From 76f5d30db1eb97bca08d8b4e03076fbe0628153e Mon Sep 17 00:00:00 2001 From: Karol Labus Date: Thu, 3 Sep 2026 11:42:09 +0200 Subject: [PATCH] Initial commit Co-authored-by: GitHub Copilot --- .env.example | 6 + .gitignore | 34 +++ Dockerfile | 14 + core/database.py | 110 +++++++ core/diagnostics.py | 48 +++ core/notifier.py | 45 +++ core/rss_generator.py | 111 +++++++ core/scrapers/jazz.py | 57 ++++ core/scrapers/radio357.py | 282 ++++++++++++++++++ core/scrapers/rns.py | 508 +++++++++++++++++++++++++++++++ core/security.py | 45 +++ docker-compose.yml | 13 + main.py | 613 ++++++++++++++++++++++++++++++++++++++ requirements.txt | 10 + static/favicon.svg | 5 + templates/base.html | 103 +++++++ templates/episodes.html | 53 ++++ templates/index.html | 233 +++++++++++++++ templates/programs.html | 67 +++++ 19 files changed, 2357 insertions(+) create mode 100644 .env.example create mode 100644 .gitignore create mode 100644 Dockerfile create mode 100644 core/database.py create mode 100644 core/diagnostics.py create mode 100644 core/notifier.py create mode 100644 core/rss_generator.py create mode 100644 core/scrapers/jazz.py create mode 100644 core/scrapers/radio357.py create mode 100644 core/scrapers/rns.py create mode 100644 core/security.py create mode 100644 docker-compose.yml create mode 100644 main.py create mode 100644 requirements.txt create mode 100644 static/favicon.svg create mode 100644 templates/base.html create mode 100644 templates/episodes.html create mode 100644 templates/index.html create mode 100644 templates/programs.html diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..5974061 --- /dev/null +++ b/.env.example @@ -0,0 +1,6 @@ +# Generate a new value for each deployment. +RADIOSYNC_SECRET_KEY= + +# Optional: Fernet key for encrypting values stored in SQLite. +# Generate with: python -c "from cryptography.fernet import Fernet; print(Fernet.generate_key().decode())" +RADIOSYNC_CONFIG_KEY= \ No newline at end of file diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..b2fbf14 --- /dev/null +++ b/.gitignore @@ -0,0 +1,34 @@ +# Secrets and local configuration +.env +.env.* +!.env.example + +# Runtime data +data/* +!data/.gitkeep +*.db +*.sqlite +*.sqlite3 + +# Python cache and local environments +__pycache__/ +*.py[cod] +*$py.class +.venv/ +venv/ +env/ + +# Test and tooling artifacts +.pytest_cache/ +.mypy_cache/ +.ruff_cache/ +.coverage +htmlcov/ + +# IDE and OS files +.vscode/* +!.vscode/extensions.json +!.vscode/settings.json +.idea/ +.DS_Store +Thumbs.db \ No newline at end of file diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..b08cb05 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,14 @@ +FROM python:3.11-slim + +WORKDIR /app + +RUN apt-get update && apt-get install -y tzdata && \ + ln -snf /usr/share/zoneinfo/Europe/Warsaw /etc/localtime && echo Europe/Warsaw > /etc/timezone && \ + apt-get clean + +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY . . + +CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/core/database.py b/core/database.py new file mode 100644 index 0000000..3f4db9f --- /dev/null +++ b/core/database.py @@ -0,0 +1,110 @@ +from sqlalchemy import create_engine, Column, String, Integer, Boolean, Float, Text, DateTime, TypeDecorator +from sqlalchemy.orm import declarative_base, sessionmaker +from .security import encrypt_config_value, decrypt_config_value + +DATABASE_URL = "sqlite:///./data/radiosync.db" + +engine = create_engine(DATABASE_URL, connect_args={"check_same_thread": False}) +SessionLocal = sessionmaker(autocommit=False, autoflush=False, bind=engine) + +Base = declarative_base() + + +class EncryptedString(TypeDecorator): + impl = String + cache_ok = True + + def process_bind_param(self, value, dialect): + return encrypt_config_value(value) + + def process_result_value(self, value, dialect): + return decrypt_config_value(value) + +class Config(Base): + __tablename__ = "config" + key = Column(String, primary_key=True, index=True) + value = Column(EncryptedString) + +class Program(Base): + __tablename__ = "programs" + id = Column(Integer, primary_key=True, index=True) + station = Column(String, index=True) # "radio357", "rns", "jazz" + slug = Column(String, index=True) + name = Column(String) + description = Column(Text, nullable=True) + image = Column(String, nullable=True) + + # Sync tracking + backfill_page = Column(Integer, default=1) + backfill_complete = Column(Boolean, default=False) + total_pages = Column(Integer, default=0) + last_catchup = Column(Float, default=0.0) + next_catchup_after = Column(Float, default=0.0) # timestamp – nie sprawdzaj przed tym czasem + +class Episode(Base): + __tablename__ = "episodes" + id = Column(Integer, primary_key=True, index=True) + station = Column(String, index=True) + program_slug = Column(String, index=True) + ep_id = Column(String, index=True) # ID from the radio API + + title = Column(String) + authors = Column(String, nullable=True) + url = Column(String, nullable=True) + image = Column(String, nullable=True) + pub_date = Column(String, nullable=True) # ISO format or YYYY-MM-DD + duration_secs = Column(Integer, default=0) + description = Column(Text, nullable=True) + + is_broken = Column(Boolean, default=False) + + +class ErrorEvent(Base): + __tablename__ = "error_events" + id = Column(Integer, primary_key=True, index=True) + station = Column(String, index=True, nullable=False) + source = Column(String, nullable=False) + message = Column(Text, nullable=False) + severity = Column(String, default="error", nullable=False) + fingerprint = Column(String, index=True, nullable=False) + occurrences = Column(Integer, default=1, nullable=False) + first_seen = Column(DateTime, nullable=False) + last_seen = Column(DateTime, nullable=False) + acknowledged = Column(Boolean, default=False, nullable=False) + acknowledged_at = Column(DateTime, nullable=True) + snapshot_path = Column(String, nullable=True) + +def init_db(): + Base.metadata.create_all(bind=engine) + import sqlite3, os + db_path = DATABASE_URL.replace("sqlite:///", "") + if os.path.exists(db_path): + with sqlite3.connect(db_path) as conn: + cursor = conn.cursor() + existing_columns = { + row[1] for row in cursor.execute("PRAGMA table_info(programs)") + } + for col, typedef in [ + ("next_catchup_after", "REAL DEFAULT 0.0"), + ("total_pages", "INTEGER DEFAULT 0"), + ]: + if col not in existing_columns: + cursor.execute(f"ALTER TABLE programs ADD COLUMN {col} {typedef}") + error_columns = {row[1] for row in cursor.execute("PRAGMA table_info(error_events)")} + if "snapshot_path" not in error_columns: + cursor.execute("ALTER TABLE error_events ADD COLUMN snapshot_path TEXT") + + # Encrypt legacy plaintext values using the raw SQLite connection; + # assigning the same decrypted value through the ORM is not dirty. + rows = cursor.execute("SELECT key, value FROM config").fetchall() + for key, value in rows: + if value is not None and not value.startswith("enc:v1:"): + cursor.execute("UPDATE config SET value = ? WHERE key = ?", + (encrypt_config_value(value), key)) + +def get_db(): + db = SessionLocal() + try: + yield db + finally: + db.close() diff --git a/core/diagnostics.py b/core/diagnostics.py new file mode 100644 index 0000000..4c900fb --- /dev/null +++ b/core/diagnostics.py @@ -0,0 +1,48 @@ +import json +import os +import re +import hashlib +from datetime import datetime + + +class ScraperError(Exception): + def __init__(self, station, code, message, url=None, status_code=None): + self.station = station + self.code = code + self.url = url + self.status_code = status_code + details = f"[{code}] {message}" + if status_code is not None: + details += f" (HTTP {status_code})" + if url: + details += f" URL: {url}" + super().__init__(details) + + +def save_response_snapshot(station, code, url, content, status_code=None): + directory = os.path.join("data", "diagnostics") + try: + os.makedirs(directory, exist_ok=True) + digest = hashlib.sha256(f"{station}:{code}:{url}".encode("utf-8")).hexdigest()[:16] + safe_station = re.sub(r"[^a-zA-Z0-9_-]", "_", station) + base_path = os.path.join(directory, f"{safe_station}-{code}-{digest}") + + with open(f"{base_path}.json", "w", encoding="utf-8") as metadata_file: + json.dump({"station": station, "code": code, "url": url, "status_code": status_code, + "updated_at": datetime.now().isoformat()}, metadata_file, ensure_ascii=False, indent=2) + with open(f"{base_path}.html", "w", encoding="utf-8") as response_file: + response_file.write((content or "")[:2_000_000]) + return base_path + except OSError as exc: + return f"snapshot unavailable: {exc}" + + +def invalid_response(station, code, message, url, response=None, content=None): + snapshot = save_response_snapshot( + station, + url, + response.text if response is not None else content, + response.status_code if response is not None else None, + ) + raise ScraperError(station, code, f"{message}. Snapshot: {snapshot}", url, + response.status_code if response is not None else None) \ No newline at end of file diff --git a/core/notifier.py b/core/notifier.py new file mode 100644 index 0000000..ee992b0 --- /dev/null +++ b/core/notifier.py @@ -0,0 +1,45 @@ +import ipaddress +import socket +from urllib.parse import urlparse +import requests + + +def validate_notification_url(url: str): + parsed = urlparse(url or "") + if parsed.scheme not in {"http", "https"} or not parsed.hostname or parsed.username or parsed.password: + raise ValueError("Adres ntfy musi być adresem HTTP(S) bez danych logowania") + try: + addresses = {info[4][0] for info in socket.getaddrinfo(parsed.hostname, parsed.port, type=socket.SOCK_STREAM)} + if any(ipaddress.ip_address(address).is_private or ipaddress.ip_address(address).is_loopback or + ipaddress.ip_address(address).is_link_local for address in addresses): + raise ValueError("Adres ntfy wskazuje na prywatną lub lokalną sieć") + except socket.gaierror as exc: + raise ValueError("Nie można rozpoznać hosta ntfy") from exc + +def send_notification(url: str, topic: str, title: str, message: str, username: str = None, password: str = None): + if not url or not topic: + return False + validate_notification_url(url) + + full_url = f"{url.rstrip('/')}/{topic}" + headers = { + "Title": title.encode('utf-8') + } + + auth = None + if username and password: + auth = (username, password) + + try: + response = requests.post( + full_url, + data=message.encode('utf-8'), + headers=headers, + auth=auth, + timeout=5 + ) + response.raise_for_status() + return True + except Exception as e: + print(f"Failed to send ntfy notification: {e}") + return False diff --git a/core/rss_generator.py b/core/rss_generator.py new file mode 100644 index 0000000..8b2c0d5 --- /dev/null +++ b/core/rss_generator.py @@ -0,0 +1,111 @@ +from sqlalchemy.orm import Session +from xml.etree import ElementTree as ET +from xml.sax.saxutils import escape +from email.utils import format_datetime +from datetime import datetime +from .database import Program, Episode + +def get_base_url(request): + # Retrieve base URL for constructing absolute paths (from FastAPI Request) + return str(request.base_url).rstrip('/') + +def generate_master_opml(db: Session, base_url: str, station: str = None): + opml_outlines = [] + + # Radio 357 + r357 = db.query(Program).filter_by(station="radio357").order_by(Program.name).all() if not station or station == "radio357" else [] + if r357: + opml_outlines.append(' ') + for prog in r357: + title = escape(prog.name, {'"': """, "'": "'"}) + xmlUrl = f"{base_url}/feeds/radio357/{prog.slug}.xml" + opml_outlines.append(f' ') + opml_outlines.append(' ') + + # RNŚ + rns = db.query(Program).filter_by(station="rns").order_by(Program.name).all() if not station or station == "rns" else [] + if rns: + opml_outlines.append(' ') + for prog in rns: + title = escape(prog.name, {'"': """, "'": "'"}) + xmlUrl = f"{base_url}/feeds/rns/{prog.slug}.xml" + opml_outlines.append(f' ') + opml_outlines.append(' ') + + # Radio Jazz FM (Direct to source feeds!) + jazz = db.query(Program).filter_by(station="jazz").order_by(Program.name).all() if not station or station == "jazz" else [] + if jazz: + opml_outlines.append(' ') + for prog in jazz: + title = escape(prog.name, {'"': """, "'": "'"}) + # Link directly to their server! + xmlUrl = f"https://podkasty.radiojazz.fm/@{prog.slug}/feed.xml" + opml_outlines.append(f' ') + opml_outlines.append(' ') + + xml_content = '\n' + xml_content += '\n \n' + xml_content += '\n'.join(opml_outlines) + '\n' + xml_content += ' \n' + return xml_content + +def generate_podcast_rss(db: Session, station: str, slug: str): + prog = db.query(Program).filter_by(station=station, slug=slug).first() + if not prog: + return None + + eps = db.query(Episode).filter_by(station=station, program_slug=slug, is_broken=False).filter(Episode.url.isnot(None)).order_by(Episode.pub_date.desc()).all() + + rss = ET.Element("rss", {"version": "2.0", "xmlns:itunes": "http://www.itunes.com/dtds/podcast-1.0.dtd"}) + channel = ET.SubElement(rss, "channel") + + ET.SubElement(channel, "title").text = prog.name + ET.SubElement(channel, "description").text = prog.description or "" + + if prog.image: + ET.SubElement(channel, "itunes:image", {"href": prog.image}) + image_el = ET.SubElement(channel, "image") + ET.SubElement(image_el, "url").text = prog.image + ET.SubElement(image_el, "title").text = prog.name + + for ep in eps: + item = ET.SubElement(channel, "item") + + date_prefix = f"({ep.pub_date[:10]}) " if ep.pub_date else "" + ET.SubElement(item, "title").text = f"{date_prefix}{ep.title}" + ET.SubElement(item, "itunes:author").text = ep.authors or "" + + if ep.description: + ET.SubElement(item, "description").text = ep.description + ET.SubElement(item, "itunes:summary").text = ep.description + + if ep.pub_date: + try: + # radio357 format: 2024-03-22T08:00:00+00:00 or rns: 2024-03-22 + if 'T' in ep.pub_date: + dt = datetime.fromisoformat(ep.pub_date) + else: + dt = datetime.strptime(ep.pub_date, "%Y-%m-%d") + ET.SubElement(item, "pubDate").text = format_datetime(dt) + except: + ET.SubElement(item, "pubDate").text = ep.pub_date + + if ep.duration_secs: + ET.SubElement(item, "itunes:duration").text = str(ep.duration_secs) + + ET.SubElement(item, "guid", {"isPermaLink": "false"}).text = f"{station}-{ep.ep_id}" + ET.SubElement(item, "enclosure", {"url": ep.url, "type": "audio/mpeg", "length": "0"}) + + if ep.image or prog.image: + ET.SubElement(item, "itunes:image", {"href": ep.image or prog.image}) + + tree = ET.ElementTree(rss) + try: + ET.indent(tree, space=" ", level=0) + except AttributeError: + pass # older python versions + + from io import BytesIO + f = BytesIO() + tree.write(f, encoding="utf-8", xml_declaration=True) + return f.getvalue() diff --git a/core/scrapers/jazz.py b/core/scrapers/jazz.py new file mode 100644 index 0000000..f9f1b0a --- /dev/null +++ b/core/scrapers/jazz.py @@ -0,0 +1,57 @@ +import requests +from bs4 import BeautifulSoup +from sqlalchemy.orm import Session +from ..database import Program + +BASE_URL = "https://podkasty.radiojazz.fm" + +class JazzScraper: + def __init__(self, db: Session, logger=print): + self.db = db + self.logger = logger + + def sync_programs(self): + self.logger("JAZZ: Pobieranie audycji...") + try: + r = requests.get(BASE_URL, headers={"User-Agent": "Mozilla/5.0"}, timeout=15) + r.raise_for_status() + html = r.text + except Exception as e: + self.logger(f"JAZZ: Błąd pobierania bazy: {e}") + return + + soup = BeautifulSoup(html, 'html.parser') + links = soup.find_all('a', href=True) + + found = 0 + for link in links: + href = link['href'] + if '/@' in href: + slug = href.split('/@')[-1].split('/')[0] + if not slug: continue + + text = link.get_text(strip=True) + if not text or text.startswith('@') or "Recent activity" in text: + continue + + clean_title = text + if clean_title.endswith(f"@{slug}"): + clean_title = clean_title[:-len(f"@{slug}")].strip() + if clean_title.startswith("Ż "): + clean_title = clean_title[2:].strip() + elif clean_title.startswith("Ż"): + clean_title = clean_title[1:].strip() + + prog = self.db.query(Program).filter_by(station="jazz", slug=slug).first() + if not prog: + self.db.add(Program( + station="jazz", slug=slug, name=clean_title, + description="Radio Jazz FM", image="" + )) + found += 1 + + self.db.commit() + self.logger(f"JAZZ: Znaleziono {found} nowych audycji (łącznie zaktualizowano).") + + def run_full_sync(self): + self.sync_programs() diff --git a/core/scrapers/radio357.py b/core/scrapers/radio357.py new file mode 100644 index 0000000..4ecfbdb --- /dev/null +++ b/core/scrapers/radio357.py @@ -0,0 +1,282 @@ +import requests +import json +import base64 +import time +import os +from sqlalchemy.orm import Session +from ..database import Program, Episode, Config +from ..diagnostics import ScraperError, invalid_response + + +def decode_jwt_exp(token): + try: + parts = token.split('.') + if len(parts) != 3: return 0 + payload_b64 = parts[1] + payload_b64 += "=" * ((4 - len(payload_b64) % 4) % 4) + return json.loads(base64.b64decode(payload_b64).decode('utf-8')).get("exp", 0) + except: + return 0 + +def get_audio_length(url): + try: + r = requests.head(url, timeout=5) + return int(r.headers.get('Content-Length', 0)) + except: + return 0 + +class Radio357Scraper: + def __init__(self, db: Session, logger=print, progress_callback=None, stop_flag=None): + self.db = db + self.logger = logger + self.progress_callback = progress_callback + self.stop_flag = stop_flag + self.requests_made = 0 + limit_conf = self.db.query(Config).filter_by(key="r357_limit").first() + self.backfill_limit = int(limit_conf.value) if limit_conf and limit_conf.value.isdigit() else 600 + hard_limit_conf = self.db.query(Config).filter_by(key="r357_hard_limit").first() + self.hard_limit = int(hard_limit_conf.value) if hard_limit_conf and hard_limit_conf.value.isdigit() else 1000 + self.start_time = time.time() + self.max_execution_time = 900 # 15 minut max na cały sync + if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit) + + def check_timeout(self): + if self.stop_flag and self.stop_flag(): + raise Exception("Zatrzymano na żądanie użytkownika.") + if time.time() - self.start_time > self.max_execution_time: + raise Exception(f"Przekroczono limit czasu wykonywania skryptu ({self.max_execution_time // 60} min). Zatrzymano awaryjnie.") + + def _get(self, url, headers=None): + self.check_timeout() + if self.requests_made >= self.hard_limit: + self.logger(f"R357: Limit {self.hard_limit} zapytań osiągnięty. Przerywam.") + return None + self.requests_made += 1 + if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit) + if not headers: + headers = {} + headers["User-Agent"] = "Mozilla/5.0 (Windows NT 10.0; Win64; x64)" + try: + r = requests.get(url, headers=headers, timeout=10) + if r.status_code == 200: + try: + data = r.json() + except ValueError: + invalid_response("radio357", "INVALID_JSON", "API zwróciło nieprawidłowy JSON", url, r) + embedded = data.get("_embedded") if isinstance(data, dict) else None + if not isinstance(embedded, dict) or not isinstance(embedded.get("podcasts"), list): + invalid_response("radio357", "API_SCHEMA_CHANGED", "Brak oczekiwanej listy podcastów w odpowiedzi API", url, r) + return data + else: + invalid_response("radio357", "HTTP_ERROR", "Błąd pobierania danych API", url, r) + except Exception as e: + if isinstance(e, ScraperError): + raise + raise ScraperError("radio357", "NETWORK_ERROR", f"Błąd zapytania: {e}", url) from e + return None + + def get_token(self): + token_conf = self.db.query(Config).filter_by(key="r357_token").first() + if token_conf and decode_jwt_exp(token_conf.value) > time.time() + 3600: + return token_conf.value + + email = self.db.query(Config).filter_by(key="r357_email").first() + password = self.db.query(Config).filter_by(key="r357_password").first() + + if not email or not password: + self.logger("R357: Brak danych logowania w bazie.") + return None + + self.logger("R357: Pobieranie nowego tokenu...") + url = "https://auth.r357.eu/api/auth/login" + payload = {"email": email.value, "password": password.value} + headers = { + "Content-Type": "application/json", + "Origin": "https://konto.radio357.pl", + "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36" + } + try: + r = requests.post(url, json=payload, headers=headers, timeout=10) + if r.status_code == 200: + new_token = r.json().get("accessToken") + if new_token: + if not token_conf: + token_conf = Config(key="r357_token", value=new_token) + self.db.add(token_conf) + else: + token_conf.value = new_token + self.db.commit() + return new_token + else: + self.logger(f"R357 Błąd logowania: {r.status_code} {r.text}") + except Exception as e: + self.logger(f"R357 Wyjątek logowania: {e}") + return None + + def check_auth_status(self): + token_conf = self.db.query(Config).filter_by(key="r357_token").first() + if not token_conf: + return False, "Brak zapisanego tokenu." + exp = decode_jwt_exp(token_conf.value) + if exp < time.time(): + return False, "Token wygasł." + return True, "Token aktywny." + + def sync_episodes(self, specific_program_slug=None): + self.logger("R357: Rozpoczynam synchronizację...") + page = 0 + new_found = 0 + + is_first_run = self.db.query(Episode).filter_by(station="radio357").count() == 0 + # Przy pierwszym uruchomieniu indeks metadanych musi przejść przez całe API. + # W kolejnych uruchomieniach zostawiamy budżet na backfill URL-i audio. + catchup_limit = self.hard_limit if is_first_run else max(1, self.hard_limit - self.backfill_limit) + seen_slugs = set() # Guard przeciw duplikatom w tej sesji + + while True: + if self.requests_made >= catchup_limit: + self.logger(f"R357: [{self.requests_made}/{self.hard_limit}] Limit Catchup osiągnięty. Zostawiam resztę dla Backfill.") + break + + url = f"https://static.radio357.pl/api/content/v1/podcasts?page={page}" + self.logger(f"R357: [{self.requests_made}/{self.hard_limit}] Pobieram stronę {page}...") + data = self._get(url) + + if not data: + raise Exception(f"Błąd sieci/API podczas pobierania strony {page} (Radio 357)") + + episodes = data.get("_embedded", {}).get("podcasts", []) + if not episodes: + break + + all_known = True + page_has_target = False + for ep in episodes: + ep_id = str(ep["id"]) + + # Extract program info + programs_list = ep.get("programs") + if not programs_list: continue + program = programs_list[0] + slug = program.get("slug", f"program_{program['id']}") + + if specific_program_slug and slug != specific_program_slug: + continue # Skip if we are targeting a specific program + + page_has_target = True + + existing = self.db.query(Episode).filter_by(station="radio357", ep_id=ep_id).first() + if not existing: + all_known = False + + # Ensure program exists (cache guard against duplicates) + if slug not in seen_slugs: + db_prog = self.db.query(Program).filter_by(station="radio357", slug=slug).first() + if not db_prog: + db_prog = Program( + station="radio357", + slug=slug, + name=program.get("name", f"Podcast {program['id']}"), + image=program.get("image", ""), + description=program.get("description", "") + ) + self.db.add(db_prog) + self.db.flush() # Natychmiastowy flush żeby kolejne odcinki widziały program + seen_slugs.add(slug) + + team = ep.get("team", []) + authors = ", ".join([t.get("name", "") for t in team]) or "Radio 357" + + new_ep = Episode( + station="radio357", + program_slug=slug, + ep_id=ep_id, + title=ep.get("subTitle") or ep.get("title", ""), + duration_secs=ep.get("duration", 0), + image=ep.get("image", ""), + description=ep.get("description", ""), + pub_date=ep.get("published", ""), + authors=authors, + url=None, + is_broken=False + ) + self.db.add(new_ep) + new_found += 1 + + self.db.commit() + + if not is_first_run and all_known and not specific_program_slug: + self.logger(f"R357: Wszystkie odcinki na stronie {page} są znane. Zakończono catchup.") + break + + if specific_program_slug and page_has_target and all_known: + self.logger(f"R357: Wszystkie odcinki programu {specific_program_slug} na stronie {page} są znane. Zakończono catchup programu.") + break + + if len(episodes) < data.get("page_size", 250): + break + + page += 1 + + self.logger(f"R357: Znaleziono {new_found} nowych odcinków w bazie.") + + def backfill_urls(self, specific_program_slug=None): + backfill_count = 0 + self.logger(f"R357: Rozpoczynam pobieranie linków audio (limit {self.backfill_limit})...") + + query = self.db.query(Episode).filter_by(station="radio357", url=None, is_broken=False).order_by(Episode.pub_date.desc()) + if specific_program_slug: + query = query.filter_by(program_slug=specific_program_slug) + + missing_eps = query.all() + token = self.get_token() + if not token: + self.logger("R357: Brak tokena, przerywam backfill url") + return + + fetched = 0 + for ep in missing_eps: + self.check_timeout() + if self.requests_made >= self.hard_limit or backfill_count >= self.backfill_limit: + break + + self.logger(f"R357: Pobieram URL dla {ep.title[:30]}...") + gateway_url = f"https://gateway.r357.eu/api/content/podcast/{ep.ep_id}/url" + + self.requests_made += 1 + if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit) + backfill_count += 1 + try: + r = requests.get(gateway_url, headers={ + "Authorization": f"Bearer {token}", + "Origin": "https://radio357.pl", + "User-Agent": "Mozilla/5.0" + }, timeout=10) + + if r.status_code == 200: + audio_url = r.json().get("url") + if audio_url: + ep.url = audio_url + fetched += 1 + elif r.status_code == 401 or r.status_code == 403: + self.logger("R357: Token odrzucony przez gateway. Wymagam reloginu.") + # Usuwamy token, zeby przy nastepnym runie pobral nowy + self.db.query(Config).filter_by(key="r357_token").delete() + self.db.commit() + raise Exception("R357 Token Expired") + else: + self.logger(f"R357 Błąd HTTP {r.status_code} podczas pobierania audio: {r.text[:100]}") + except Exception as e: + self.logger(f"R357: Błąd URL dla odcinka {ep.ep_id}: {e}") + if "Token Expired" in str(e): + raise + + time.sleep(0.5) + + self.db.commit() + self.logger(f"R357: Pomyślnie pobrano {fetched} nowych linków.") + + def run_full_sync(self, specific_program_slug=None): + self.sync_episodes(specific_program_slug) + self.backfill_urls(specific_program_slug) + self.logger(f"R357: Koniec. Wysłano zapytania: {self.requests_made}/{self.hard_limit} (w tym backfill ograniczony do {self.backfill_limit}).") diff --git a/core/scrapers/rns.py b/core/scrapers/rns.py new file mode 100644 index 0000000..579db1b --- /dev/null +++ b/core/scrapers/rns.py @@ -0,0 +1,508 @@ +import requests +import time +import os +import re +import urllib.parse +from bs4 import BeautifulSoup +from sqlalchemy.orm import Session +from ..database import Program, Episode, Config +from ..diagnostics import ScraperError, invalid_response + +MAX_REQUESTS_LIMIT = 50 +BASE_URL = "https://nowyswiat.online" + + +def get_cookie_for_url(cookie_jar, name, url): + """Wybiera najbardziej szczegółowe cookie, gdy jar zawiera duplikaty nazwy.""" + parsed_url = urllib.parse.urlparse(url) + host = parsed_url.hostname.lower() + path = parsed_url.path or "/" + candidates = [] + for cookie in cookie_jar: + if cookie.name != name or (cookie.secure and parsed_url.scheme != "https"): + continue + domain = cookie.domain.lstrip(".").lower() + if domain and not (host == domain or host.endswith(f".{domain}")): + continue + cookie_path = cookie.path or "/" + if not path.startswith(cookie_path.rstrip("/") or "/"): + continue + candidates.append(cookie) + + if not candidates: + return None + selected = max(candidates, key=lambda cookie: (len(cookie.path or "/"), len(cookie.domain or ""))) + for cookie in candidates: + if cookie is not selected: + cookie_jar.clear(cookie.domain, cookie.path, cookie.name) + return selected.value + +def parse_polish_date(date_str): + months = { + "stycznia": "01", "lutego": "02", "marca": "03", "kwietnia": "04", + "maja": "05", "czerwca": "06", "lipca": "07", "sierpnia": "08", + "września": "09", "października": "10", "listopada": "11", "grudnia": "12", + "styczeń": "01", "luty": "02", "marzec": "03", "kwiecień": "04", + "maj": "05", "czerwiec": "06", "lipiec": "07", "sierpień": "08", + "wrzesień": "09", "październik": "10", "listopad": "11", "grudzień": "12" + } + parts = date_str.lower().split() + if len(parts) == 3: + day = parts[0].zfill(2) + month = months.get(parts[1], "01") + year = parts[2] + return f"{year}-{month}-{day}" + return date_str + +class RNScraper: + def __init__(self, db: Session, logger=print, progress_callback=None, warning_callback=None, stop_flag=None): + self.db = db + self.logger = logger + self.progress_callback = progress_callback + self.warning_callback = warning_callback + self.stop_flag = stop_flag + self.requests_made = 0 + self.start_time = time.time() + self.max_execution_time = 900 # 15 minut max na cały sync + limit_conf = self.db.query(Config).filter_by(key="rns_limit").first() + self.backfill_limit = int(limit_conf.value) if limit_conf and limit_conf.value.isdigit() else 50 + hard_limit_conf = self.db.query(Config).filter_by(key="rns_hard_limit").first() + self.hard_limit = int(hard_limit_conf.value) if hard_limit_conf and hard_limit_conf.value.isdigit() else 500 + self.session = requests.Session() + self.session.headers.update({"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}) + if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit) + + def check_timeout(self): + if self.stop_flag and self.stop_flag(): + raise Exception("Zatrzymano na żądanie użytkownika.") + if time.time() - self.start_time > self.max_execution_time: + raise Exception(f"Przekroczono limit czasu wykonywania skryptu ({self.max_execution_time // 60} min). Zatrzymano awaryjnie.") + + def _get_html(self, url): + self.check_timeout() + if self.requests_made >= self.hard_limit: + return None + for attempt in range(1, 4): + if self.requests_made >= self.hard_limit: + return None + self.requests_made += 1 + if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit) + try: + r = self.session.get(url, timeout=15) + if r.status_code == 200: + return r.text + invalid_response("rns", "HTTP_ERROR", "Błąd pobierania strony", url, r) + except requests.exceptions.Timeout as e: + if attempt == 3: + raise ScraperError("rns", "NETWORK_TIMEOUT", "Trzy próby pobrania strony zakończyły się timeoutem", url) from e + self.logger(f"RNŚ: Timeout dla {url}, próba {attempt}/3. Ponawiam za 30 sekund.") + self.check_timeout() + time.sleep(30) + except Exception as e: + if isinstance(e, ScraperError): + raise + raise ScraperError("rns", "NETWORK_ERROR", f"Błąd zapytania: {e}", url) from e + return None + + def perform_login(self): + email = self.db.query(Config).filter_by(key="rns_email").first() + password = self.db.query(Config).filter_by(key="rns_password").first() + if not email or not password: + self.logger("RNŚ: Brak danych logowania w bazie.") + return False + + self.logger("RNŚ: Inicjalizacja logowania (pobieranie CSRF)...") + try: + r1 = self.session.get("https://nowyswiat.online/konto/zaloguj", timeout=15) + except Exception as e: + self.logger(f"RNŚ: Sieć zablokowała pobieranie CSRF: {e}") + return False + + csrf_token = get_cookie_for_url( + self.session.cookies, + "csrf_cookie_neocms", + "https://nowyswiat.online/konto/zaloguj", + ) + if not csrf_token: + self.logger("RNŚ: Nie udało się pobrać tokenu CSRF.") + return False + + self.logger("RNŚ: Wysyłanie formularza...") + payload = {"csrf_neocms": csrf_token, "login": email.value, "password": password.value, "ufd_data": "{}"} + try: + r2 = self.session.post("https://nowyswiat.online/konto/zaloguj", data=payload, headers={ + "X-Requested-With": "XMLHttpRequest" + }, timeout=15) + except Exception as e: + self.logger(f"RNŚ: Błąd sieci przy wysyłaniu formularza: {e}") + return False + + # Check if login succeeded by looking for a session cookie or a success response + if '"status":"OK"' in r2.text or "logowanie udane" in r2.text.lower(): + # Save cookies to DB (serialize) + cookies_dict = requests.utils.dict_from_cookiejar(self.session.cookies) + import json + cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first() + if not cookie_conf: + self.db.add(Config(key="rns_cookies", value=json.dumps(cookies_dict))) + else: + cookie_conf.value = json.dumps(cookies_dict) + self.db.commit() + return True + else: + self.logger(f"RNŚ: Błędne dane logowania (lub zmiana mechanizmu). Odpowiedź: {r2.text[:100]}") + return False + + def ensure_auth(self): + cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first() + if cookie_conf: + import json + try: + cookies_dict = json.loads(cookie_conf.value) + self.session.cookies = requests.utils.cookiejar_from_dict(cookies_dict) + html = self._get_html(f"{BASE_URL}/") + if html and "wyloguj" in html.lower(): + return True + self.logger("RNŚ: Zapisana sesja wygasła, wykonuję ponowne logowanie.") + except: + pass + return self.perform_login() + + def check_auth_status(self): + cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first() + if not cookie_conf: + return False, "Brak zapisanych ciasteczek." + import json + try: + self.session.cookies = requests.utils.cookiejar_from_dict(json.loads(cookie_conf.value)) + html = self._get_html("https://nowyswiat.online/") + if html and "wyloguj" in html.lower(): + return True, "Ciasteczka aktywne, sesja poprawna." + return False, "Ciasteczka nieaktywne lub wygasły (brak dostępu do profilu)." + except Exception as e: + return False, f"Błąd: {e}" + + def update_programs(self): + self.logger("RNŚ: Aktualizacja listy programów...") + html = self._get_html("https://nowyswiat.online/podcasty") + if not html: return + soup = BeautifulSoup(html, "html.parser") + links = soup.find_all("a", href=True) + program_links = [link for link in links if "rbroadcast=" in link["href"]] + if not program_links: + invalid_response("rns", "HTML_SCHEMA_CHANGED", "Nie znaleziono programów w stronie podcastów", "https://nowyswiat.online/podcasty", content=html) + for link in program_links: + href = link["href"] + if "rbroadcast=" in href: + parsed = urllib.parse.urlparse(href) + slug = urllib.parse.parse_qs(parsed.query).get("rbroadcast", [None])[0] + if not slug: continue + + title_el = link.find("h2", class_="rns-search-dropdown-title") + title = title_el.text.strip() if title_el else slug + + img_el = link.find("img") + img_url = img_el["src"] if img_el and img_el.has_attr("src") else "" + if img_url and not img_url.startswith("http"): img_url = f"{BASE_URL}/{img_url.lstrip('/')}" + + prog = self.db.query(Program).filter_by(station="rns", slug=slug).first() + if not prog: + self.db.add(Program( + station="rns", slug=slug, name=title, image=img_url, + description=f"Radio Nowy Świat: {title}", backfill_page=2, backfill_complete=False + )) + self.db.commit() + + def _verify_audio_teaser(self, ep: Episode): + """Wykrywa 1-minutowy teaser (rozmiar mniejszy niż ~2MB, mimo że audycja trwa > 5 min).""" + self.check_timeout() + if not ep.url: return False + + if self.requests_made >= self.hard_limit: return False + self.requests_made += 1 + if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit) + + try: + r = self.session.head(ep.url, timeout=5) + cl = int(r.headers.get("Content-Length", 0)) + if cl > 0 and cl < 2_500_000 and ep.duration_secs > 300: + self.logger(f"RNŚ: Wykryto uszkodzony link (Teaser 1-min) dla {ep.title}. Usuwam URL.") + ep.url = None + ep.is_broken = True + return True + except Exception as e: + pass + return False + + def fetch_program_page(self, program_slug, page): + url = f"https://nowyswiat.online/podcasty?rbroadcast={program_slug}&page={page}" + html = self._get_html(url) + if not html: + raise Exception(f"Błąd sieci podczas pobierania strony {page} audycji {program_slug}") + + soup = BeautifulSoup(html, "html.parser") + cards = soup.find_all("a", class_="rns-grid-podcast-card") + known_episode_count = self.db.query(Episode).filter_by( + station="rns", program_slug=program_slug + ).count() + page_has_podcast_content = bool(soup.find(string=re.compile(r"podcast|podkast", re.IGNORECASE))) + if not cards and page > 1: + return 0, 0, page - 1 + if not cards and (known_episode_count or not page_has_podcast_content): + invalid_response("rns", "HTML_SCHEMA_CHANGED", "Nie znaleziono kart odcinków dla istniejącego programu", url, content=html) + + new_found = 0 + missing_player_count = 0 + valid_cards = 0 + + for card in cards: + href = card.get("href", "") + if not href or "/podcasty/" not in href: continue + valid_cards += 1 + ep_id = href.split("/podcasty/")[-1].split("?")[0] + + player_box = card.find("div", class_="rns-play-btn-box") + audio_url = player_box.get("data-neo-player-src") if player_box else None + + if not player_box: + missing_player_count += 1 + + if audio_url and not audio_url.startswith("http"): + audio_url = f"{BASE_URL}/{audio_url.lstrip('/')}" + + title_el = card.find("p", class_="rns-post-title") + raw_title = player_box.get("data-neo-player-title", "") if player_box else (title_el.text.strip() if title_el else ep_id) + title = BeautifulSoup(raw_title, "html.parser").text.strip() if raw_title else ep_id + + raw_subtitle = player_box.get("data-neo-player-subtitle", "") if player_box else "" + authors = BeautifulSoup(raw_subtitle, "html.parser").text.strip() if raw_subtitle else "Radio Nowy Świat" + + date_el = card.find("p", class_="rns-podcast-details-date") + pub_date = parse_polish_date(date_el.text.strip()) if date_el else "" + + time_el = card.find("p", class_="rns-podcast-details-long") + duration_str = time_el.text.strip() if time_el else "0:00" + duration_secs = 0 + if ":" in duration_str: + parts = duration_str.split(":") + if len(parts) == 3: duration_secs = int(parts[0])*3600 + int(parts[1])*60 + int(parts[2]) + elif len(parts) == 2: duration_secs = int(parts[0])*60 + int(parts[1]) + + img_url = player_box.get("data-neo-player-img", "") if player_box else "" + if img_url and not img_url.startswith("http"): img_url = f"{BASE_URL}/{img_url.lstrip('/')}" + + desc_el = card.find("p", class_="rns-post-card-desc") + description = desc_el.get_text(separator="\n").strip() if desc_el else "" + + ep = self.db.query(Episode).filter_by(station="rns", ep_id=ep_id).first() + if not ep: + ep = Episode( + station="rns", program_slug=program_slug, ep_id=ep_id, + title=title, authors=authors, url=audio_url, image=img_url, + pub_date=pub_date, duration_secs=duration_secs, description=description, + is_broken=False + ) + self.db.add(ep) + new_found += 1 + else: + if not ep.url and audio_url: + ep.url = audio_url + ep.is_broken = False + # Celowo nie zwiększamy new_found, aby Daily Catchup nie wchodził w tryb głębokiego archiwum. + # Łataniem starych dziur zajmie się faza Backfill. + + if cards and not valid_cards: + invalid_response("rns", "HTML_SCHEMA_CHANGED", "Znaleziono karty podcastów bez oczekiwanych identyfikatorów", url, content=html) + + if missing_player_count: + warning = f"{missing_player_count} odcinków bez dostępnego playera audio; pomijam ich URL-e." + if self.warning_callback: + self.warning_callback(warning) + else: + self.logger(f"RNŚ: WARNING: {warning}") + + self.db.commit() + + page_links = soup.find_all("a", href=re.compile(r"page=\d+")) + max_page = page + for link in page_links: + href = link.get("href") + m = re.search(r"page=(\d+)", href) + if m: + max_page = max(max_page, int(m.group(1))) + + return new_found, len(cards), max_page + + def _compute_next_catchup(self, prog: Program) -> float: + """Wylicza kiedy najwcześniej warto znowu sprawdzać tę audycję.""" + episodes = self.db.query(Episode).filter_by( + station="rns", program_slug=prog.slug + ).order_by(Episode.pub_date.desc()).limit(20).all() + + if len(episodes) < 3: + # Za mało danych – sprawdzamy przy każdym uruchomieniu + return 0.0 + + # Wylicz średnią przerwę między odcinkami w dniach + import re + dates = [] + for ep in episodes: + if ep.pub_date and re.match(r'\d{4}-\d{2}-\d{2}', ep.pub_date): + try: + import datetime + dates.append(datetime.date.fromisoformat(ep.pub_date[:10])) + except ValueError: + pass + + if len(dates) < 3: + return 0.0 + + dates.sort(reverse=True) + gaps = [(dates[i] - dates[i+1]).days for i in range(len(dates)-1)] + avg_gap = sum(gaps) / len(gaps) + + if avg_gap < 2: + # Codziennie lub częściej → zawsze sprawdzamy (brak cooldownu) + return 0.0 + elif avg_gap <= 8: + # Tygodniowo → cooldown = 70% cyklu (żeby sprawdzić przed następnym odcinkiem) + cooldown_days = avg_gap * 0.7 + else: + # Rzadziej niż tygodniowo → max 7 dni + cooldown_days = 7.0 + + return time.time() + cooldown_days * 86400 + + def phase_1_catchup(self, specific_program_slug=None): + self.logger("RNŚ: Phase 1 (Daily Catchup)...") + query = self.db.query(Program).filter_by(station="rns") + if specific_program_slug: + query = query.filter_by(slug=specific_program_slug) + else: + query = query.order_by(Program.last_catchup.asc()) + + catchup_limit = max(1, self.hard_limit - self.backfill_limit) + skipped = 0 + + timeout_program = None + for prog in query.all(): + if self.requests_made >= catchup_limit: + self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Limit Catchup osiągnięty. Zostawiam resztę dla Backfill.") + break + + # Pomiń jeśli za wcześnie (adaptive cooldown) + if not specific_program_slug and prog.next_catchup_after and time.time() < prog.next_catchup_after: + skipped += 1 + continue + + is_first_sync = (prog.last_catchup == 0.0) + + page = 1 + catchup_succeeded = False + while True: + if self.requests_made >= catchup_limit: break + self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Sprawdzam {prog.slug} (strona {page})") + try: + new_found, total_cards, max_page = self.fetch_program_page(prog.slug, page) + except ScraperError as exc: + if exc.code != "NETWORK_TIMEOUT": + raise + if timeout_program: + raise ScraperError( + "rns", "NETWORK_TIMEOUT", + f"Timeout także dla kolejnej audycji {prog.slug}; poprzednia: {timeout_program}", + exc.url, + ) from exc + timeout_program = prog.slug + self.logger(f"RNŚ: Pomijam {prog.slug} po trzech timeoutach i sprawdzam następną audycję.") + break + + if timeout_program: + warning = f"Poprzednia audycja {timeout_program} miała trzy timeouty; kolejna audycja odpowiada poprawnie." + if self.warning_callback: + self.warning_callback(warning) + else: + self.logger(f"RNŚ: WARNING: {warning}") + timeout_program = None + + if max_page > prog.total_pages: + prog.total_pages = max_page + + if total_cards == 0 or new_found == 0: + catchup_succeeded = True + break + + if is_first_sync: + catchup_succeeded = True + break + + if page >= max_page: + catchup_succeeded = True + break + + page += 1 + time.sleep(0.5) + + if catchup_succeeded: + prog.last_catchup = time.time() + prog.next_catchup_after = self._compute_next_catchup(prog) + self.db.commit() + + if skipped: + self.logger(f"RNŚ: Pominięto {skipped} audycji (cooldown adaptacyjny).") + + def phase_2_backfill(self, specific_program_slug=None): + backfill_count = 0 + if specific_program_slug: + prog = self.db.query(Program).filter_by(station="rns", slug=specific_program_slug).first() + if prog: + page = prog.backfill_page + self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill manualny dla {prog.slug} (od strony {page})") + while self.requests_made < self.hard_limit and backfill_count < self.backfill_limit: + new_found, total, max_p = self.fetch_program_page(prog.slug, page) + backfill_count += 1 + if max_p > prog.total_pages: + prog.total_pages = max_p + self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill {prog.slug} strona {page} z {prog.total_pages or '?'} ({new_found} nowych)") + if total == 0 or page >= max_p: + prog.backfill_complete = True + self.logger(f"RNŚ: Backfill {prog.slug} – zakończono.") + break + prog.backfill_page = page + 1 + page += 1 + self.db.commit() + time.sleep(0.5) + return + + while self.requests_made < self.hard_limit and backfill_count < self.backfill_limit: + prog = self.db.query(Program).filter_by(station="rns", backfill_complete=False).order_by(Program.backfill_page.asc()).first() + if not prog: + self.logger("RNŚ: Wszystkie audycje w pełni uzupełnione!") + break + + page = prog.backfill_page + self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill dla {prog.slug} (strona {page} z {prog.total_pages or '?'})") + new_found, total_cards, max_p = self.fetch_program_page(prog.slug, page) + backfill_count += 1 + + if max_p > prog.total_pages: + prog.total_pages = max_p + + if total_cards == 0 or page >= max_p: + prog.backfill_complete = True + else: + prog.backfill_page = page + 1 + self.db.commit() + time.sleep(0.5) + + def run_full_sync(self, specific_program_slug=None): + if not self.ensure_auth(): + self.logger("RNŚ: Błąd logowania przed startem!") + raise Exception("RNS Login Failed") + + self.update_programs() + self.phase_1_catchup(specific_program_slug) + if self.requests_made < self.hard_limit: + self.phase_2_backfill(specific_program_slug) + + self.logger(f"RNŚ: Koniec. Wysłano zapytania: {self.requests_made}/{self.hard_limit} (w tym backfill ograniczony do {self.backfill_limit}).") diff --git a/core/security.py b/core/security.py new file mode 100644 index 0000000..d085ef2 --- /dev/null +++ b/core/security.py @@ -0,0 +1,45 @@ +import base64 +import hashlib +import os + +from cryptography.fernet import Fernet, InvalidToken + +_ENCRYPTED_PREFIX = "enc:v1:" + + +def get_session_secret() -> str: + secret = os.environ.get("RADIOSYNC_SECRET_KEY") + if not secret: + raise RuntimeError("RADIOSYNC_SECRET_KEY must be set") + return secret + + +def _get_fernet() -> Fernet: + configured_key = os.environ.get("RADIOSYNC_CONFIG_KEY") + if configured_key: + try: + return Fernet(configured_key.encode("ascii")) + except Exception as exc: + raise RuntimeError("RADIOSYNC_CONFIG_KEY must be a valid Fernet key") from exc + + digest = hashlib.sha256(get_session_secret().encode("utf-8")).digest() + return Fernet(base64.urlsafe_b64encode(digest)) + + +def encrypt_config_value(value: str) -> str: + if value is None: + return None + if value.startswith(_ENCRYPTED_PREFIX): + return value + token = _get_fernet().encrypt(value.encode("utf-8")).decode("ascii") + return f"{_ENCRYPTED_PREFIX}{token}" + + +def decrypt_config_value(value: str) -> str: + if value is None or not value.startswith(_ENCRYPTED_PREFIX): + return value + try: + token = value[len(_ENCRYPTED_PREFIX):].encode("ascii") + return _get_fernet().decrypt(token).decode("utf-8") + except InvalidToken as exc: + raise RuntimeError("Cannot decrypt config value with the configured key") from exc diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..ca46806 --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,13 @@ +services: + radiosync: + build: . + container_name: radiosync_hub + restart: unless-stopped + ports: + - "8000:8000" + volumes: + - ./data:/app/data + environment: + - TZ=Europe/Warsaw + - RADIOSYNC_SECRET_KEY=${RADIOSYNC_SECRET_KEY:?Set RADIOSYNC_SECRET_KEY in .env} + - RADIOSYNC_CONFIG_KEY=${RADIOSYNC_CONFIG_KEY:-} diff --git a/main.py b/main.py new file mode 100644 index 0000000..41c321c --- /dev/null +++ b/main.py @@ -0,0 +1,613 @@ +import os +import datetime +import threading +import hashlib +import re +import secrets +from fastapi import FastAPI, HTTPException, Depends, Request, Form, BackgroundTasks +from fastapi.responses import HTMLResponse, Response, RedirectResponse +from fastapi.staticfiles import StaticFiles +from fastapi.templating import Jinja2Templates +from sqlalchemy.orm import Session +from apscheduler.schedulers.background import BackgroundScheduler + +from core.database import SessionLocal, init_db, Config, Program, Episode, ErrorEvent +from core.scrapers.radio357 import Radio357Scraper +from core.scrapers.rns import RNScraper +from core.scrapers.jazz import JazzScraper +from core.rss_generator import generate_master_opml, generate_podcast_rss +from core.notifier import send_notification, validate_notification_url + +from starlette.middleware.sessions import SessionMiddleware + +app = FastAPI(title="RadioSync Hub") +session_secret = os.environ.get("RADIOSYNC_SECRET_KEY") +if not session_secret: + raise RuntimeError("RADIOSYNC_SECRET_KEY must be set") +app.add_middleware(SessionMiddleware, secret_key=session_secret) + +# Set up templates +base_dir = os.path.dirname(os.path.abspath(__file__)) +templates = Jinja2Templates(directory=os.path.join(base_dir, "templates")) +app.mount("/static", StaticFiles(directory=os.path.join(base_dir, "static")), name="static") + +init_db() + +scheduler = BackgroundScheduler() +sync_locks = {station: threading.Lock() for station in ("radio357", "rns", "jazz")} + +# State variables for logging +sync_status = { + "radio357": {"last_run": None, "status": "Oczekuje", "progress": "", "stop_requested": False}, + "rns": {"last_run": None, "status": "Oczekuje", "progress": "", "stop_requested": False}, + "jazz": {"last_run": None, "status": "Oczekuje", "progress": "", "stop_requested": False}, + "logs": [] +} + +def add_log(msg: str): + ts = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") + sync_status["logs"].insert(0, f"[{ts}] {msg}") + if len(sync_status["logs"]) > 100: + sync_status["logs"].pop() + print(msg) + +def db_session(): + db = SessionLocal() + try: + yield db + finally: + db.close() + +def notify_error(db: Session, station: str, message: str): + url = db.query(Config).filter_by(key="ntfy_url").first() + topic = db.query(Config).filter_by(key="ntfy_topic").first() + user = db.query(Config).filter_by(key="ntfy_user").first() + pw = db.query(Config).filter_by(key="ntfy_password").first() + + if url and topic: + try: + validate_notification_url(url.value) + send_notification(url.value, topic.value, f"RadioSync Błąd: {station}", message, + username=user.value if user else None, + password=pw.value if pw else None) + except Exception as exc: + add_log(f"Nie udało się wysłać powiadomienia ntfy: {exc}") + +def record_event(db: Session, station: str, message: str, source: str = "sync", severity: str = "error"): + now = datetime.datetime.now() + snapshot_match = re.search(r"Snapshot: (.+?)(?: URL:|$)", message) + snapshot_path = snapshot_match.group(1) if snapshot_match else None + stable_message = re.sub(r"\. Snapshot: .+?(?= URL:|$)", "", message) + fingerprint_message = re.sub(r" URL: https?://\S+", "", stable_message) + fingerprint_message = re.sub(r"\b\d+\b", "", fingerprint_message) + code_match = re.search(r"\[([A-Z0-9_]+)\]", stable_message) + stable_source = code_match.group(1) if code_match else source + fingerprint = hashlib.sha256(f"{station}:{stable_source}:{fingerprint_message}".encode("utf-8")).hexdigest() + try: + event = db.query(ErrorEvent).filter_by(fingerprint=fingerprint).first() + if event: + event.occurrences += 1 + event.last_seen = now + event.acknowledged = False + event.acknowledged_at = None + if snapshot_path: + event.snapshot_path = snapshot_path + else: + db.add(ErrorEvent(station=station, source=stable_source, message=stable_message, + severity=severity, fingerprint=fingerprint, first_seen=now, + last_seen=now, snapshot_path=snapshot_path)) + db.commit() + except Exception as exc: + db.rollback() + print(f"Failed to persist error event: {exc}") + label = "BŁĄD" if severity == "error" else "WARNING" + add_log(f"[{label}][{station}] {stable_message}") + +def record_error(db: Session, station: str, message: str, source: str = "sync"): + record_event(db, station, message, source, severity="error") + +def record_warning(db: Session, station: str, message: str, source: str = "sync"): + record_event(db, station, message, source, severity="warning") + +def is_user_stop(message: str): + return "Zatrzymano na żądanie użytkownika" in message + +def csrf_token(request: Request): + token = request.session.get("csrf_token") + if not token: + token = secrets.token_urlsafe(32) + request.session["csrf_token"] = token + return token + +def validate_csrf(request: Request, token: str): + if not token or token != request.session.get("csrf_token"): + raise HTTPException(status_code=403, detail="Nieprawidłowy token CSRF") + +def validate_interval(value: str, field: str): + try: + interval = int(value) + except (TypeError, ValueError) as exc: + raise HTTPException(status_code=400, detail=f"{field}: podaj liczbę godzin") from exc + if not 1 <= interval <= 168: + raise HTTPException(status_code=400, detail=f"{field}: zakres to 1-168 godzin") + return interval + +def validate_limit(value: str, field: str, maximum: int): + try: + limit = int(value) + except (TypeError, ValueError) as exc: + raise HTTPException(status_code=400, detail=f"{field}: podaj liczbę") from exc + if not 1 <= limit <= maximum: + raise HTTPException(status_code=400, detail=f"{field}: zakres to 1-{maximum}") + return limit + +def configured_interval(db: Session, key: str, default: int): + row = db.query(Config).filter_by(key=key).first() + try: + value = int(row.value) if row else default + except (TypeError, ValueError): + return default + return value if 1 <= value <= 168 else default + +def station_status_key(station: str): + return { + "Radio 357": "radio357", + "RNŚ": "rns", + "Radio Jazz FM": "jazz", + }.get(station) + +def persist_last_run(station: str, dt: datetime.datetime): + """Zapisuje datę ostatniego udanego syncu do bazy danych.""" + db = SessionLocal() + try: + existing = db.query(Config).filter_by(key=f"{station}_last_run").first() + if existing: + existing.value = dt.isoformat() + else: + db.add(Config(key=f"{station}_last_run", value=dt.isoformat())) + db.commit() + finally: + db.close() + +def persist_status(station: str, status: str): + db = SessionLocal() + try: + key = f"{station}_status" + existing = db.query(Config).filter_by(key=key).first() + if existing: + existing.value = status + else: + db.add(Config(key=key, value=status)) + db.commit() + finally: + db.close() + +def set_status(station: str, status: str): + sync_status[station]["status"] = status + persist_status(station, status) + +def load_last_runs(): + """Wczytuje daty ostatnich syncronizacji z bazy danych przy starcie.""" + db = SessionLocal() + try: + for station in ["radio357", "rns", "jazz"]: + row = db.query(Config).filter_by(key=f"{station}_last_run").first() + if row and row.value: + try: + sync_status[station]["last_run"] = datetime.datetime.fromisoformat(row.value) + except ValueError: + pass + status_row = db.query(Config).filter_by(key=f"{station}_status").first() + if status_row and status_row.value != "W trakcie": + sync_status[station]["status"] = status_row.value + finally: + db.close() + +def update_progress(station: str, made: int, limit: int): + sync_status[station]["progress"] = f"{made} / {limit}" + +# --- BACKGROUND JOBS --- + +def job_sync_radio357(specific_program=None): + if not sync_locks["radio357"].acquire(blocking=False): + add_log("R357: Pomijam synchronizację, ponieważ już trwa.") + return + db = None + try: + set_status("radio357", "W trakcie") + sync_status["radio357"]["progress"] = "0 / ?" + sync_status["radio357"]["stop_requested"] = False + add_log(f"Rozpoczęto synchronizację Radia 357 (program: {specific_program or 'wszystkie'})") + db = SessionLocal() + stop_fn = lambda: sync_status["radio357"]["stop_requested"] + scraper = Radio357Scraper(db, logger=add_log, + progress_callback=lambda m, l: update_progress("radio357", m, l), + stop_flag=stop_fn) + scraper.run_full_sync(specific_program) + now = datetime.datetime.now() + sync_status["radio357"]["last_run"] = now + set_status("radio357", "OK") + persist_last_run("radio357", now) + except Exception as e: + msg = f"Błąd R357: {e}" + if db: + if is_user_stop(msg): + record_warning(db, "Radio 357", "Synchronizacja zatrzymana na żądanie użytkownika.", "user_stop") + else: + record_error(db, "Radio 357", msg) + set_status("radio357", "Przerwano" if is_user_stop(msg) else "Błąd") + if db and not is_user_stop(msg): + notify_error(db, "Radio 357", msg) + finally: + sync_status["radio357"]["progress"] = "" + sync_status["radio357"]["stop_requested"] = False + if db: + db.close() + sync_locks["radio357"].release() + +def job_sync_rns(specific_program=None): + if not sync_locks["rns"].acquire(blocking=False): + add_log("RNŚ: Pomijam synchronizację, ponieważ już trwa.") + return + db = None + try: + set_status("rns", "W trakcie") + sync_status["rns"]["progress"] = "0 / ?" + sync_status["rns"]["stop_requested"] = False + add_log(f"Rozpoczęto synchronizację RNŚ (program: {specific_program or 'wszystkie'})") + db = SessionLocal() + stop_fn = lambda: sync_status["rns"]["stop_requested"] + scraper = RNScraper(db, logger=add_log, + progress_callback=lambda m, l: update_progress("rns", m, l), + warning_callback=lambda message: record_warning(db, "RNŚ", message, "missing_player"), + stop_flag=stop_fn) + scraper.run_full_sync(specific_program) + now = datetime.datetime.now() + sync_status["rns"]["last_run"] = now + set_status("rns", "OK") + persist_last_run("rns", now) + except Exception as e: + msg = f"Błąd RNŚ: {e}" + if db: + if is_user_stop(msg): + record_warning(db, "RNŚ", "Synchronizacja zatrzymana na żądanie użytkownika.", "user_stop") + else: + record_error(db, "RNŚ", msg) + set_status("rns", "Przerwano" if is_user_stop(msg) else "Błąd") + if db and not is_user_stop(msg): + notify_error(db, "RNŚ", msg) + finally: + sync_status["rns"]["progress"] = "" + sync_status["rns"]["stop_requested"] = False + if db: + db.close() + sync_locks["rns"].release() + +def job_sync_jazz(): + if not sync_locks["jazz"].acquire(blocking=False): + add_log("JAZZ: Pomijam synchronizację, ponieważ już trwa.") + return + db = None + try: + set_status("jazz", "W trakcie") + add_log("Rozpoczęto synchronizację Radio Jazz FM") + db = SessionLocal() + scraper = JazzScraper(db, logger=add_log) + scraper.run_full_sync() + sync_status["jazz"]["last_run"] = datetime.datetime.now() + now = datetime.datetime.now() + sync_status["jazz"]["last_run"] = now + set_status("jazz", "OK") + persist_last_run("jazz", now) + except Exception as e: + msg = f"Błąd JAZZ: {e}" + if db: + if is_user_stop(msg): + record_warning(db, "Radio Jazz FM", "Synchronizacja zatrzymana na żądanie użytkownika.", "user_stop") + else: + record_error(db, "Radio Jazz FM", msg) + set_status("jazz", "Przerwano" if is_user_stop(msg) else "Błąd") + if db and not is_user_stop(msg): + notify_error(db, "Radio Jazz FM", msg) + finally: + if db: + db.close() + sync_locks["jazz"].release() + +@app.on_event("startup") +def startup_event(): + db = SessionLocal() + r357_int = configured_interval(db, "r357_interval", 6) + rns_int = configured_interval(db, "rns_interval", 6) + jazz_int = configured_interval(db, "jazz_interval", 24) + db.close() + + scheduler.add_job(job_sync_radio357, 'interval', hours=r357_int, id='sync_r357', max_instances=1, coalesce=True) + scheduler.add_job(job_sync_rns, 'interval', hours=rns_int, id='sync_rns', max_instances=1, coalesce=True) + scheduler.add_job(job_sync_jazz, 'interval', hours=jazz_int, id='sync_jazz', max_instances=1, coalesce=True) + scheduler.start() + load_last_runs() + add_log("Aplikacja i harmonogram uruchomione.") + +@app.on_event("shutdown") +def shutdown_event(): + scheduler.shutdown() + +# --- ROUTES --- + +@app.get("/", response_class=HTMLResponse) +def index(request: Request, db: Session = Depends(db_session)): + msg = request.session.pop("msg", None) + stats = { + "r357_progs": db.query(Program).filter_by(station="radio357").count(), + "r357_eps": db.query(Episode).filter_by(station="radio357").count(), + "r357_audio_eps": db.query(Episode).filter_by(station="radio357", is_broken=False).filter(Episode.url.isnot(None)).count(), + "rns_progs": db.query(Program).filter_by(station="rns").count(), + "rns_eps": db.query(Episode).filter_by(station="rns").count(), + "rns_audio_eps": db.query(Episode).filter_by(station="rns", is_broken=False).filter(Episode.url.isnot(None)).count(), + } + + config = {c.key: c.value for c in db.query(Config).all()} + pending_errors = db.query(ErrorEvent).filter_by(acknowledged=False).order_by(ErrorEvent.last_seen.desc()).all() + acknowledged_errors = db.query(ErrorEvent).filter_by(acknowledged=True).order_by(ErrorEvent.last_seen.desc()).limit(20).all() + + return templates.TemplateResponse(request, "index.html", { + "status": sync_status, + "stats": stats, + "config": config, + "msg": msg, + "pending_errors": pending_errors, + "acknowledged_errors": acknowledged_errors, + "csrf_token": csrf_token(request), + }) + +@app.post("/errors/{error_id}/acknowledge") +def acknowledge_error(request: Request, error_id: int, csrf: str = Form(...), db: Session = Depends(db_session)): + validate_csrf(request, csrf) + event = db.query(ErrorEvent).filter_by(id=error_id).first() + if event: + event.acknowledged = True + event.acknowledged_at = datetime.datetime.now() + db.commit() + add_log(f"Potwierdzono odczytanie błędu: {event.message}") + status_key = station_status_key(event.station) + pending_for_station = db.query(ErrorEvent).filter_by( + station=event.station, acknowledged=False + ).count() + if status_key and pending_for_station == 0 and sync_status[status_key]["status"] == "Błąd": + set_status(status_key, "Oczekuje") + return RedirectResponse(url="/", status_code=303) + +@app.post("/config") +def save_config( + request: Request, + csrf: str = Form(...), + r357_email: str = Form(""), r357_password: str = Form(""), + rns_email: str = Form(""), rns_password: str = Form(""), + ntfy_url: str = Form(""), ntfy_topic: str = Form(""), + ntfy_user: str = Form(""), ntfy_password: str = Form(""), + r357_limit: str = Form("600"), rns_limit: str = Form("50"), + r357_hard_limit: str = Form("1000"), rns_hard_limit: str = Form("500"), + r357_interval: str = Form("6"), rns_interval: str = Form("6"), jazz_interval: str = Form("24"), + db: Session = Depends(db_session) +): + validate_csrf(request, csrf) + r357_interval_value = validate_interval(r357_interval, "r357_interval") + rns_interval_value = validate_interval(rns_interval, "rns_interval") + jazz_interval_value = validate_interval(jazz_interval, "jazz_interval") + validate_limit(r357_limit, "r357_limit", 10000) + validate_limit(rns_limit, "rns_limit", 10000) + validate_limit(r357_hard_limit, "r357_hard_limit", 20000) + validate_limit(rns_hard_limit, "rns_hard_limit", 20000) + if ntfy_url: + try: + validate_notification_url(ntfy_url) + except ValueError as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + import urllib.parse + updates = { + "r357_email": r357_email, "r357_password": r357_password, + "rns_email": rns_email, "rns_password": rns_password, + "ntfy_url": ntfy_url, "ntfy_topic": ntfy_topic, + "ntfy_user": ntfy_user, "ntfy_password": ntfy_password, + "r357_limit": r357_limit, "rns_limit": rns_limit, + "r357_hard_limit": r357_hard_limit, "rns_hard_limit": rns_hard_limit, + "r357_interval": r357_interval, "rns_interval": rns_interval, "jazz_interval": jazz_interval + } + + for k, v in updates.items(): + if v: + c = db.query(Config).filter_by(key=k).first() + if not c: + db.add(Config(key=k, value=v)) + else: + c.value = v + db.commit() + + scheduler.reschedule_job('sync_r357', trigger='interval', hours=r357_interval_value) + scheduler.reschedule_job('sync_rns', trigger='interval', hours=rns_interval_value) + scheduler.reschedule_job('sync_jazz', trigger='interval', hours=jazz_interval_value) + + if r357_email or r357_password: + db.query(Config).filter(Config.key == "r357_token").delete(synchronize_session=False) + if rns_email or rns_password: + db.query(Config).filter(Config.key == "rns_cookies").delete(synchronize_session=False) + db.commit() + + m = "Zapisano nową konfigurację. Harmonogram i sesje zresetowane." + add_log(m) + request.session["msg"] = m + return RedirectResponse(url="/", status_code=303) + +@app.post("/test-ntfy") +def test_ntfy(request: Request, csrf: str = Form(...), db: Session = Depends(db_session)): + validate_csrf(request, csrf) + import urllib.parse + notify_error(db, "TEST", "To jest wiadomość testowa z RadioSync Hub.") + m = "Wysłano testowe powiadomienie ntfy." + add_log(m) + request.session["msg"] = m + return RedirectResponse(url="/", status_code=303) + +@app.post("/test-auth") +def test_auth(request: Request, station: str = Form(...), csrf: str = Form(...), db: Session = Depends(db_session)): + validate_csrf(request, csrf) + import urllib.parse + m = "" + if station == "radio357": + scraper = Radio357Scraper(db, logger=add_log) + ok, msg = scraper.check_auth_status() + m = f"R357: {'✅' if ok else '❌'} {msg}" + elif station == "rns": + scraper = RNScraper(db, logger=add_log) + ok, msg = scraper.check_auth_status() + m = f"RNŚ: {'✅' if ok else '❌'} {msg}" + add_log(m) + request.session["msg"] = m + return RedirectResponse(url="/", status_code=303) + +@app.post("/force-relogin") +def force_relogin(request: Request, station: str = Form(...), csrf: str = Form(...), db: Session = Depends(db_session)): + validate_csrf(request, csrf) + import urllib.parse + add_log(f"Wymuszony relogin dla: {station}...") + m = "" + if station == "radio357": + db.query(Config).filter_by(key="r357_token").delete() + db.commit() + scraper = Radio357Scraper(db, logger=add_log) + if scraper.get_token(): + m = "R357: ✅ Relogin poprawny!" + else: + m = "R357: ❌ Relogin nieudany (sprawdź dane)." + elif station == "rns": + db.query(Config).filter_by(key="rns_cookies").delete() + db.commit() + scraper = RNScraper(db, logger=add_log) + if scraper.perform_login(): + m = "RNŚ: ✅ Relogin poprawny!" + else: + m = "RNŚ: ❌ Relogin nieudany (sprawdź dane)." + add_log(m) + request.session["msg"] = m + return RedirectResponse(url="/", status_code=303) + +@app.post("/force-sync") +def force_sync(request: Request, station: str = Form(...), csrf: str = Form(...), bg_tasks: BackgroundTasks = BackgroundTasks()): + validate_csrf(request, csrf) + import urllib.parse + if station not in sync_locks: + raise HTTPException(status_code=400, detail="Unknown station") + if sync_locks[station].locked(): + m = f"Synchronizacja dla {station} już trwa." + add_log(m) + request.session["msg"] = m + return RedirectResponse(url="/", status_code=303) + if station == "radio357": + bg_tasks.add_task(job_sync_radio357) + elif station == "rns": + bg_tasks.add_task(job_sync_rns) + elif station == "jazz": + bg_tasks.add_task(job_sync_jazz) + + names = {"radio357": "Radia 357", "rns": "Radia Nowy Świat", "jazz": "Radio Jazz FM"} + m = f"Rozpoczęto w tle synchronizację: {names.get(station, station)}." + add_log(m) + request.session["msg"] = m + return RedirectResponse(url="/", status_code=303) + +@app.post("/stop-sync") +def stop_sync(request: Request, station: str = Form(...), csrf: str = Form(...)): + validate_csrf(request, csrf) + if station in sync_status and sync_status[station]["status"] == "W trakcie": + sync_status[station]["stop_requested"] = True + m = f"Wysłano sygnał zatrzymania synchronizacji: {station}." + else: + m = f"Sync dla {station} nie jest aktualnie uruchomiony." + add_log(m) + request.session["msg"] = m + return RedirectResponse(url="/", status_code=303) + +@app.post("/force-sync-program") +def force_sync_program(request: Request, station: str = Form(...), slug: str = Form(...), csrf: str = Form(...), bg_tasks: BackgroundTasks = BackgroundTasks()): + validate_csrf(request, csrf) + import urllib.parse + if station not in {"radio357", "rns"}: + raise HTTPException(status_code=400, detail="Station does not support program sync") + if sync_locks[station].locked(): + m = f"Synchronizacja dla {station} już trwa." + add_log(m) + request.session["msg"] = m + return RedirectResponse(url=f"/programs/{station}", status_code=303) + if station == "radio357": + bg_tasks.add_task(job_sync_radio357, specific_program=slug) + else: + bg_tasks.add_task(job_sync_rns, specific_program=slug) + m = f"Zlecono wymuszoną synchronizację pojedynczego programu: {slug} ({station})" + add_log(m) + request.session["msg"] = m + return RedirectResponse(url=f"/programs/{station}", status_code=303) + +@app.get("/programs/{station}", response_class=HTMLResponse) +def list_programs(request: Request, station: str, db: Session = Depends(db_session)): + msg = request.session.pop("msg", None) + import datetime + progs = db.query(Program).filter_by(station=station).order_by(Program.name).all() + + for p in progs: + p.missing_urls_count = db.query(Episode).filter_by(station=station, program_slug=p.slug, url=None).count() + p.total_eps = db.query(Episode).filter_by(station=station, program_slug=p.slug).count() + if p.last_catchup: + p.last_catchup_str = datetime.datetime.fromtimestamp(p.last_catchup).strftime('%Y-%m-%d %H:%M') + else: + p.last_catchup_str = "Nigdy" + + return templates.TemplateResponse(request, "programs.html", {"station": station, "programs": progs, "msg": msg, "csrf_token": csrf_token(request)}) + +@app.get("/episodes/{station}/{slug:path}", response_class=HTMLResponse) +def list_episodes(request: Request, station: str, slug: str, db: Session = Depends(db_session)): + msg = request.session.pop("msg", None) + prog = db.query(Program).filter_by(station=station, slug=slug).first() + if not prog: + raise HTTPException(status_code=404, detail="Program not found") + eps = db.query(Episode).filter_by(station=station, program_slug=slug).order_by(Episode.pub_date.desc()).all() + return templates.TemplateResponse(request, "episodes.html", {"station": station, "program": prog, "episodes": eps, "msg": msg, "csrf_token": csrf_token(request)}) + +@app.post("/episodes/delete-url") +def delete_episode_url(request: Request, ep_id: int = Form(...), csrf: str = Form(...), db: Session = Depends(db_session)): + validate_csrf(request, csrf) + import urllib.parse + ep = db.query(Episode).filter_by(id=ep_id).first() + if ep: + ep.url = None + ep.is_broken = True + db.commit() + m = f"Usunięto błędny link audio dla odcinka: {ep.title}" + add_log(m) + request.session["msg"] = m + return RedirectResponse(url=f"/episodes/{ep.station}/{ep.program_slug}", status_code=303) + return RedirectResponse(url="/", status_code=303) + +# --- RSS ENDPOINTS --- + +@app.get("/feeds/Podcasts.opml") +def get_master_opml(request: Request, db: Session = Depends(db_session)): + xml = generate_master_opml(db, str(request.base_url).rstrip('/')) + return Response(content=xml, media_type="application/xml") + +@app.get("/feeds/{station}.opml") +def get_station_opml(request: Request, station: str, db: Session = Depends(db_session)): + if station not in {"radio357", "rns", "jazz"}: + raise HTTPException(status_code=404, detail="Station not found") + xml = generate_master_opml(db, str(request.base_url).rstrip('/'), station=station) + return Response(content=xml, media_type="application/xml") + +@app.get("/feeds/{station}/{slug_with_ext:path}") +def podcast_rss(station: str, slug_with_ext: str, request: Request, db: Session = Depends(db_session)): + if not slug_with_ext.endswith(".xml"): + raise HTTPException(status_code=404, detail="Not Found") + slug = slug_with_ext[:-4] + xml = generate_podcast_rss(db, station, slug) + if not xml: + return Response(status_code=404) + return Response(content=xml, media_type="application/xml") diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..b213345 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,10 @@ +fastapi +uvicorn[standard] +sqlalchemy +jinja2 +beautifulsoup4 +requests +apscheduler +python-multipart +itsdangerous +cryptography diff --git a/static/favicon.svg b/static/favicon.svg new file mode 100644 index 0000000..b2818e6 --- /dev/null +++ b/static/favicon.svg @@ -0,0 +1,5 @@ + + + + + diff --git a/templates/base.html b/templates/base.html new file mode 100644 index 0000000..192c25b --- /dev/null +++ b/templates/base.html @@ -0,0 +1,103 @@ + + + + + + + RadioSync Hub + + + + + + + + +
+ {% block content %}{% endblock %} +
+ + + diff --git a/templates/episodes.html b/templates/episodes.html new file mode 100644 index 0000000..bba411f --- /dev/null +++ b/templates/episodes.html @@ -0,0 +1,53 @@ +{% extends "base.html" %} + +{% block content %} + + +{% if msg %} + +{% endif %} + +
+
Odcinki

{{ program.name }}

+ Skopiuj URL Feedu (RSS) +
+ +
+ {% for ep in episodes %} +
+
+
{{ ep.title }}
+ {{ ep.pub_date }} +
+

{{ (ep.description or '')[:200] }}{% if ep.description and ep.description|length > 200 %}...{% endif %}

+
+
+ {% if ep.url %} + + {% else %} + Brak pliku audio (lub oznaczony jako błąd) + {% endif %} +
+ {% if ep.url %} +
+ + + +
+ {% endif %} +
+
+ {% endfor %} +
+{% endblock %} diff --git a/templates/index.html b/templates/index.html new file mode 100644 index 0000000..0fefffd --- /dev/null +++ b/templates/index.html @@ -0,0 +1,233 @@ +{% extends "base.html" %} + +{% block content %} +{% set is_running = (status.radio357.status == 'W trakcie') or (status.rns.status == 'W trakcie') or (status.jazz.status == 'W trakcie') %} +{% if is_running %} + +{% endif %} +
+ {% if msg %} +
+ +
+ {% endif %} + +
+
+
Panel operacyjny
+

Status systemu

+
+
+ {% for st, name in [('radio357', 'Radio 357'), ('rns', 'Radio Nowy Świat'), ('jazz', 'Radio Jazz')] %} +
+
+
+ {{ name }} +
+
+
+ Status + {% if status[st].status == 'OK' %}OK + {% elif status[st].status == 'Błąd' %}Błąd + {% elif status[st].status == 'W trakcie' %} + W trakcie + {% if status[st].progress %}
Zapytania: {{ status[st].progress }}{% endif %} + {% else %}{{ status[st].status }}{% endif %} +
+
+
Ostatni sync{% if status[st].last_run %}{{ status[st].last_run.strftime('%d.%m.%Y %H:%M') }}{% else %}Nigdy{% endif %}
+ {% if st in ['radio357', 'rns'] %} +
Programy{{ stats['r357_progs' if st == 'radio357' else 'rns_progs'] }}
+
Audio / odcinki{{ stats['r357_audio_eps' if st == 'radio357' else 'rns_audio_eps'] }} / {{ stats['r357_eps' if st == 'radio357' else 'rns_eps'] }}
+ {% endif %} +
+
+ +
+
+ {% endfor %} +
+
+

Błędy wymagające uwagi

+ + {{ pending_errors|length }} + +
+ {% if pending_errors %} +
+ {% for error in pending_errors %} +
+
+ {% if error.severity == 'warning' %}Ostrzeżenie{% else %}Błąd{% endif %}: {{ error.station }} + {{ error.last_seen.strftime('%Y-%m-%d %H:%M') }} +
+
{{ error.message }}
+
+ Powtórzenia: {{ error.occurrences }} +
+ + +
+
+
+ {% endfor %} +
+ {% else %} +
Brak błędów wymagających odczytania.
+ {% endif %} + + {% if acknowledged_errors %} +
+ Ostatnio odczytane błędy ({{ acknowledged_errors|length }}) +
+ {% for error in acknowledged_errors %} +
+
+ {{ error.station }} + {{ error.last_seen.strftime('%Y-%m-%d %H:%M') }} +
+
{{ error.message }}
+ Powtórzenia: {{ error.occurrences }} +
+ {% endfor %} +
+
+ {% endif %} + +

Ostatnie logi (z pamięci podręcznej)

+
+
+
+ {% for line in status.logs %} +
{{ line }}
+ {% endfor %} +
+
+
+
+ +
+
+
Konfiguracja Systemu
+
+
+ +
Radio 357
+
+ +
+
+ +
+
+
+ + +
+
+ + +
+
+ + +
+
+ +
Radio Nowy Świat
+
+ +
+
+ +
+
+
+ + +
+
+ + +
+
+ + +
+
+ +
Radio Jazz
+
+
+ + +
+
+ +
Powiadomienia ntfy.sh
+
+ +
+
+ +
+
+
+ +
+
+ +
+
+ +
+ +
+
+
+ + +
+
+
+
+
+{% endblock %} diff --git a/templates/programs.html b/templates/programs.html new file mode 100644 index 0000000..e01ded9 --- /dev/null +++ b/templates/programs.html @@ -0,0 +1,67 @@ +{% extends "base.html" %} + +{% block content %} +
+
Biblioteka

Programy: {{ station }}

+ OPML dla tej stacji +
+ +{% if msg %} + +{% endif %} + +
+ {% for prog in programs %} +
+
+
+
+ {% if prog.image %} + Okładka + {% endif %} +
{{ prog.name }}
+
+

{{ prog.description or 'Brak opisu programu.' }}

+
+
Ostatni sync: {{ prog.last_catchup_str }}
+ {% if station == 'rns' %} +
Status backfill: + {% if prog.backfill_complete or (prog.total_pages > 0 and prog.backfill_page > prog.total_pages) %} + ✅ Zakończony + {% elif prog.backfill_page <= 2 %} + {% if prog.total_eps > 0 %} + Catch-up wykonany; od strony 2 + {% else %} + Nie rozpoczęty + {% endif %} + {% else %} + Strona {{ prog.backfill_page - 1 }} z {{ prog.total_pages if prog.total_pages > 0 else '?' }} + {% endif %} +
+ {% endif %} + {% if station != 'jazz' %} +
Odcinki: {{ prog.total_eps }} (brak audio: {% if prog.missing_urls_count > 0 %}{{ prog.missing_urls_count }}{% else %}0{% endif %})
+ {% endif %} +
+
+ +
+
+ {% endfor %} +
+{% endblock %}