Files
radiosync/core/scrapers/rns.py
T
karol 76f5d30db1 Initial commit
Co-authored-by: GitHub Copilot <noreply@github.com>
2026-09-03 11:57:16 +02:00

509 lines
23 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import requests
import time
import os
import re
import urllib.parse
from bs4 import BeautifulSoup
from sqlalchemy.orm import Session
from ..database import Program, Episode, Config
from ..diagnostics import ScraperError, invalid_response
MAX_REQUESTS_LIMIT = 50
BASE_URL = "https://nowyswiat.online"
def get_cookie_for_url(cookie_jar, name, url):
"""Wybiera najbardziej szczegółowe cookie, gdy jar zawiera duplikaty nazwy."""
parsed_url = urllib.parse.urlparse(url)
host = parsed_url.hostname.lower()
path = parsed_url.path or "/"
candidates = []
for cookie in cookie_jar:
if cookie.name != name or (cookie.secure and parsed_url.scheme != "https"):
continue
domain = cookie.domain.lstrip(".").lower()
if domain and not (host == domain or host.endswith(f".{domain}")):
continue
cookie_path = cookie.path or "/"
if not path.startswith(cookie_path.rstrip("/") or "/"):
continue
candidates.append(cookie)
if not candidates:
return None
selected = max(candidates, key=lambda cookie: (len(cookie.path or "/"), len(cookie.domain or "")))
for cookie in candidates:
if cookie is not selected:
cookie_jar.clear(cookie.domain, cookie.path, cookie.name)
return selected.value
def parse_polish_date(date_str):
months = {
"stycznia": "01", "lutego": "02", "marca": "03", "kwietnia": "04",
"maja": "05", "czerwca": "06", "lipca": "07", "sierpnia": "08",
"września": "09", "października": "10", "listopada": "11", "grudnia": "12",
"styczeń": "01", "luty": "02", "marzec": "03", "kwiecień": "04",
"maj": "05", "czerwiec": "06", "lipiec": "07", "sierpień": "08",
"wrzesień": "09", "październik": "10", "listopad": "11", "grudzień": "12"
}
parts = date_str.lower().split()
if len(parts) == 3:
day = parts[0].zfill(2)
month = months.get(parts[1], "01")
year = parts[2]
return f"{year}-{month}-{day}"
return date_str
class RNScraper:
def __init__(self, db: Session, logger=print, progress_callback=None, warning_callback=None, stop_flag=None):
self.db = db
self.logger = logger
self.progress_callback = progress_callback
self.warning_callback = warning_callback
self.stop_flag = stop_flag
self.requests_made = 0
self.start_time = time.time()
self.max_execution_time = 900 # 15 minut max na cały sync
limit_conf = self.db.query(Config).filter_by(key="rns_limit").first()
self.backfill_limit = int(limit_conf.value) if limit_conf and limit_conf.value.isdigit() else 50
hard_limit_conf = self.db.query(Config).filter_by(key="rns_hard_limit").first()
self.hard_limit = int(hard_limit_conf.value) if hard_limit_conf and hard_limit_conf.value.isdigit() else 500
self.session = requests.Session()
self.session.headers.update({"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"})
if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit)
def check_timeout(self):
if self.stop_flag and self.stop_flag():
raise Exception("Zatrzymano na żądanie użytkownika.")
if time.time() - self.start_time > self.max_execution_time:
raise Exception(f"Przekroczono limit czasu wykonywania skryptu ({self.max_execution_time // 60} min). Zatrzymano awaryjnie.")
def _get_html(self, url):
self.check_timeout()
if self.requests_made >= self.hard_limit:
return None
for attempt in range(1, 4):
if self.requests_made >= self.hard_limit:
return None
self.requests_made += 1
if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit)
try:
r = self.session.get(url, timeout=15)
if r.status_code == 200:
return r.text
invalid_response("rns", "HTTP_ERROR", "Błąd pobierania strony", url, r)
except requests.exceptions.Timeout as e:
if attempt == 3:
raise ScraperError("rns", "NETWORK_TIMEOUT", "Trzy próby pobrania strony zakończyły się timeoutem", url) from e
self.logger(f"RNŚ: Timeout dla {url}, próba {attempt}/3. Ponawiam za 30 sekund.")
self.check_timeout()
time.sleep(30)
except Exception as e:
if isinstance(e, ScraperError):
raise
raise ScraperError("rns", "NETWORK_ERROR", f"Błąd zapytania: {e}", url) from e
return None
def perform_login(self):
email = self.db.query(Config).filter_by(key="rns_email").first()
password = self.db.query(Config).filter_by(key="rns_password").first()
if not email or not password:
self.logger("RNŚ: Brak danych logowania w bazie.")
return False
self.logger("RNŚ: Inicjalizacja logowania (pobieranie CSRF)...")
try:
r1 = self.session.get("https://nowyswiat.online/konto/zaloguj", timeout=15)
except Exception as e:
self.logger(f"RNŚ: Sieć zablokowała pobieranie CSRF: {e}")
return False
csrf_token = get_cookie_for_url(
self.session.cookies,
"csrf_cookie_neocms",
"https://nowyswiat.online/konto/zaloguj",
)
if not csrf_token:
self.logger("RNŚ: Nie udało się pobrać tokenu CSRF.")
return False
self.logger("RNŚ: Wysyłanie formularza...")
payload = {"csrf_neocms": csrf_token, "login": email.value, "password": password.value, "ufd_data": "{}"}
try:
r2 = self.session.post("https://nowyswiat.online/konto/zaloguj", data=payload, headers={
"X-Requested-With": "XMLHttpRequest"
}, timeout=15)
except Exception as e:
self.logger(f"RNŚ: Błąd sieci przy wysyłaniu formularza: {e}")
return False
# Check if login succeeded by looking for a session cookie or a success response
if '"status":"OK"' in r2.text or "logowanie udane" in r2.text.lower():
# Save cookies to DB (serialize)
cookies_dict = requests.utils.dict_from_cookiejar(self.session.cookies)
import json
cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first()
if not cookie_conf:
self.db.add(Config(key="rns_cookies", value=json.dumps(cookies_dict)))
else:
cookie_conf.value = json.dumps(cookies_dict)
self.db.commit()
return True
else:
self.logger(f"RNŚ: Błędne dane logowania (lub zmiana mechanizmu). Odpowiedź: {r2.text[:100]}")
return False
def ensure_auth(self):
cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first()
if cookie_conf:
import json
try:
cookies_dict = json.loads(cookie_conf.value)
self.session.cookies = requests.utils.cookiejar_from_dict(cookies_dict)
html = self._get_html(f"{BASE_URL}/")
if html and "wyloguj" in html.lower():
return True
self.logger("RNŚ: Zapisana sesja wygasła, wykonuję ponowne logowanie.")
except:
pass
return self.perform_login()
def check_auth_status(self):
cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first()
if not cookie_conf:
return False, "Brak zapisanych ciasteczek."
import json
try:
self.session.cookies = requests.utils.cookiejar_from_dict(json.loads(cookie_conf.value))
html = self._get_html("https://nowyswiat.online/")
if html and "wyloguj" in html.lower():
return True, "Ciasteczka aktywne, sesja poprawna."
return False, "Ciasteczka nieaktywne lub wygasły (brak dostępu do profilu)."
except Exception as e:
return False, f"Błąd: {e}"
def update_programs(self):
self.logger("RNŚ: Aktualizacja listy programów...")
html = self._get_html("https://nowyswiat.online/podcasty")
if not html: return
soup = BeautifulSoup(html, "html.parser")
links = soup.find_all("a", href=True)
program_links = [link for link in links if "rbroadcast=" in link["href"]]
if not program_links:
invalid_response("rns", "HTML_SCHEMA_CHANGED", "Nie znaleziono programów w stronie podcastów", "https://nowyswiat.online/podcasty", content=html)
for link in program_links:
href = link["href"]
if "rbroadcast=" in href:
parsed = urllib.parse.urlparse(href)
slug = urllib.parse.parse_qs(parsed.query).get("rbroadcast", [None])[0]
if not slug: continue
title_el = link.find("h2", class_="rns-search-dropdown-title")
title = title_el.text.strip() if title_el else slug
img_el = link.find("img")
img_url = img_el["src"] if img_el and img_el.has_attr("src") else ""
if img_url and not img_url.startswith("http"): img_url = f"{BASE_URL}/{img_url.lstrip('/')}"
prog = self.db.query(Program).filter_by(station="rns", slug=slug).first()
if not prog:
self.db.add(Program(
station="rns", slug=slug, name=title, image=img_url,
description=f"Radio Nowy Świat: {title}", backfill_page=2, backfill_complete=False
))
self.db.commit()
def _verify_audio_teaser(self, ep: Episode):
"""Wykrywa 1-minutowy teaser (rozmiar mniejszy niż ~2MB, mimo że audycja trwa > 5 min)."""
self.check_timeout()
if not ep.url: return False
if self.requests_made >= self.hard_limit: return False
self.requests_made += 1
if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit)
try:
r = self.session.head(ep.url, timeout=5)
cl = int(r.headers.get("Content-Length", 0))
if cl > 0 and cl < 2_500_000 and ep.duration_secs > 300:
self.logger(f"RNŚ: Wykryto uszkodzony link (Teaser 1-min) dla {ep.title}. Usuwam URL.")
ep.url = None
ep.is_broken = True
return True
except Exception as e:
pass
return False
def fetch_program_page(self, program_slug, page):
url = f"https://nowyswiat.online/podcasty?rbroadcast={program_slug}&page={page}"
html = self._get_html(url)
if not html:
raise Exception(f"Błąd sieci podczas pobierania strony {page} audycji {program_slug}")
soup = BeautifulSoup(html, "html.parser")
cards = soup.find_all("a", class_="rns-grid-podcast-card")
known_episode_count = self.db.query(Episode).filter_by(
station="rns", program_slug=program_slug
).count()
page_has_podcast_content = bool(soup.find(string=re.compile(r"podcast|podkast", re.IGNORECASE)))
if not cards and page > 1:
return 0, 0, page - 1
if not cards and (known_episode_count or not page_has_podcast_content):
invalid_response("rns", "HTML_SCHEMA_CHANGED", "Nie znaleziono kart odcinków dla istniejącego programu", url, content=html)
new_found = 0
missing_player_count = 0
valid_cards = 0
for card in cards:
href = card.get("href", "")
if not href or "/podcasty/" not in href: continue
valid_cards += 1
ep_id = href.split("/podcasty/")[-1].split("?")[0]
player_box = card.find("div", class_="rns-play-btn-box")
audio_url = player_box.get("data-neo-player-src") if player_box else None
if not player_box:
missing_player_count += 1
if audio_url and not audio_url.startswith("http"):
audio_url = f"{BASE_URL}/{audio_url.lstrip('/')}"
title_el = card.find("p", class_="rns-post-title")
raw_title = player_box.get("data-neo-player-title", "") if player_box else (title_el.text.strip() if title_el else ep_id)
title = BeautifulSoup(raw_title, "html.parser").text.strip() if raw_title else ep_id
raw_subtitle = player_box.get("data-neo-player-subtitle", "") if player_box else ""
authors = BeautifulSoup(raw_subtitle, "html.parser").text.strip() if raw_subtitle else "Radio Nowy Świat"
date_el = card.find("p", class_="rns-podcast-details-date")
pub_date = parse_polish_date(date_el.text.strip()) if date_el else ""
time_el = card.find("p", class_="rns-podcast-details-long")
duration_str = time_el.text.strip() if time_el else "0:00"
duration_secs = 0
if ":" in duration_str:
parts = duration_str.split(":")
if len(parts) == 3: duration_secs = int(parts[0])*3600 + int(parts[1])*60 + int(parts[2])
elif len(parts) == 2: duration_secs = int(parts[0])*60 + int(parts[1])
img_url = player_box.get("data-neo-player-img", "") if player_box else ""
if img_url and not img_url.startswith("http"): img_url = f"{BASE_URL}/{img_url.lstrip('/')}"
desc_el = card.find("p", class_="rns-post-card-desc")
description = desc_el.get_text(separator="\n").strip() if desc_el else ""
ep = self.db.query(Episode).filter_by(station="rns", ep_id=ep_id).first()
if not ep:
ep = Episode(
station="rns", program_slug=program_slug, ep_id=ep_id,
title=title, authors=authors, url=audio_url, image=img_url,
pub_date=pub_date, duration_secs=duration_secs, description=description,
is_broken=False
)
self.db.add(ep)
new_found += 1
else:
if not ep.url and audio_url:
ep.url = audio_url
ep.is_broken = False
# Celowo nie zwiększamy new_found, aby Daily Catchup nie wchodził w tryb głębokiego archiwum.
# Łataniem starych dziur zajmie się faza Backfill.
if cards and not valid_cards:
invalid_response("rns", "HTML_SCHEMA_CHANGED", "Znaleziono karty podcastów bez oczekiwanych identyfikatorów", url, content=html)
if missing_player_count:
warning = f"{missing_player_count} odcinków bez dostępnego playera audio; pomijam ich URL-e."
if self.warning_callback:
self.warning_callback(warning)
else:
self.logger(f"RNŚ: WARNING: {warning}")
self.db.commit()
page_links = soup.find_all("a", href=re.compile(r"page=\d+"))
max_page = page
for link in page_links:
href = link.get("href")
m = re.search(r"page=(\d+)", href)
if m:
max_page = max(max_page, int(m.group(1)))
return new_found, len(cards), max_page
def _compute_next_catchup(self, prog: Program) -> float:
"""Wylicza kiedy najwcześniej warto znowu sprawdzać tę audycję."""
episodes = self.db.query(Episode).filter_by(
station="rns", program_slug=prog.slug
).order_by(Episode.pub_date.desc()).limit(20).all()
if len(episodes) < 3:
# Za mało danych – sprawdzamy przy każdym uruchomieniu
return 0.0
# Wylicz średnią przerwę między odcinkami w dniach
import re
dates = []
for ep in episodes:
if ep.pub_date and re.match(r'\d{4}-\d{2}-\d{2}', ep.pub_date):
try:
import datetime
dates.append(datetime.date.fromisoformat(ep.pub_date[:10]))
except ValueError:
pass
if len(dates) < 3:
return 0.0
dates.sort(reverse=True)
gaps = [(dates[i] - dates[i+1]).days for i in range(len(dates)-1)]
avg_gap = sum(gaps) / len(gaps)
if avg_gap < 2:
# Codziennie lub częściej → zawsze sprawdzamy (brak cooldownu)
return 0.0
elif avg_gap <= 8:
# Tygodniowo → cooldown = 70% cyklu (żeby sprawdzić przed następnym odcinkiem)
cooldown_days = avg_gap * 0.7
else:
# Rzadziej niż tygodniowo → max 7 dni
cooldown_days = 7.0
return time.time() + cooldown_days * 86400
def phase_1_catchup(self, specific_program_slug=None):
self.logger("RNŚ: Phase 1 (Daily Catchup)...")
query = self.db.query(Program).filter_by(station="rns")
if specific_program_slug:
query = query.filter_by(slug=specific_program_slug)
else:
query = query.order_by(Program.last_catchup.asc())
catchup_limit = max(1, self.hard_limit - self.backfill_limit)
skipped = 0
timeout_program = None
for prog in query.all():
if self.requests_made >= catchup_limit:
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Limit Catchup osiągnięty. Zostawiam resztę dla Backfill.")
break
# Pomiń jeśli za wcześnie (adaptive cooldown)
if not specific_program_slug and prog.next_catchup_after and time.time() < prog.next_catchup_after:
skipped += 1
continue
is_first_sync = (prog.last_catchup == 0.0)
page = 1
catchup_succeeded = False
while True:
if self.requests_made >= catchup_limit: break
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Sprawdzam {prog.slug} (strona {page})")
try:
new_found, total_cards, max_page = self.fetch_program_page(prog.slug, page)
except ScraperError as exc:
if exc.code != "NETWORK_TIMEOUT":
raise
if timeout_program:
raise ScraperError(
"rns", "NETWORK_TIMEOUT",
f"Timeout także dla kolejnej audycji {prog.slug}; poprzednia: {timeout_program}",
exc.url,
) from exc
timeout_program = prog.slug
self.logger(f"RNŚ: Pomijam {prog.slug} po trzech timeoutach i sprawdzam następną audycję.")
break
if timeout_program:
warning = f"Poprzednia audycja {timeout_program} miała trzy timeouty; kolejna audycja odpowiada poprawnie."
if self.warning_callback:
self.warning_callback(warning)
else:
self.logger(f"RNŚ: WARNING: {warning}")
timeout_program = None
if max_page > prog.total_pages:
prog.total_pages = max_page
if total_cards == 0 or new_found == 0:
catchup_succeeded = True
break
if is_first_sync:
catchup_succeeded = True
break
if page >= max_page:
catchup_succeeded = True
break
page += 1
time.sleep(0.5)
if catchup_succeeded:
prog.last_catchup = time.time()
prog.next_catchup_after = self._compute_next_catchup(prog)
self.db.commit()
if skipped:
self.logger(f"RNŚ: Pominięto {skipped} audycji (cooldown adaptacyjny).")
def phase_2_backfill(self, specific_program_slug=None):
backfill_count = 0
if specific_program_slug:
prog = self.db.query(Program).filter_by(station="rns", slug=specific_program_slug).first()
if prog:
page = prog.backfill_page
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill manualny dla {prog.slug} (od strony {page})")
while self.requests_made < self.hard_limit and backfill_count < self.backfill_limit:
new_found, total, max_p = self.fetch_program_page(prog.slug, page)
backfill_count += 1
if max_p > prog.total_pages:
prog.total_pages = max_p
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill {prog.slug} strona {page} z {prog.total_pages or '?'} ({new_found} nowych)")
if total == 0 or page >= max_p:
prog.backfill_complete = True
self.logger(f"RNŚ: Backfill {prog.slug} – zakończono.")
break
prog.backfill_page = page + 1
page += 1
self.db.commit()
time.sleep(0.5)
return
while self.requests_made < self.hard_limit and backfill_count < self.backfill_limit:
prog = self.db.query(Program).filter_by(station="rns", backfill_complete=False).order_by(Program.backfill_page.asc()).first()
if not prog:
self.logger("RNŚ: Wszystkie audycje w pełni uzupełnione!")
break
page = prog.backfill_page
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill dla {prog.slug} (strona {page} z {prog.total_pages or '?'})")
new_found, total_cards, max_p = self.fetch_program_page(prog.slug, page)
backfill_count += 1
if max_p > prog.total_pages:
prog.total_pages = max_p
if total_cards == 0 or page >= max_p:
prog.backfill_complete = True
else:
prog.backfill_page = page + 1
self.db.commit()
time.sleep(0.5)
def run_full_sync(self, specific_program_slug=None):
if not self.ensure_auth():
self.logger("RNŚ: Błąd logowania przed startem!")
raise Exception("RNS Login Failed")
self.update_programs()
self.phase_1_catchup(specific_program_slug)
if self.requests_made < self.hard_limit:
self.phase_2_backfill(specific_program_slug)
self.logger(f"RNŚ: Koniec. Wysłano zapytania: {self.requests_made}/{self.hard_limit} (w tym backfill ograniczony do {self.backfill_limit}).")