76f5d30db1
Co-authored-by: GitHub Copilot <noreply@github.com>
509 lines
23 KiB
Python
509 lines
23 KiB
Python
import requests
|
||
import time
|
||
import os
|
||
import re
|
||
import urllib.parse
|
||
from bs4 import BeautifulSoup
|
||
from sqlalchemy.orm import Session
|
||
from ..database import Program, Episode, Config
|
||
from ..diagnostics import ScraperError, invalid_response
|
||
|
||
MAX_REQUESTS_LIMIT = 50
|
||
BASE_URL = "https://nowyswiat.online"
|
||
|
||
|
||
def get_cookie_for_url(cookie_jar, name, url):
|
||
"""Wybiera najbardziej szczegółowe cookie, gdy jar zawiera duplikaty nazwy."""
|
||
parsed_url = urllib.parse.urlparse(url)
|
||
host = parsed_url.hostname.lower()
|
||
path = parsed_url.path or "/"
|
||
candidates = []
|
||
for cookie in cookie_jar:
|
||
if cookie.name != name or (cookie.secure and parsed_url.scheme != "https"):
|
||
continue
|
||
domain = cookie.domain.lstrip(".").lower()
|
||
if domain and not (host == domain or host.endswith(f".{domain}")):
|
||
continue
|
||
cookie_path = cookie.path or "/"
|
||
if not path.startswith(cookie_path.rstrip("/") or "/"):
|
||
continue
|
||
candidates.append(cookie)
|
||
|
||
if not candidates:
|
||
return None
|
||
selected = max(candidates, key=lambda cookie: (len(cookie.path or "/"), len(cookie.domain or "")))
|
||
for cookie in candidates:
|
||
if cookie is not selected:
|
||
cookie_jar.clear(cookie.domain, cookie.path, cookie.name)
|
||
return selected.value
|
||
|
||
def parse_polish_date(date_str):
|
||
months = {
|
||
"stycznia": "01", "lutego": "02", "marca": "03", "kwietnia": "04",
|
||
"maja": "05", "czerwca": "06", "lipca": "07", "sierpnia": "08",
|
||
"września": "09", "października": "10", "listopada": "11", "grudnia": "12",
|
||
"styczeń": "01", "luty": "02", "marzec": "03", "kwiecień": "04",
|
||
"maj": "05", "czerwiec": "06", "lipiec": "07", "sierpień": "08",
|
||
"wrzesień": "09", "październik": "10", "listopad": "11", "grudzień": "12"
|
||
}
|
||
parts = date_str.lower().split()
|
||
if len(parts) == 3:
|
||
day = parts[0].zfill(2)
|
||
month = months.get(parts[1], "01")
|
||
year = parts[2]
|
||
return f"{year}-{month}-{day}"
|
||
return date_str
|
||
|
||
class RNScraper:
|
||
def __init__(self, db: Session, logger=print, progress_callback=None, warning_callback=None, stop_flag=None):
|
||
self.db = db
|
||
self.logger = logger
|
||
self.progress_callback = progress_callback
|
||
self.warning_callback = warning_callback
|
||
self.stop_flag = stop_flag
|
||
self.requests_made = 0
|
||
self.start_time = time.time()
|
||
self.max_execution_time = 900 # 15 minut max na cały sync
|
||
limit_conf = self.db.query(Config).filter_by(key="rns_limit").first()
|
||
self.backfill_limit = int(limit_conf.value) if limit_conf and limit_conf.value.isdigit() else 50
|
||
hard_limit_conf = self.db.query(Config).filter_by(key="rns_hard_limit").first()
|
||
self.hard_limit = int(hard_limit_conf.value) if hard_limit_conf and hard_limit_conf.value.isdigit() else 500
|
||
self.session = requests.Session()
|
||
self.session.headers.update({"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"})
|
||
if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit)
|
||
|
||
def check_timeout(self):
|
||
if self.stop_flag and self.stop_flag():
|
||
raise Exception("Zatrzymano na żądanie użytkownika.")
|
||
if time.time() - self.start_time > self.max_execution_time:
|
||
raise Exception(f"Przekroczono limit czasu wykonywania skryptu ({self.max_execution_time // 60} min). Zatrzymano awaryjnie.")
|
||
|
||
def _get_html(self, url):
|
||
self.check_timeout()
|
||
if self.requests_made >= self.hard_limit:
|
||
return None
|
||
for attempt in range(1, 4):
|
||
if self.requests_made >= self.hard_limit:
|
||
return None
|
||
self.requests_made += 1
|
||
if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit)
|
||
try:
|
||
r = self.session.get(url, timeout=15)
|
||
if r.status_code == 200:
|
||
return r.text
|
||
invalid_response("rns", "HTTP_ERROR", "Błąd pobierania strony", url, r)
|
||
except requests.exceptions.Timeout as e:
|
||
if attempt == 3:
|
||
raise ScraperError("rns", "NETWORK_TIMEOUT", "Trzy próby pobrania strony zakończyły się timeoutem", url) from e
|
||
self.logger(f"RNŚ: Timeout dla {url}, próba {attempt}/3. Ponawiam za 30 sekund.")
|
||
self.check_timeout()
|
||
time.sleep(30)
|
||
except Exception as e:
|
||
if isinstance(e, ScraperError):
|
||
raise
|
||
raise ScraperError("rns", "NETWORK_ERROR", f"Błąd zapytania: {e}", url) from e
|
||
return None
|
||
|
||
def perform_login(self):
|
||
email = self.db.query(Config).filter_by(key="rns_email").first()
|
||
password = self.db.query(Config).filter_by(key="rns_password").first()
|
||
if not email or not password:
|
||
self.logger("RNŚ: Brak danych logowania w bazie.")
|
||
return False
|
||
|
||
self.logger("RNŚ: Inicjalizacja logowania (pobieranie CSRF)...")
|
||
try:
|
||
r1 = self.session.get("https://nowyswiat.online/konto/zaloguj", timeout=15)
|
||
except Exception as e:
|
||
self.logger(f"RNŚ: Sieć zablokowała pobieranie CSRF: {e}")
|
||
return False
|
||
|
||
csrf_token = get_cookie_for_url(
|
||
self.session.cookies,
|
||
"csrf_cookie_neocms",
|
||
"https://nowyswiat.online/konto/zaloguj",
|
||
)
|
||
if not csrf_token:
|
||
self.logger("RNŚ: Nie udało się pobrać tokenu CSRF.")
|
||
return False
|
||
|
||
self.logger("RNŚ: Wysyłanie formularza...")
|
||
payload = {"csrf_neocms": csrf_token, "login": email.value, "password": password.value, "ufd_data": "{}"}
|
||
try:
|
||
r2 = self.session.post("https://nowyswiat.online/konto/zaloguj", data=payload, headers={
|
||
"X-Requested-With": "XMLHttpRequest"
|
||
}, timeout=15)
|
||
except Exception as e:
|
||
self.logger(f"RNŚ: Błąd sieci przy wysyłaniu formularza: {e}")
|
||
return False
|
||
|
||
# Check if login succeeded by looking for a session cookie or a success response
|
||
if '"status":"OK"' in r2.text or "logowanie udane" in r2.text.lower():
|
||
# Save cookies to DB (serialize)
|
||
cookies_dict = requests.utils.dict_from_cookiejar(self.session.cookies)
|
||
import json
|
||
cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first()
|
||
if not cookie_conf:
|
||
self.db.add(Config(key="rns_cookies", value=json.dumps(cookies_dict)))
|
||
else:
|
||
cookie_conf.value = json.dumps(cookies_dict)
|
||
self.db.commit()
|
||
return True
|
||
else:
|
||
self.logger(f"RNŚ: Błędne dane logowania (lub zmiana mechanizmu). Odpowiedź: {r2.text[:100]}")
|
||
return False
|
||
|
||
def ensure_auth(self):
|
||
cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first()
|
||
if cookie_conf:
|
||
import json
|
||
try:
|
||
cookies_dict = json.loads(cookie_conf.value)
|
||
self.session.cookies = requests.utils.cookiejar_from_dict(cookies_dict)
|
||
html = self._get_html(f"{BASE_URL}/")
|
||
if html and "wyloguj" in html.lower():
|
||
return True
|
||
self.logger("RNŚ: Zapisana sesja wygasła, wykonuję ponowne logowanie.")
|
||
except:
|
||
pass
|
||
return self.perform_login()
|
||
|
||
def check_auth_status(self):
|
||
cookie_conf = self.db.query(Config).filter_by(key="rns_cookies").first()
|
||
if not cookie_conf:
|
||
return False, "Brak zapisanych ciasteczek."
|
||
import json
|
||
try:
|
||
self.session.cookies = requests.utils.cookiejar_from_dict(json.loads(cookie_conf.value))
|
||
html = self._get_html("https://nowyswiat.online/")
|
||
if html and "wyloguj" in html.lower():
|
||
return True, "Ciasteczka aktywne, sesja poprawna."
|
||
return False, "Ciasteczka nieaktywne lub wygasły (brak dostępu do profilu)."
|
||
except Exception as e:
|
||
return False, f"Błąd: {e}"
|
||
|
||
def update_programs(self):
|
||
self.logger("RNŚ: Aktualizacja listy programów...")
|
||
html = self._get_html("https://nowyswiat.online/podcasty")
|
||
if not html: return
|
||
soup = BeautifulSoup(html, "html.parser")
|
||
links = soup.find_all("a", href=True)
|
||
program_links = [link for link in links if "rbroadcast=" in link["href"]]
|
||
if not program_links:
|
||
invalid_response("rns", "HTML_SCHEMA_CHANGED", "Nie znaleziono programów w stronie podcastów", "https://nowyswiat.online/podcasty", content=html)
|
||
for link in program_links:
|
||
href = link["href"]
|
||
if "rbroadcast=" in href:
|
||
parsed = urllib.parse.urlparse(href)
|
||
slug = urllib.parse.parse_qs(parsed.query).get("rbroadcast", [None])[0]
|
||
if not slug: continue
|
||
|
||
title_el = link.find("h2", class_="rns-search-dropdown-title")
|
||
title = title_el.text.strip() if title_el else slug
|
||
|
||
img_el = link.find("img")
|
||
img_url = img_el["src"] if img_el and img_el.has_attr("src") else ""
|
||
if img_url and not img_url.startswith("http"): img_url = f"{BASE_URL}/{img_url.lstrip('/')}"
|
||
|
||
prog = self.db.query(Program).filter_by(station="rns", slug=slug).first()
|
||
if not prog:
|
||
self.db.add(Program(
|
||
station="rns", slug=slug, name=title, image=img_url,
|
||
description=f"Radio Nowy Świat: {title}", backfill_page=2, backfill_complete=False
|
||
))
|
||
self.db.commit()
|
||
|
||
def _verify_audio_teaser(self, ep: Episode):
|
||
"""Wykrywa 1-minutowy teaser (rozmiar mniejszy niż ~2MB, mimo że audycja trwa > 5 min)."""
|
||
self.check_timeout()
|
||
if not ep.url: return False
|
||
|
||
if self.requests_made >= self.hard_limit: return False
|
||
self.requests_made += 1
|
||
if self.progress_callback: self.progress_callback(self.requests_made, self.hard_limit)
|
||
|
||
try:
|
||
r = self.session.head(ep.url, timeout=5)
|
||
cl = int(r.headers.get("Content-Length", 0))
|
||
if cl > 0 and cl < 2_500_000 and ep.duration_secs > 300:
|
||
self.logger(f"RNŚ: Wykryto uszkodzony link (Teaser 1-min) dla {ep.title}. Usuwam URL.")
|
||
ep.url = None
|
||
ep.is_broken = True
|
||
return True
|
||
except Exception as e:
|
||
pass
|
||
return False
|
||
|
||
def fetch_program_page(self, program_slug, page):
|
||
url = f"https://nowyswiat.online/podcasty?rbroadcast={program_slug}&page={page}"
|
||
html = self._get_html(url)
|
||
if not html:
|
||
raise Exception(f"Błąd sieci podczas pobierania strony {page} audycji {program_slug}")
|
||
|
||
soup = BeautifulSoup(html, "html.parser")
|
||
cards = soup.find_all("a", class_="rns-grid-podcast-card")
|
||
known_episode_count = self.db.query(Episode).filter_by(
|
||
station="rns", program_slug=program_slug
|
||
).count()
|
||
page_has_podcast_content = bool(soup.find(string=re.compile(r"podcast|podkast", re.IGNORECASE)))
|
||
if not cards and page > 1:
|
||
return 0, 0, page - 1
|
||
if not cards and (known_episode_count or not page_has_podcast_content):
|
||
invalid_response("rns", "HTML_SCHEMA_CHANGED", "Nie znaleziono kart odcinków dla istniejącego programu", url, content=html)
|
||
|
||
new_found = 0
|
||
missing_player_count = 0
|
||
valid_cards = 0
|
||
|
||
for card in cards:
|
||
href = card.get("href", "")
|
||
if not href or "/podcasty/" not in href: continue
|
||
valid_cards += 1
|
||
ep_id = href.split("/podcasty/")[-1].split("?")[0]
|
||
|
||
player_box = card.find("div", class_="rns-play-btn-box")
|
||
audio_url = player_box.get("data-neo-player-src") if player_box else None
|
||
|
||
if not player_box:
|
||
missing_player_count += 1
|
||
|
||
if audio_url and not audio_url.startswith("http"):
|
||
audio_url = f"{BASE_URL}/{audio_url.lstrip('/')}"
|
||
|
||
title_el = card.find("p", class_="rns-post-title")
|
||
raw_title = player_box.get("data-neo-player-title", "") if player_box else (title_el.text.strip() if title_el else ep_id)
|
||
title = BeautifulSoup(raw_title, "html.parser").text.strip() if raw_title else ep_id
|
||
|
||
raw_subtitle = player_box.get("data-neo-player-subtitle", "") if player_box else ""
|
||
authors = BeautifulSoup(raw_subtitle, "html.parser").text.strip() if raw_subtitle else "Radio Nowy Świat"
|
||
|
||
date_el = card.find("p", class_="rns-podcast-details-date")
|
||
pub_date = parse_polish_date(date_el.text.strip()) if date_el else ""
|
||
|
||
time_el = card.find("p", class_="rns-podcast-details-long")
|
||
duration_str = time_el.text.strip() if time_el else "0:00"
|
||
duration_secs = 0
|
||
if ":" in duration_str:
|
||
parts = duration_str.split(":")
|
||
if len(parts) == 3: duration_secs = int(parts[0])*3600 + int(parts[1])*60 + int(parts[2])
|
||
elif len(parts) == 2: duration_secs = int(parts[0])*60 + int(parts[1])
|
||
|
||
img_url = player_box.get("data-neo-player-img", "") if player_box else ""
|
||
if img_url and not img_url.startswith("http"): img_url = f"{BASE_URL}/{img_url.lstrip('/')}"
|
||
|
||
desc_el = card.find("p", class_="rns-post-card-desc")
|
||
description = desc_el.get_text(separator="\n").strip() if desc_el else ""
|
||
|
||
ep = self.db.query(Episode).filter_by(station="rns", ep_id=ep_id).first()
|
||
if not ep:
|
||
ep = Episode(
|
||
station="rns", program_slug=program_slug, ep_id=ep_id,
|
||
title=title, authors=authors, url=audio_url, image=img_url,
|
||
pub_date=pub_date, duration_secs=duration_secs, description=description,
|
||
is_broken=False
|
||
)
|
||
self.db.add(ep)
|
||
new_found += 1
|
||
else:
|
||
if not ep.url and audio_url:
|
||
ep.url = audio_url
|
||
ep.is_broken = False
|
||
# Celowo nie zwiększamy new_found, aby Daily Catchup nie wchodził w tryb głębokiego archiwum.
|
||
# Łataniem starych dziur zajmie się faza Backfill.
|
||
|
||
if cards and not valid_cards:
|
||
invalid_response("rns", "HTML_SCHEMA_CHANGED", "Znaleziono karty podcastów bez oczekiwanych identyfikatorów", url, content=html)
|
||
|
||
if missing_player_count:
|
||
warning = f"{missing_player_count} odcinków bez dostępnego playera audio; pomijam ich URL-e."
|
||
if self.warning_callback:
|
||
self.warning_callback(warning)
|
||
else:
|
||
self.logger(f"RNŚ: WARNING: {warning}")
|
||
|
||
self.db.commit()
|
||
|
||
page_links = soup.find_all("a", href=re.compile(r"page=\d+"))
|
||
max_page = page
|
||
for link in page_links:
|
||
href = link.get("href")
|
||
m = re.search(r"page=(\d+)", href)
|
||
if m:
|
||
max_page = max(max_page, int(m.group(1)))
|
||
|
||
return new_found, len(cards), max_page
|
||
|
||
def _compute_next_catchup(self, prog: Program) -> float:
|
||
"""Wylicza kiedy najwcześniej warto znowu sprawdzać tę audycję."""
|
||
episodes = self.db.query(Episode).filter_by(
|
||
station="rns", program_slug=prog.slug
|
||
).order_by(Episode.pub_date.desc()).limit(20).all()
|
||
|
||
if len(episodes) < 3:
|
||
# Za mało danych – sprawdzamy przy każdym uruchomieniu
|
||
return 0.0
|
||
|
||
# Wylicz średnią przerwę między odcinkami w dniach
|
||
import re
|
||
dates = []
|
||
for ep in episodes:
|
||
if ep.pub_date and re.match(r'\d{4}-\d{2}-\d{2}', ep.pub_date):
|
||
try:
|
||
import datetime
|
||
dates.append(datetime.date.fromisoformat(ep.pub_date[:10]))
|
||
except ValueError:
|
||
pass
|
||
|
||
if len(dates) < 3:
|
||
return 0.0
|
||
|
||
dates.sort(reverse=True)
|
||
gaps = [(dates[i] - dates[i+1]).days for i in range(len(dates)-1)]
|
||
avg_gap = sum(gaps) / len(gaps)
|
||
|
||
if avg_gap < 2:
|
||
# Codziennie lub częściej → zawsze sprawdzamy (brak cooldownu)
|
||
return 0.0
|
||
elif avg_gap <= 8:
|
||
# Tygodniowo → cooldown = 70% cyklu (żeby sprawdzić przed następnym odcinkiem)
|
||
cooldown_days = avg_gap * 0.7
|
||
else:
|
||
# Rzadziej niż tygodniowo → max 7 dni
|
||
cooldown_days = 7.0
|
||
|
||
return time.time() + cooldown_days * 86400
|
||
|
||
def phase_1_catchup(self, specific_program_slug=None):
|
||
self.logger("RNŚ: Phase 1 (Daily Catchup)...")
|
||
query = self.db.query(Program).filter_by(station="rns")
|
||
if specific_program_slug:
|
||
query = query.filter_by(slug=specific_program_slug)
|
||
else:
|
||
query = query.order_by(Program.last_catchup.asc())
|
||
|
||
catchup_limit = max(1, self.hard_limit - self.backfill_limit)
|
||
skipped = 0
|
||
|
||
timeout_program = None
|
||
for prog in query.all():
|
||
if self.requests_made >= catchup_limit:
|
||
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Limit Catchup osiągnięty. Zostawiam resztę dla Backfill.")
|
||
break
|
||
|
||
# Pomiń jeśli za wcześnie (adaptive cooldown)
|
||
if not specific_program_slug and prog.next_catchup_after and time.time() < prog.next_catchup_after:
|
||
skipped += 1
|
||
continue
|
||
|
||
is_first_sync = (prog.last_catchup == 0.0)
|
||
|
||
page = 1
|
||
catchup_succeeded = False
|
||
while True:
|
||
if self.requests_made >= catchup_limit: break
|
||
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Sprawdzam {prog.slug} (strona {page})")
|
||
try:
|
||
new_found, total_cards, max_page = self.fetch_program_page(prog.slug, page)
|
||
except ScraperError as exc:
|
||
if exc.code != "NETWORK_TIMEOUT":
|
||
raise
|
||
if timeout_program:
|
||
raise ScraperError(
|
||
"rns", "NETWORK_TIMEOUT",
|
||
f"Timeout także dla kolejnej audycji {prog.slug}; poprzednia: {timeout_program}",
|
||
exc.url,
|
||
) from exc
|
||
timeout_program = prog.slug
|
||
self.logger(f"RNŚ: Pomijam {prog.slug} po trzech timeoutach i sprawdzam następną audycję.")
|
||
break
|
||
|
||
if timeout_program:
|
||
warning = f"Poprzednia audycja {timeout_program} miała trzy timeouty; kolejna audycja odpowiada poprawnie."
|
||
if self.warning_callback:
|
||
self.warning_callback(warning)
|
||
else:
|
||
self.logger(f"RNŚ: WARNING: {warning}")
|
||
timeout_program = None
|
||
|
||
if max_page > prog.total_pages:
|
||
prog.total_pages = max_page
|
||
|
||
if total_cards == 0 or new_found == 0:
|
||
catchup_succeeded = True
|
||
break
|
||
|
||
if is_first_sync:
|
||
catchup_succeeded = True
|
||
break
|
||
|
||
if page >= max_page:
|
||
catchup_succeeded = True
|
||
break
|
||
|
||
page += 1
|
||
time.sleep(0.5)
|
||
|
||
if catchup_succeeded:
|
||
prog.last_catchup = time.time()
|
||
prog.next_catchup_after = self._compute_next_catchup(prog)
|
||
self.db.commit()
|
||
|
||
if skipped:
|
||
self.logger(f"RNŚ: Pominięto {skipped} audycji (cooldown adaptacyjny).")
|
||
|
||
def phase_2_backfill(self, specific_program_slug=None):
|
||
backfill_count = 0
|
||
if specific_program_slug:
|
||
prog = self.db.query(Program).filter_by(station="rns", slug=specific_program_slug).first()
|
||
if prog:
|
||
page = prog.backfill_page
|
||
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill manualny dla {prog.slug} (od strony {page})")
|
||
while self.requests_made < self.hard_limit and backfill_count < self.backfill_limit:
|
||
new_found, total, max_p = self.fetch_program_page(prog.slug, page)
|
||
backfill_count += 1
|
||
if max_p > prog.total_pages:
|
||
prog.total_pages = max_p
|
||
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill {prog.slug} strona {page} z {prog.total_pages or '?'} ({new_found} nowych)")
|
||
if total == 0 or page >= max_p:
|
||
prog.backfill_complete = True
|
||
self.logger(f"RNŚ: Backfill {prog.slug} – zakończono.")
|
||
break
|
||
prog.backfill_page = page + 1
|
||
page += 1
|
||
self.db.commit()
|
||
time.sleep(0.5)
|
||
return
|
||
|
||
while self.requests_made < self.hard_limit and backfill_count < self.backfill_limit:
|
||
prog = self.db.query(Program).filter_by(station="rns", backfill_complete=False).order_by(Program.backfill_page.asc()).first()
|
||
if not prog:
|
||
self.logger("RNŚ: Wszystkie audycje w pełni uzupełnione!")
|
||
break
|
||
|
||
page = prog.backfill_page
|
||
self.logger(f"RNŚ: [{self.requests_made}/{self.hard_limit}] Backfill dla {prog.slug} (strona {page} z {prog.total_pages or '?'})")
|
||
new_found, total_cards, max_p = self.fetch_program_page(prog.slug, page)
|
||
backfill_count += 1
|
||
|
||
if max_p > prog.total_pages:
|
||
prog.total_pages = max_p
|
||
|
||
if total_cards == 0 or page >= max_p:
|
||
prog.backfill_complete = True
|
||
else:
|
||
prog.backfill_page = page + 1
|
||
self.db.commit()
|
||
time.sleep(0.5)
|
||
|
||
def run_full_sync(self, specific_program_slug=None):
|
||
if not self.ensure_auth():
|
||
self.logger("RNŚ: Błąd logowania przed startem!")
|
||
raise Exception("RNS Login Failed")
|
||
|
||
self.update_programs()
|
||
self.phase_1_catchup(specific_program_slug)
|
||
if self.requests_made < self.hard_limit:
|
||
self.phase_2_backfill(specific_program_slug)
|
||
|
||
self.logger(f"RNŚ: Koniec. Wysłano zapytania: {self.requests_made}/{self.hard_limit} (w tym backfill ograniczony do {self.backfill_limit}).")
|