76f5d30db1
Co-authored-by: GitHub Copilot <noreply@github.com>
58 lines
2.1 KiB
Python
58 lines
2.1 KiB
Python
import requests
|
|
from bs4 import BeautifulSoup
|
|
from sqlalchemy.orm import Session
|
|
from ..database import Program
|
|
|
|
BASE_URL = "https://podkasty.radiojazz.fm"
|
|
|
|
class JazzScraper:
|
|
def __init__(self, db: Session, logger=print):
|
|
self.db = db
|
|
self.logger = logger
|
|
|
|
def sync_programs(self):
|
|
self.logger("JAZZ: Pobieranie audycji...")
|
|
try:
|
|
r = requests.get(BASE_URL, headers={"User-Agent": "Mozilla/5.0"}, timeout=15)
|
|
r.raise_for_status()
|
|
html = r.text
|
|
except Exception as e:
|
|
self.logger(f"JAZZ: Błąd pobierania bazy: {e}")
|
|
return
|
|
|
|
soup = BeautifulSoup(html, 'html.parser')
|
|
links = soup.find_all('a', href=True)
|
|
|
|
found = 0
|
|
for link in links:
|
|
href = link['href']
|
|
if '/@' in href:
|
|
slug = href.split('/@')[-1].split('/')[0]
|
|
if not slug: continue
|
|
|
|
text = link.get_text(strip=True)
|
|
if not text or text.startswith('@') or "Recent activity" in text:
|
|
continue
|
|
|
|
clean_title = text
|
|
if clean_title.endswith(f"@{slug}"):
|
|
clean_title = clean_title[:-len(f"@{slug}")].strip()
|
|
if clean_title.startswith("Ż "):
|
|
clean_title = clean_title[2:].strip()
|
|
elif clean_title.startswith("Ż"):
|
|
clean_title = clean_title[1:].strip()
|
|
|
|
prog = self.db.query(Program).filter_by(station="jazz", slug=slug).first()
|
|
if not prog:
|
|
self.db.add(Program(
|
|
station="jazz", slug=slug, name=clean_title,
|
|
description="Radio Jazz FM", image=""
|
|
))
|
|
found += 1
|
|
|
|
self.db.commit()
|
|
self.logger(f"JAZZ: Znaleziono {found} nowych audycji (łącznie zaktualizowano).")
|
|
|
|
def run_full_sync(self):
|
|
self.sync_programs()
|