Files
radiosync/core/scrapers/jazz.py
T
karol 76f5d30db1 Initial commit
Co-authored-by: GitHub Copilot <noreply@github.com>
2026-09-03 11:57:16 +02:00

58 lines
2.1 KiB
Python

import requests
from bs4 import BeautifulSoup
from sqlalchemy.orm import Session
from ..database import Program
BASE_URL = "https://podkasty.radiojazz.fm"
class JazzScraper:
def __init__(self, db: Session, logger=print):
self.db = db
self.logger = logger
def sync_programs(self):
self.logger("JAZZ: Pobieranie audycji...")
try:
r = requests.get(BASE_URL, headers={"User-Agent": "Mozilla/5.0"}, timeout=15)
r.raise_for_status()
html = r.text
except Exception as e:
self.logger(f"JAZZ: Błąd pobierania bazy: {e}")
return
soup = BeautifulSoup(html, 'html.parser')
links = soup.find_all('a', href=True)
found = 0
for link in links:
href = link['href']
if '/@' in href:
slug = href.split('/@')[-1].split('/')[0]
if not slug: continue
text = link.get_text(strip=True)
if not text or text.startswith('@') or "Recent activity" in text:
continue
clean_title = text
if clean_title.endswith(f"@{slug}"):
clean_title = clean_title[:-len(f"@{slug}")].strip()
if clean_title.startswith("Ż "):
clean_title = clean_title[2:].strip()
elif clean_title.startswith("Ż"):
clean_title = clean_title[1:].strip()
prog = self.db.query(Program).filter_by(station="jazz", slug=slug).first()
if not prog:
self.db.add(Program(
station="jazz", slug=slug, name=clean_title,
description="Radio Jazz FM", image=""
))
found += 1
self.db.commit()
self.logger(f"JAZZ: Znaleziono {found} nowych audycji (łącznie zaktualizowano).")
def run_full_sync(self):
self.sync_programs()