@@ -0,0 +1,57 @@
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
from sqlalchemy.orm import Session
|
||||
from ..database import Program
|
||||
|
||||
BASE_URL = "https://podkasty.radiojazz.fm"
|
||||
|
||||
class JazzScraper:
|
||||
def __init__(self, db: Session, logger=print):
|
||||
self.db = db
|
||||
self.logger = logger
|
||||
|
||||
def sync_programs(self):
|
||||
self.logger("JAZZ: Pobieranie audycji...")
|
||||
try:
|
||||
r = requests.get(BASE_URL, headers={"User-Agent": "Mozilla/5.0"}, timeout=15)
|
||||
r.raise_for_status()
|
||||
html = r.text
|
||||
except Exception as e:
|
||||
self.logger(f"JAZZ: Błąd pobierania bazy: {e}")
|
||||
return
|
||||
|
||||
soup = BeautifulSoup(html, 'html.parser')
|
||||
links = soup.find_all('a', href=True)
|
||||
|
||||
found = 0
|
||||
for link in links:
|
||||
href = link['href']
|
||||
if '/@' in href:
|
||||
slug = href.split('/@')[-1].split('/')[0]
|
||||
if not slug: continue
|
||||
|
||||
text = link.get_text(strip=True)
|
||||
if not text or text.startswith('@') or "Recent activity" in text:
|
||||
continue
|
||||
|
||||
clean_title = text
|
||||
if clean_title.endswith(f"@{slug}"):
|
||||
clean_title = clean_title[:-len(f"@{slug}")].strip()
|
||||
if clean_title.startswith("Ż "):
|
||||
clean_title = clean_title[2:].strip()
|
||||
elif clean_title.startswith("Ż"):
|
||||
clean_title = clean_title[1:].strip()
|
||||
|
||||
prog = self.db.query(Program).filter_by(station="jazz", slug=slug).first()
|
||||
if not prog:
|
||||
self.db.add(Program(
|
||||
station="jazz", slug=slug, name=clean_title,
|
||||
description="Radio Jazz FM", image=""
|
||||
))
|
||||
found += 1
|
||||
|
||||
self.db.commit()
|
||||
self.logger(f"JAZZ: Znaleziono {found} nowych audycji (łącznie zaktualizowano).")
|
||||
|
||||
def run_full_sync(self):
|
||||
self.sync_programs()
|
||||
Reference in New Issue
Block a user