feat: add Goldfish Recap YouTube summary PWA
YouTube subtitles via OpenRouter become persistent, shareable recaps with timestamp jump links, PIN-protected creation, and Traefik deploy. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -0,0 +1,178 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
from youtube_transcript_api import YouTubeTranscriptApi
|
||||
from youtube_transcript_api._errors import (
|
||||
NoTranscriptFound,
|
||||
TranscriptsDisabled,
|
||||
VideoUnavailable,
|
||||
YouTubeRequestFailed,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PREFERRED_LANGUAGES = ("de", "en", "de-DE", "en-US", "en-GB")
|
||||
|
||||
|
||||
@dataclass
|
||||
class TranscriptSegment:
|
||||
start: float
|
||||
text: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class Transcript:
|
||||
language: str
|
||||
segments: list[TranscriptSegment]
|
||||
source: str
|
||||
|
||||
|
||||
def _hash_transcript(segments: list[TranscriptSegment]) -> str:
|
||||
payload = "\n".join(f"{s.start:.2f}:{s.text}" for s in segments)
|
||||
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def transcript_hash(segments: list[TranscriptSegment]) -> str:
|
||||
return _hash_transcript(segments)
|
||||
|
||||
|
||||
def format_transcript_for_prompt(segments: list[TranscriptSegment], *, max_chars: int = 120_000) -> str:
|
||||
lines: list[str] = []
|
||||
total = 0
|
||||
for seg in segments:
|
||||
minutes = int(seg.start // 60)
|
||||
seconds = int(seg.start % 60)
|
||||
line = f"[{minutes:02d}:{seconds:02d}] {seg.text.strip()}"
|
||||
if total + len(line) + 1 > max_chars:
|
||||
lines.append("[... Transcript gekürzt wegen Länge ...]")
|
||||
break
|
||||
lines.append(line)
|
||||
total += len(line) + 1
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _normalize_ytt_segments(raw: list[dict]) -> list[TranscriptSegment]:
|
||||
return [TranscriptSegment(start=float(item["start"]), text=str(item["text"])) for item in raw]
|
||||
|
||||
|
||||
def _fetch_with_youtube_transcript_api(video_id: str) -> Transcript:
|
||||
api = YouTubeTranscriptApi()
|
||||
try:
|
||||
fetched = api.fetch(video_id, languages=list(PREFERRED_LANGUAGES))
|
||||
segments = _normalize_ytt_segments(
|
||||
[{"start": s.start, "text": s.text} for s in fetched]
|
||||
)
|
||||
return Transcript(language=fetched.language_code, segments=segments, source="youtube-transcript-api")
|
||||
except (TranscriptsDisabled, NoTranscriptFound, VideoUnavailable):
|
||||
raise
|
||||
except YouTubeRequestFailed as exc:
|
||||
logger.warning("youtube-transcript-api failed for %s: %s", video_id, exc)
|
||||
raise
|
||||
|
||||
|
||||
def _parse_vtt(content: str) -> list[TranscriptSegment]:
|
||||
segments: list[TranscriptSegment] = []
|
||||
blocks = re.split(r"\n\n+", content.strip())
|
||||
for block in blocks:
|
||||
lines = [ln.strip() for ln in block.splitlines() if ln.strip()]
|
||||
if len(lines) < 2:
|
||||
continue
|
||||
time_line = lines[0] if "-->" in lines[0] else (lines[1] if len(lines) > 1 and "-->" in lines[1] else "")
|
||||
if "-->" not in time_line:
|
||||
continue
|
||||
start_raw = time_line.split("-->")[0].strip()
|
||||
match = re.match(r"(?:(\d+):)?(\d+):(\d+(?:\.\d+)?)", start_raw)
|
||||
if not match:
|
||||
continue
|
||||
hours = int(match.group(1) or 0)
|
||||
minutes = int(match.group(2))
|
||||
seconds = float(match.group(3))
|
||||
start = hours * 3600 + minutes * 60 + seconds
|
||||
text_lines = [ln for ln in lines if "-->" not in ln and not ln.isdigit()]
|
||||
text = " ".join(text_lines).strip()
|
||||
text = re.sub(r"<[^>]+>", "", text)
|
||||
if text:
|
||||
segments.append(TranscriptSegment(start=start, text=text))
|
||||
return segments
|
||||
|
||||
|
||||
def _fetch_with_ytdlp(video_id: str) -> Transcript:
|
||||
import yt_dlp
|
||||
|
||||
url = f"https://www.youtube.com/watch?v={video_id}"
|
||||
opts: dict = {
|
||||
"skip_download": True,
|
||||
"quiet": True,
|
||||
"no_warnings": True,
|
||||
"writesubtitles": True,
|
||||
"writeautomaticsub": True,
|
||||
"subtitleslangs": list(PREFERRED_LANGUAGES),
|
||||
"subtitlesformat": "vtt",
|
||||
}
|
||||
with yt_dlp.YoutubeDL(opts) as ydl:
|
||||
info = ydl.extract_info(url, download=False)
|
||||
|
||||
subtitles = info.get("subtitles") or {}
|
||||
automatic = info.get("automatic_captions") or {}
|
||||
tracks = {**automatic, **subtitles}
|
||||
|
||||
chosen_lang = None
|
||||
for lang in PREFERRED_LANGUAGES:
|
||||
if lang in tracks:
|
||||
chosen_lang = lang
|
||||
break
|
||||
if not chosen_lang:
|
||||
for lang in tracks:
|
||||
chosen_lang = lang
|
||||
break
|
||||
if not chosen_lang:
|
||||
raise ValueError("Keine Untertitel gefunden.")
|
||||
|
||||
formats = tracks[chosen_lang]
|
||||
vtt_url = None
|
||||
for fmt in formats:
|
||||
if fmt.get("ext") == "vtt" or "vtt" in (fmt.get("url") or ""):
|
||||
vtt_url = fmt.get("url")
|
||||
break
|
||||
if not vtt_url and formats:
|
||||
vtt_url = formats[0].get("url")
|
||||
if not vtt_url:
|
||||
raise ValueError("Untertitel-URL nicht verfügbar.")
|
||||
|
||||
import httpx
|
||||
|
||||
resp = httpx.get(vtt_url, timeout=30.0, follow_redirects=True)
|
||||
resp.raise_for_status()
|
||||
segments = _parse_vtt(resp.text)
|
||||
if not segments:
|
||||
raise ValueError("Untertitel konnten nicht geparst werden.")
|
||||
|
||||
return Transcript(language=chosen_lang, segments=segments, source="yt-dlp")
|
||||
|
||||
|
||||
def fetch_transcript(video_id: str) -> Transcript:
|
||||
try:
|
||||
return _fetch_with_youtube_transcript_api(video_id)
|
||||
except (TranscriptsDisabled, NoTranscriptFound):
|
||||
pass
|
||||
except VideoUnavailable as exc:
|
||||
raise ValueError("Video nicht verfügbar.") from exc
|
||||
|
||||
try:
|
||||
return _fetch_with_ytdlp(video_id)
|
||||
except Exception as exc:
|
||||
logger.exception("yt-dlp transcript fallback failed for %s", video_id)
|
||||
raise ValueError(
|
||||
"Dieses Video hat keine Captions. Dein Goldfisch kann leider nicht ins Leere starren."
|
||||
) from exc
|
||||
|
||||
|
||||
def estimate_duration_sec(segments: list[TranscriptSegment]) -> int | None:
|
||||
if not segments:
|
||||
return None
|
||||
last = segments[-1]
|
||||
return int(last.start) + 30
|
||||
Reference in New Issue
Block a user