second-brain/src/second_brain/adapters/youtube.py
Travis Herbranson a2d46a7c6e gauntlet: fix lint — drop unused imports + reorder models.py + add dev deps
Ruff autofix removed pre-existing F401 unused imports across adapters,
context, compiler, extractor, scheduler, web, and main. Also reordered
models.py so its SQLAlchemy imports sit at the top of the file
(E402 was triggered by the inline `utcnow` helper definition).

Added a [dependency-groups] dev block (pytest, pytest-cov, ruff) so the
zero-check test gauntlet — which runs `pytest --cov --cov-report=term-missing`
— can resolve its plugins without a manual `uv pip install pytest-cov`.
2026-05-24 22:52:08 -04:00

232 lines
7.9 KiB
Python

"""
YouTube pull + Whisper transcribe adapter.
Ported from xtract/main.py (VideoDownloader + SubtitleExtractor classes).
Adapted to work with the second-brain Source model and Config.
"""
from __future__ import annotations
from datetime import datetime
from pathlib import Path
from typing import Optional
import whisper
import yt_dlp
from second_brain.config import Config
from second_brain.models import Source, SourceStatus
class VideoDownloader:
"""Download a YouTube video (or playlist) via yt-dlp.
Ported from xtract VideoDownloader — same core yt-dlp options,
adapted to write into config.media_dir.
"""
def __init__(self, config: Config) -> None:
self.config = config
def download(self, url: str) -> list[Path]:
"""Download video(s) and return paths to the downloaded files."""
config = self.config
config.media_dir.mkdir(parents=True, exist_ok=True)
ydl_opts = {
"format": "bestvideo[height<=720]+bestaudio/best[height<=720]",
"outtmpl": str(config.media_dir / "%(id)s.%(ext)s"),
"writeinfojson": True,
"ignoreerrors": True,
"no_warnings": False,
"quiet": False,
}
downloaded: list[Path] = []
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
print(f"[yt-dlp] Fetching: {url}")
info = ydl.extract_info(url, download=True)
if info is None:
print(f"[yt-dlp] Error: could not extract info from {url}")
return downloaded
if "entries" in info:
for entry in info["entries"]:
if entry:
p = self._find_media_file(entry.get("id", ""))
if p:
downloaded.append(p)
else:
p = self._find_media_file(info.get("id", ""))
if p:
downloaded.append(p)
return downloaded
def _find_media_file(self, video_id: str) -> Optional[Path]:
for f in self.config.media_dir.glob(f"{video_id}.*"):
if f.suffix in {".mp4", ".mkv", ".webm", ".m4a"}:
return f
return None
def extract_metadata(self, url: str) -> dict:
"""Fetch video metadata without downloading the media file."""
ydl_opts = {"quiet": True, "skip_download": True}
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(url, download=False) or {}
return {
"title": info.get("title"),
"author": info.get("uploader"),
"duration_seconds": info.get("duration"),
"published_at": _parse_upload_date(info.get("upload_date")),
}
class SubtitleExtractor:
"""Transcribe a video file to an SRT subtitle file using OpenAI Whisper.
Ported from xtract SubtitleExtractor — identical core logic,
writes SRTs to config.subtitles_dir.
"""
def __init__(self, config: Config) -> None:
self.config = config
self._model = None
def _load_model(self) -> None:
if self._model is None:
print(f"[Whisper] Loading {self.config.whisper_model} model…")
self._model = whisper.load_model(self.config.whisper_model)
print("[Whisper] Model ready")
def extract(self, media_file: Path) -> Optional[Path]:
"""Transcribe *media_file* and return the path to the SRT file."""
self.config.subtitles_dir.mkdir(parents=True, exist_ok=True)
srt_file = self.config.subtitles_dir / f"{media_file.stem}.srt"
if srt_file.exists():
print(f"[Whisper] Already transcribed: {srt_file.name}")
return srt_file
self._load_model()
print(f"[Whisper] Transcribing: {media_file.name}")
try:
result = self._model.transcribe(
str(media_file),
verbose=False,
language=None,
)
self._write_srt(result, srt_file)
print(f"[Whisper] Saved: {srt_file.name}")
return srt_file
except Exception as exc:
print(f"[Whisper] Error transcribing {media_file.name}: {exc}")
return None
def extract_text(self, media_file: Path) -> Optional[str]:
"""Return the plain-text transcript (no SRT formatting)."""
self._load_model()
try:
result = self._model.transcribe(str(media_file), verbose=False)
return " ".join(seg["text"].strip() for seg in result["segments"])
except Exception as exc:
print(f"[Whisper] Error: {exc}")
return None
@staticmethod
def _write_srt(result: dict, output_file: Path) -> None:
with open(output_file, "w", encoding="utf-8") as fh:
for i, seg in enumerate(result["segments"], start=1):
start = _fmt_timestamp(seg["start"])
end = _fmt_timestamp(seg["end"])
fh.write(f"{i}\n{start} --> {end}\n{seg['text'].strip()}\n\n")
@staticmethod
def read_srt(srt_file: Path) -> str:
"""Read an SRT file and return its raw text content."""
return srt_file.read_text(encoding="utf-8")
# ---------------------------------------------------------------------------
# High-level adapter
# ---------------------------------------------------------------------------
class YouTubeAdapter:
"""Orchestrates pull + transcribe for a YouTube Source record."""
def __init__(self, config: Config) -> None:
self.config = config
self.downloader = VideoDownloader(config)
self.transcriber = SubtitleExtractor(config)
def pull(self, source: Source) -> bool:
"""Download the video. Updates source in-place, returns success."""
try:
meta = self.downloader.extract_metadata(source.url)
source.title = source.title or meta.get("title")
source.author = meta.get("author")
source.duration_seconds = meta.get("duration_seconds")
source.published_at = meta.get("published_at")
files = self.downloader.download(source.url)
if not files:
source.error_message = "yt-dlp returned no files"
source.status = SourceStatus.FAILED
return False
source.media_path = str(files[0])
source.status = SourceStatus.PULLED
return True
except Exception as exc:
source.error_message = str(exc)
source.status = SourceStatus.FAILED
return False
def transcribe(self, source: Source) -> bool:
"""Transcribe the downloaded video. Updates source in-place."""
if not source.media_path:
source.error_message = "No media_path; run pull first"
return False
media_file = Path(source.media_path)
srt_file = self.transcriber.extract(media_file)
if srt_file is None:
source.error_message = "Whisper transcription failed"
source.status = SourceStatus.FAILED
return False
source.transcript_path = str(srt_file)
source.transcript_text = SubtitleExtractor.read_srt(srt_file)
source.status = SourceStatus.TRANSCRIBED
return True
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _fmt_timestamp(seconds: float) -> str:
h = int(seconds // 3600)
m = int((seconds % 3600) // 60)
s = int(seconds % 60)
ms = int((seconds % 1) * 1000)
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
def _parse_upload_date(upload_date: Optional[str]) -> Optional[datetime]:
"""Parse yt-dlp's YYYYMMDD upload_date string."""
if not upload_date or len(upload_date) != 8:
return None
try:
return datetime.strptime(upload_date, "%Y%m%d")
except ValueError:
return None
__all__ = ["VideoDownloader", "SubtitleExtractor", "YouTubeAdapter"]