From 3ba58b0217dae92749a338d653dd5273b9c575bc Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 24 Aug 2026 15:01:53 +0000 Subject: [PATCH] Retry the downloads that drop, and remember the ones that fail MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A backfill of thirty videos rarely gets through without YouTube hanging up part-way through one of them: [download] Got error: ('Connection aborted.', ConnectionResetError(10054, '远程主机强迫关闭了一个现有的连接。', None, 10054, None)) That is the network, not the video, and it was costing the whole video: the run reported a failure and moved on. Two changes, one for each timescale. During a run, a request that fails on something that looks like network trouble is made again after retry_backoff seconds, doubling for each of download_retries extra attempts. yt-dlp resumes from the .part file, so a drop at 63% costs the pause rather than the 63%. Transient failures are told apart from refusals by walking the exception chain for connection and timeout types and their messages, so the check survives a localised OS string; a members-only or deleted video still fails on the first try, as before. Between runs, a video that still fails goes into the state file under "failures" with the error, an attempt count and enough metadata to fetch it again. This is what a backfill needed: a plain run only looks at the newest check_limit videos, so a video that failed during the initial thirty was out of the window by the next run and would never have been seen again. `run --retry-failed` puts that list back in front of the queue whatever its age, `run --only-failed` does those and nothing else without listing the channel, and retry_failed = true makes every run do it. A video that keeps failing is picked up retry_max_attempts times and then left alone, which --only-failed overrides and `ytscript failures --clear` resets. `ytscript failures` shows the list; a video that succeeds drops off it. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_0116BFtHjqyoP9CQoe7mVcym --- README.md | 81 +++++++++++++++++++++++- src/ytscript/cli.py | 87 ++++++++++++++++++++++++++ src/ytscript/config.py | 43 +++++++++++++ src/ytscript/models.py | 6 ++ src/ytscript/pipeline.py | 80 +++++++++++++++++++++++- src/ytscript/state.py | 58 +++++++++++++++++- src/ytscript/youtube.py | 129 +++++++++++++++++++++++++++++++++++++-- tests/test_cli.py | 77 +++++++++++++++++++++++ tests/test_config.py | 12 ++++ tests/test_pipeline.py | 109 +++++++++++++++++++++++++++++++++ tests/test_state.py | 49 +++++++++++++++ tests/test_youtube.py | 109 +++++++++++++++++++++++++++++++++ 12 files changed, 830 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index 9b292b8..7c1b1e5 100644 --- a/README.md +++ b/README.md @@ -80,12 +80,18 @@ ytscript run # first run: the latest 30 videos ytscript run # later runs: only what is new ytscript polish scripts # re-clean scripts already written + +ytscript failures # what an earlier run could not finish +ytscript run --only-failed # take another run at exactly those ``` Scripts land in `output_dir` as `2024-05-01_Video-title_VIDEOID.txt`, and every finished video is recorded in the state file, so re-running is cheap and safe. State is written after each video, so an interrupted backfill resumes where it stopped. A video that -fails is not recorded and is retried on the next run. +fails is not recorded as done — it goes on the state file's failure list instead, so +`ytscript failures` can show it and `--retry-failed` can pick it back up long after it +has scrolled out of the `check_limit` window. See [When a download +drops](#when-a-download-drops). Useful flags on `run`: @@ -100,6 +106,9 @@ Useful flags on `run`: | `--format txt,md,json` | Write more than one rendering | | `--timestamps` | Prefix each paragraph with `[hh:mm:ss]` | | `--dry-run` | List what is missing without downloading anything | +| `--retry-failed` | Also re-attempt videos an earlier run could not finish | +| `--only-failed` | Do just those, skipping the channel listing and the attempt limit | +| `--retries N` | Extra attempts a download gets when the connection drops (default 3) | | `--keep-audio` | Keep the downloaded audio next to the scripts | | `--drive` / `--no-drive` | Also copy each script into Google Drive, or not | | `--drive-folder ID` | The Drive folder they go into | @@ -110,6 +119,10 @@ Useful flags on `run`: The cookie and members-only flags work on `list` too. +`ytscript failures` prints the failure list — id, attempts, when, title and the error — +and `ytscript failures --clear [ID ...]` forgets entries, which also resets their +attempt count. + `ytscript polish` runs the same clean-up a run does over scripts that already exist — useful after adding a term to the vocabulary, or on a backlog transcribed before it had one. It takes files or directories (`.txt` and `.md`), rewrites them in place, and has @@ -128,6 +141,12 @@ language = "zh" # main spoken language, ISO 639-1; "auto" to detect initial_backfill = 30 # videos transcribed on the very first run check_limit = 5 # videos inspected on later runs +download_retries = 3 # extra attempts when the connection drops; 0 means one try +retry_backoff = 5.0 # seconds before the second attempt, doubling after that +socket_timeout = 30.0 # seconds a stalled connection gets before it counts as failed +retry_failed = false # every run also re-attempts what the failure list holds +retry_max_attempts = 3 # times a failed video is picked up again before it is left alone + backend = "faster-whisper" # or "openai" whisper_model = "large-v3" # tiny | base | small | medium | large-v3 | distil-large-v3 whisper_device = "cuda" # "cpu", "cuda", ... @@ -168,6 +187,62 @@ include_members_only = false # true also transcribes members-only video Every key has a matching environment variable: `YTSCRIPT_CHANNEL`, `YTSCRIPT_LANGUAGE`, `YTSCRIPT_BACKEND`, and so on. +### When a download drops + +YouTube hangs up part-way through a download often enough that a backfill of thirty +videos rarely gets through untouched: + +``` +[download] 63.2% of 48.19MiB at 1.02MiB/s ETA 00:17 +[download] Got error: ('Connection aborted.', ConnectionResetError(10054, +'远程主机强迫关闭了一个现有的连接。', None, 10054, None)) +``` + +That is the network, not the video — the same URL usually works seconds later. ytscript +handles it in two places. + +**During the run.** A request that fails on something that looks like network trouble is +made again, waiting `retry_backoff` seconds, then twice that, then twice that again, for +`download_retries` extra attempts. yt-dlp resumes from the `.part` file it already has, +so a drop at 63% costs the pause, not the 63%. Refusals are told apart from drops and +are not retried: a members-only video or a deleted one fails immediately, as it should. +`--retries 6` widens it for a bad line, `--retries 0` turns it off. + +**Between runs.** A video that still fails — its retries used up, the backend out of +memory, Drive refusing the upload — is written to the state file under `failures`, with +the error, the attempt count and enough about the video to fetch it again: + +```bash +ytscript failures +# 1 video(s) failed; 'ytscript run --retry-failed' tries them again: +# dQw4w9WgXcQ 2 attempt(s) 2024-05-02T09:14:31+00:00 Market wrap, May 1 +# could not download audio for dQw4w9WgXcQ: ('Connection aborted.', ...) +``` + +This matters for a backfill. A plain run only looks at the newest `check_limit` videos, +so a video that failed while thirty were being transcribed is out of the window by the +next run and would never be seen again. `--retry-failed` puts the failure list back in +front of the queue whatever its age, and `--only-failed` does those and nothing else, +skipping the channel listing entirely: + +```bash +ytscript run --retry-failed # the newest few, plus everything that failed before +ytscript run --only-failed # just the failures, no listing +``` + +Set `retry_failed = true` to make every run do it. A video that keeps failing is picked +up `retry_max_attempts` times and then left alone, so a genuinely broken one does not +cost a download on every run: + +``` +left 1 failed video(s) alone after retry_max_attempts; 'ytscript run --only-failed' +tries them anyway +``` + +`--only-failed` ignores that limit — it is an explicit request — and `ytscript failures +--clear [ID ...]` forgets entries entirely, resetting their counts. A video that +succeeds drops off the list by itself. + ### Members-only videos A channel's members-only videos show up in its uploads listing, but YouTube refuses the @@ -583,7 +658,9 @@ print(report.written) `Pipeline` takes an optional `client` and `transcriber`, so a different source or speech-to-text engine only has to match the small protocol in -`ytscript/transcribers/base.py`. +`ytscript/transcribers/base.py`. `run()` takes the retry switches too — +`run(retry_failed=True)` and `run(only_failed=True)` — and the report it returns carries +`failed`, `retried` and `given_up` alongside `written`. ## Development diff --git a/src/ytscript/cli.py b/src/ytscript/cli.py index 8389d1a..1aa0cc8 100644 --- a/src/ytscript/cli.py +++ b/src/ytscript/cli.py @@ -12,6 +12,7 @@ from .models import RunReport from .pipeline import Pipeline from .polish import polish_text +from .state import State from .vocabulary import VocabularyError, load_vocabulary from .youtube import YouTubeError @@ -110,6 +111,33 @@ def add_common(target: argparse.ArgumentParser) -> None: metavar="ID_OR_URL", help="Google Drive folder the scripts go into", ) + retry = run.add_mutually_exclusive_group() + retry.add_argument( + "--retry-failed", + dest="retry_failed", + action="store_true", + default=None, + help="also re-attempt videos an earlier run could not finish, however old they are", + ) + retry.add_argument( + "--no-retry-failed", + dest="retry_failed", + action="store_false", + default=None, + help="only look at the newest videos (the default)", + ) + retry.add_argument( + "--only-failed", + action="store_true", + help="re-attempt just those, skipping the channel listing and the attempt limit", + ) + run.add_argument( + "--retries", + dest="download_retries", + type=int, + metavar="N", + help="extra attempts a download gets when the connection drops (default 3)", + ) run.add_argument("--keep-audio", dest="keep_audio", action="store_true", default=None) run.add_argument("--state-file", dest="state_file", type=Path) run.add_argument( @@ -147,6 +175,19 @@ def add_common(target: argparse.ArgumentParser) -> None: add_common(listing) listing.add_argument("--limit", type=int, default=10) + failures = sub.add_parser( + "failures", + help="show the videos an earlier run could not finish", + ) + failures.add_argument("--state-file", dest="state_file", type=Path) + failures.add_argument( + "--clear", + nargs="*", + metavar="ID", + default=None, + help="forget these failures, or all of them when given no id", + ) + sub.add_parser( "drive-auth", help="sign in to Google Drive once and cache the token for later runs", @@ -162,6 +203,8 @@ def add_common(target: argparse.ArgumentParser) -> None: _OVERRIDE_FIELDS = ( "channel", "language", + "retry_failed", + "download_retries", "backend", "whisper_model", "whisper_batch_size", @@ -192,6 +235,13 @@ def _config_from_args(args: argparse.Namespace) -> Config: def _print_report(report: RunReport, dry_run: bool) -> None: print(f"checked {report.checked} video(s); {len(report.skipped)} already had a script") + if report.retried: + print(f"picked {len(report.retried)} video(s) back up from the failure list") + if report.given_up: + print( + f"left {len(report.given_up)} failed video(s) alone after retry_max_attempts; " + "'ytscript run --only-failed' tries them anyway" + ) if report.members_only: print( f"skipped {len(report.members_only)} members-only video(s); " @@ -212,6 +262,12 @@ def _print_report(report: RunReport, dry_run: bool) -> None: print(f" {item}") for video_id, error in report.failed: print(f" failed: {video_id}: {error}", file=sys.stderr) + if report.failed and not dry_run: + print( + f"{len(report.failed)} failure(s) recorded; " + "'ytscript run --retry-failed' takes another run at them", + file=sys.stderr, + ) def cmd_run(args: argparse.Namespace) -> int: @@ -224,6 +280,7 @@ def cmd_run(args: argparse.Namespace) -> int: limit=limit, dry_run=args.dry_run, on_progress=lambda label: print(label, flush=True), + only_failed=args.only_failed, ) _print_report(report, args.dry_run) return 1 if report.failed else 0 @@ -293,6 +350,35 @@ def cmd_polish(args: argparse.Namespace) -> int: return 0 +def cmd_failures(args: argparse.Namespace) -> int: + # No channel is needed to read the state file. + overrides = {"state_file": args.state_file} if args.state_file else {} + config = load_config(path=args.config, overrides=overrides) + state = State.load(config.state_file) + + if args.clear is not None: + dropped = state.forget_failures(args.clear or None) + if dropped: + state.save() + print(f"forgot {len(dropped)} failure(s)") + for video_id in dropped: + print(f" {video_id}") + return 0 + + entries = state.failed_videos() + if not entries: + print("no failures on record") + return 0 + print(f"{len(entries)} video(s) failed; 'ytscript run --retry-failed' tries them again:") + for entry in entries: + attempts = entry.get("attempts", 1) + when = entry.get("last_failed_at", "") + title = entry.get("title", "") + print(f" {entry['id']} {attempts} attempt(s) {when} {title}") + print(f" {entry.get('error', '')}") + return 0 + + def cmd_drive_auth(args: argparse.Namespace) -> int: # No channel is needed to authorise, so this skips the usual validation. config = load_config(path=args.config) @@ -334,6 +420,7 @@ def main(argv: list[str] | None = None) -> int: "run": cmd_run, "list": cmd_list, "polish": cmd_polish, + "failures": cmd_failures, "init": cmd_init, "drive-auth": cmd_drive_auth, } diff --git a/src/ytscript/config.py b/src/ytscript/config.py index 643930e..c9f6732 100644 --- a/src/ytscript/config.py +++ b/src/ytscript/config.py @@ -39,6 +39,28 @@ class Config: check_limit: int = 5 """Newest videos inspected on later runs; unseen ones get transcribed.""" + # --- when something goes wrong --------------------------------------- + download_retries: int = 3 + """Extra attempts a YouTube request gets when the connection drops. ``0`` means one try. + + This is the fix for ``('Connection aborted.', ConnectionResetError(10054, ...))`` + and its friends: the download starts again and yt-dlp resumes the part it has.""" + + retry_backoff: float = 5.0 + """Seconds before the second attempt; each further wait doubles it (5, 10, 20...).""" + + socket_timeout: float = 30.0 + """Seconds a stalled connection is given before it counts as a failed attempt.""" + + retry_failed: bool = False + """Also re-attempt the videos recorded as failed, even when they have fallen out of + the ``check_limit`` window. ``--retry-failed`` turns it on for a single run.""" + + retry_max_attempts: int = 3 + """How many times a failed video is picked up again before it is left alone. + ``ytscript run --only-failed`` retries it regardless, and ``ytscript failures --clear`` + starts the count over.""" + # --- speech-to-text ------------------------------------------------- backend: str = "faster-whisper" whisper_model: str = "large-v3" @@ -176,6 +198,14 @@ def validate(self) -> None: raise ConfigError("check_limit must be at least 1") if self.whisper_batch_size < 1: raise ConfigError("whisper_batch_size must be at least 1 (1 turns batching off)") + if self.download_retries < 0: + raise ConfigError("download_retries cannot be negative (0 means one attempt)") + if self.retry_backoff < 0: + raise ConfigError("retry_backoff cannot be negative") + if self.socket_timeout <= 0: + raise ConfigError("socket_timeout must be greater than 0") + if self.retry_max_attempts < 1: + raise ConfigError("retry_max_attempts must be at least 1") # Reading the file now means a typo fails the command, not the first video. try: load_vocabulary(self.vocabulary) @@ -304,6 +334,19 @@ def load_config( initial_backfill = 30 check_limit = 5 +# A dropped connection mid-download ("Connection aborted", ConnectionResetError) is +# retried this many extra times, waiting 5s, then 10s, then 20s between attempts. +download_retries = 3 +retry_backoff = 5.0 +socket_timeout = 30.0 + +# A video that still fails is written to the state file's "failures" list. Turn this +# on to re-attempt those on every run, even once they are older than `check_limit`; +# `ytscript run --retry-failed` does it for one run, and `ytscript failures` shows +# what is on the list. Each one is picked up at most `retry_max_attempts` times. +retry_failed = false +retry_max_attempts = 3 + # "faster-whisper" runs locally, "openai" calls the hosted transcription API. backend = "faster-whisper" diff --git a/src/ytscript/models.py b/src/ytscript/models.py index 62b2f65..0c0ac2c 100644 --- a/src/ytscript/models.py +++ b/src/ytscript/models.py @@ -81,5 +81,11 @@ class RunReport: members_only: list[str] = field(default_factory=list) """Videos passed over because ``include_members_only`` is off.""" + retried: list[str] = field(default_factory=list) + """Videos brought back from the state file's failure list, outside the usual window.""" + + given_up: list[str] = field(default_factory=list) + """Failures left alone because they have already had ``retry_max_attempts`` goes.""" + uploaded: list[str] = field(default_factory=list) """Scripts copied into Google Drive, as ``name link``. Empty unless ``drive_upload`` is on.""" diff --git a/src/ytscript/pipeline.py b/src/ytscript/pipeline.py index 9a121a3..83acd02 100644 --- a/src/ytscript/pipeline.py +++ b/src/ytscript/pipeline.py @@ -6,7 +6,9 @@ import tempfile from collections.abc import Callable, Iterable from contextlib import contextmanager +from datetime import date from pathlib import Path +from typing import Any from .config import Config from .drive import DriveError, DriveFile, DriveUploader @@ -42,6 +44,25 @@ def select_videos(videos: Iterable[Video], state: State) -> list[Video]: return pending +def _parse_date(raw: Any) -> date | None: + try: + return date.fromisoformat(str(raw)) + except (TypeError, ValueError): + return None + + +def video_from_failure(entry: dict[str, Any]) -> Video: + """Rebuild the video a failure record describes, so a retry needs no fresh listing.""" + video_id = str(entry["id"]) + return Video( + id=video_id, + title=str(entry.get("title") or video_id), + url=str(entry.get("url") or f"https://www.youtube.com/watch?v={video_id}"), + upload_date=_parse_date(entry.get("upload_date")), + members_only=bool(entry.get("members_only")), + ) + + class Pipeline: def __init__( self, @@ -56,6 +77,9 @@ def __init__( audio_format=config.audio_format, cookies_file=config.cookies_file, cookies_from_browser=config.cookies_from_browser, + retries=config.download_retries, + retry_backoff=config.retry_backoff, + socket_timeout=config.socket_timeout, ) self._transcriber = transcriber self._uploader = uploader @@ -96,16 +120,26 @@ def run( limit: int | None = None, dry_run: bool = False, on_progress: Callable[[str], None] | None = None, + retry_failed: bool | None = None, + only_failed: bool = False, ) -> RunReport: + """Transcribe what the channel has that the state file does not. + + ``retry_failed`` (``retry_failed`` in the config when left unset) also picks up + the videos an earlier run recorded as failed, however old they are. + ``only_failed`` does just those, skipping the channel listing entirely, and + ignores ``retry_max_attempts``. + """ config = self.config state = State.load(config.state_file) state.channel = state.channel or config.channel + retry = only_failed or (config.retry_failed if retry_failed is None else retry_failed) if limit is None: limit = config.initial_backfill if state.is_empty else config.check_limit report = RunReport() - videos = self.client.latest_videos(config.channel, limit) + videos = [] if only_failed else self.client.latest_videos(config.channel, limit) report.checked = len(videos) if not config.include_members_only: report.members_only = [video.id for video in videos if video.members_only] @@ -113,6 +147,13 @@ def run( pending = select_videos(videos, state) report.skipped = [video.id for video in videos if state.seen(video.id)] + if retry: + # The failures come first: the oldest stuck video is the one most likely + # to fall out of the listing window for good. + pending = self._failed_videos(state, report, pending, force=only_failed) + pending + if only_failed: + report.checked = len(pending) + if not pending: return report if dry_run: @@ -134,6 +175,17 @@ def run( except (YouTubeError, TranscriptionError, DriveError) as exc: log.warning("%s failed: %s", video.id, exc) report.failed.append((video.id, str(exc))) + # Written down so a later run can pick it up again, even once the + # video is older than check_limit: see run(retry_failed=True). + state.record_failure( + video.id, + str(exc), + title=video.title, + url=video.url, + upload_date=video.upload_date.isoformat() if video.upload_date else None, + members_only=video.members_only or None, + ) + state.save() continue report.written.extend(str(path) for path in paths) report.uploaded.extend(str(upload) for upload in uploads) @@ -151,6 +203,32 @@ def run( state.save() return report + def _failed_videos( + self, state: State, report: RunReport, already: list[Video], force: bool + ) -> list[Video]: + """The videos on the failure list that this run should try again.""" + config = self.config + cap = None if force else config.retry_max_attempts + known = {video.id for video in already} + pending: list[Video] = [] + for entry in state.failed_videos(max_attempts=cap): + # One still inside the listing window is already lined up for its turn. + if entry["id"] in known: + continue + video = video_from_failure(entry) + if video.members_only and not config.include_members_only: + report.members_only.append(video.id) + continue + pending.append(video) + report.retried.append(video.id) + if cap is not None: + report.given_up = [ + entry["id"] + for entry in state.failed_videos() + if int(entry.get("attempts", 0)) >= cap + ] + return pending + def _process(self, video: Video, audio_dir: Path) -> tuple[list[Path], list[DriveFile]]: config = self.config audio_path, video = self.client.download_audio(video, audio_dir) diff --git a/src/ytscript/state.py b/src/ytscript/state.py index b25bdb5..c6e9684 100644 --- a/src/ytscript/state.py +++ b/src/ytscript/state.py @@ -1,4 +1,4 @@ -"""Which videos have already been turned into scripts.""" +"""Which videos have already been turned into scripts, and which ones went wrong.""" from __future__ import annotations @@ -8,7 +8,11 @@ from pathlib import Path from typing import Any -STATE_VERSION = 1 +STATE_VERSION = 2 + + +def _now() -> str: + return datetime.now(UTC).isoformat(timespec="seconds") @dataclass @@ -18,6 +22,8 @@ class State: path: Path channel: str | None = None videos: dict[str, dict[str, Any]] = field(default_factory=dict) + failures: dict[str, dict[str, Any]] = field(default_factory=dict) + """Videos whose last attempt raised, keyed by id, so a later run can retry them.""" @classmethod def load(cls, path: Path) -> State: @@ -32,6 +38,8 @@ def load(cls, path: Path) -> State: path=path, channel=data.get("channel"), videos=dict(data.get("videos") or {}), + # Absent in a version 1 file, which simply has no failures on record. + failures=dict(data.get("failures") or {}), ) @property @@ -42,15 +50,59 @@ def seen(self, video_id: str) -> bool: return video_id in self.videos def record(self, video_id: str, **details: Any) -> None: - entry = {"transcribed_at": datetime.now(UTC).isoformat(timespec="seconds")} + entry = {"transcribed_at": _now()} entry.update({key: value for key, value in details.items() if value is not None}) self.videos[video_id] = entry + # The video is done, so whatever went wrong before no longer needs retrying. + self.failures.pop(video_id, None) + + def record_failure(self, video_id: str, error: str, **details: Any) -> dict[str, Any]: + """Remember that ``video_id`` raised, keeping the count of attempts so far.""" + previous = self.failures.get(video_id, {}) + # Built on top of what is already there, so a detail the retry did not have to + # hand (a title taken from the listing) is not lost on the second failure. + entry = dict(previous) + entry.update({key: value for key, value in details.items() if value is not None}) + entry.update( + { + "error": error, + "attempts": int(previous.get("attempts", 0)) + 1, + "first_failed_at": previous.get("first_failed_at") or _now(), + "last_failed_at": _now(), + } + ) + self.failures[video_id] = entry + return entry + + def attempts(self, video_id: str) -> int: + return int(self.failures.get(video_id, {}).get("attempts", 0)) + + def failed_videos(self, max_attempts: int | None = None) -> list[dict[str, Any]]: + """Recorded failures, oldest first, as entries carrying their own ``id``. + + ``max_attempts`` leaves out the ones that have already been tried that many + times; ``None`` returns every failure on record. + """ + entries = [ + {"id": video_id, **entry} + for video_id, entry in self.failures.items() + if not self.seen(video_id) + and (max_attempts is None or int(entry.get("attempts", 0)) < max_attempts) + ] + entries.sort(key=lambda entry: (str(entry.get("first_failed_at") or ""), entry["id"])) + return entries + + def forget_failures(self, video_ids: list[str] | None = None) -> list[str]: + """Drop failure records, so their attempt counts start over. All of them by default.""" + targets = list(self.failures) if video_ids is None else video_ids + return [video_id for video_id in targets if self.failures.pop(video_id, None) is not None] def save(self) -> None: payload = { "version": STATE_VERSION, "channel": self.channel, "videos": self.videos, + "failures": self.failures, } self.path.parent.mkdir(parents=True, exist_ok=True) tmp = self.path.with_suffix(self.path.suffix + ".tmp") diff --git a/src/ytscript/youtube.py b/src/ytscript/youtube.py index 4193214..cfdc359 100644 --- a/src/ytscript/youtube.py +++ b/src/ytscript/youtube.py @@ -2,13 +2,20 @@ from __future__ import annotations +import logging import re +import time +from collections.abc import Callable, Iterator from datetime import date, datetime from pathlib import Path -from typing import Any +from typing import Any, TypeVar from .models import Video +log = logging.getLogger("ytscript") + +T = TypeVar("T") + _CHANNEL_ID = re.compile(r"^UC[\w-]{22}$") _WATCH_URL = "https://www.youtube.com/watch?v={id}" @@ -19,6 +26,33 @@ # What YouTube says when the request is not signed in as a member of the channel. _MEMBERS_ONLY_ERRORS = ("members-only", "members only", "join this channel") +# Network trouble that another attempt can get past. Matched against the whole +# exception chain in lower case, so a localised OS message ("远程主机强迫关闭了一个现有 +# 的连接。") is still caught by the English class name Python wraps it in. +_TRANSIENT_HINTS = ( + "connection aborted", + "connection reset", + "connectionreseterror", + "connection broken", + "connection refused", + "connectionerror", + "remote end closed connection", + "incompleteread", + "timed out", + "timeout", + "temporary failure in name resolution", + "unable to download webpage", + "unable to download api page", + "giving up after", + "content too short", + "http error 429", + "http error 500", + "http error 502", + "http error 503", + "http error 504", + "eof occurred in violation of protocol", +) + _MEMBERSHIP_HINT = ( "sign in as a member: set cookies_file (a cookies.txt export) or " 'cookies_from_browser (e.g. "firefox") to an account that holds the membership' @@ -38,6 +72,38 @@ class YouTubeError(RuntimeError): """Raised when yt-dlp cannot list a channel or fetch a video.""" + def __init__(self, message: str, *, transient: bool = False) -> None: + super().__init__(message) + self.transient = transient + """Whether the same call has a fair chance of working on another try.""" + + +def _causes(exc: BaseException) -> Iterator[BaseException]: + """Walk an exception and everything it was raised from.""" + seen: set[int] = set() + current: BaseException | None = exc + while current is not None and id(current) not in seen: + seen.add(id(current)) + yield current + current = current.__cause__ or current.__context__ + + +def is_transient(exc: BaseException) -> bool: + """Whether ``exc`` looks like network trouble rather than a refusal. + + A dropped connection mid-download is the common one: + ``('Connection aborted.', ConnectionResetError(10054, ...))``. + """ + for error in _causes(exc): + if isinstance(error, ConnectionError | TimeoutError): + return True + if isinstance(error, OSError) and error.errno in (54, 60, 104, 110, 10054, 10060): + return True + lowered = str(error).lower() + if any(hint in lowered for hint in _TRANSIENT_HINTS): + return True + return False + def _load_yt_dlp() -> Any: try: @@ -136,11 +202,23 @@ def __init__( cookies_file: Path | None = None, cookies_from_browser: str | None = None, quiet: bool = True, + retries: int = 3, + retry_backoff: float = 5.0, + socket_timeout: float = 30.0, + sleep: Callable[[float], None] = time.sleep, ) -> None: self.audio_format = audio_format self.cookies_file = cookies_file self.cookies_from_browser = cookies_from_browser self.quiet = quiet + self.retries = max(0, retries) + """Extra attempts a request gets after network trouble; 0 means one try only.""" + + self.retry_backoff = max(0.0, retry_backoff) + """Seconds before the second attempt; each further wait doubles it.""" + + self.socket_timeout = socket_timeout + self._sleep = sleep @property def signed_in(self) -> bool: @@ -153,6 +231,13 @@ def _base_opts(self) -> dict[str, Any]: "no_warnings": self.quiet, "noprogress": self.quiet, "ignoreerrors": False, + # yt-dlp's own retries come first: they resume a half-downloaded file + # from its .part, where our outer retry starts the request again. + "retries": self.retries, + "fragment_retries": self.retries, + "extractor_retries": self.retries, + "socket_timeout": self.socket_timeout, + "continuedl": True, } if self.cookies_file: opts["cookiefile"] = str(self.cookies_file) @@ -160,8 +245,31 @@ def _base_opts(self) -> dict[str, Any]: opts["cookiesfrombrowser"] = parse_browser_spec(self.cookies_from_browser) return opts + def _attempt(self, what: str, call: Callable[[], T]) -> T: + """Run ``call``, giving it another go while the failure looks like network trouble.""" + for attempt in range(1, self.retries + 2): + try: + return call() + except YouTubeError as exc: + if not exc.transient or attempt > self.retries: + raise + delay = self.retry_backoff * 2 ** (attempt - 1) + log.warning( + "%s failed (attempt %d of %d): %s; retrying in %.0fs", + what, + attempt, + self.retries + 1, + exc, + delay, + ) + self._sleep(delay) + raise AssertionError("unreachable") # pragma: no cover + def latest_videos(self, channel: str, limit: int) -> list[Video]: """Return up to ``limit`` of the channel's newest uploads, newest first.""" + return self._attempt(f"listing {channel}", lambda: self._list_videos_once(channel, limit)) + + def _list_videos_once(self, channel: str, limit: int) -> list[Video]: yt_dlp = _load_yt_dlp() url = channel_uploads_url(channel) opts = self._base_opts() | { @@ -173,14 +281,25 @@ def latest_videos(self, channel: str, limit: int) -> list[Video]: with yt_dlp.YoutubeDL(opts) as ydl: info = ydl.extract_info(url, download=False) except Exception as exc: # yt-dlp raises a family of DownloadError subclasses - raise YouTubeError(f"could not list videos for {channel!r}: {exc}") from exc + raise YouTubeError( + f"could not list videos for {channel!r}: {exc}", transient=is_transient(exc) + ) from exc entries = list(_iter_entries(info)) videos = [_to_video(entry) for entry in entries if entry.get("id")] return videos[:limit] def download_audio(self, video: Video, dest_dir: Path) -> tuple[Path, Video]: - """Download the audio track and return its path plus enriched metadata.""" + """Download the audio track and return its path plus enriched metadata. + + A connection dropped mid-download is retried with a widening pause; yt-dlp + picks the file up from the part it already has. + """ + return self._attempt( + f"downloading {video.id}", lambda: self._download_audio_once(video, dest_dir) + ) + + def _download_audio_once(self, video: Video, dest_dir: Path) -> tuple[Path, Video]: yt_dlp = _load_yt_dlp() dest_dir = Path(dest_dir) dest_dir.mkdir(parents=True, exist_ok=True) @@ -200,7 +319,9 @@ def download_audio(self, video: Video, dest_dir: Path) -> tuple[Path, Video]: if _mentions_membership(str(exc)): # A listing without badges (or a video made members-only later) lands here. raise YouTubeError(f"{video.id} is members-only; {_MEMBERSHIP_HINT}") from exc - raise YouTubeError(f"could not download audio for {video.id}: {exc}") from exc + raise YouTubeError( + f"could not download audio for {video.id}: {exc}", transient=is_transient(exc) + ) from exc if not path.is_file(): matches = sorted(dest_dir.glob(f"{video.id}.*")) diff --git a/tests/test_cli.py b/tests/test_cli.py index 7fc79b4..afdbec8 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -7,6 +7,7 @@ from fakes import FakeDriveUploader, FakeTranscriber, FakeYouTubeClient, make_videos from ytscript import cli +from ytscript.state import State @pytest.fixture() @@ -267,3 +268,79 @@ def test_polish_says_when_a_path_is_missing(project: Path, capsys: pytest.Captur def test_an_unknown_vocabulary_stops_the_run(project: Path, capsys: pytest.CaptureFixture) -> None: assert cli.main(["run", "--vocabulary", "nope"]) == 1 assert "no vocabulary 'nope'" in capsys.readouterr().err + + +def _seed_failure(project: Path, video_id: str = "vid007") -> None: + state = State.load(project / "state.json") + state.record_failure( + video_id, + "could not download audio: ('Connection aborted.', ConnectionResetError(10054, ...))", + title=f"Episode {video_id[-1]}", + url=f"https://www.youtube.com/watch?v={video_id}", + ) + state.save() + + +def test_run_only_failed_retries_what_the_state_file_recorded( + project: Path, fake_pipeline, capsys: pytest.CaptureFixture +) -> None: + _seed_failure(project) + assert cli.main(["run", "--only-failed"]) == 0 + + out = capsys.readouterr().out + assert "picked 1 video(s) back up from the failure list" in out + # Nothing was listed: the retry works straight off the state file. + assert fake_pipeline.listed == [] and fake_pipeline.downloaded == ["vid007"] + assert State.load(project / "state.json").failures == {} + + +def test_run_retry_failed_flag_adds_them_to_the_usual_window( + project: Path, fake_pipeline, capsys: pytest.CaptureFixture +) -> None: + _seed_failure(project) + assert cli.main(["run", "--retry-failed"]) == 0 + assert fake_pipeline.listed == [("@testchannel", 3)] + assert "vid007" in fake_pipeline.downloaded + + _seed_failure(project) + assert cli.main(["run", "--no-retry-failed"]) == 0 + assert State.load(project / "state.json").failures # left where it was + + +def test_failures_lists_and_clears_the_record(project: Path, capsys: pytest.CaptureFixture) -> None: + assert cli.main(["failures"]) == 0 + assert "no failures on record" in capsys.readouterr().out + + _seed_failure(project) + assert cli.main(["failures"]) == 0 + out = capsys.readouterr().out + assert "vid007" in out and "1 attempt(s)" in out and "Connection aborted" in out + + assert cli.main(["failures", "--clear"]) == 0 + assert "forgot 1 failure(s)" in capsys.readouterr().out + assert State.load(project / "state.json").failures == {} + + +def test_failures_clear_takes_specific_ids(project: Path, capsys: pytest.CaptureFixture) -> None: + _seed_failure(project, "vid007") + _seed_failure(project, "vid008") + assert cli.main(["failures", "--clear", "vid007"]) == 0 + assert list(State.load(project / "state.json").failures) == ["vid008"] + + +def test_a_failed_run_points_at_the_retry_flag( + project: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture +) -> None: + client = FakeYouTubeClient(make_videos(2)) + real_init = cli.Pipeline.__init__ + monkeypatch.setattr( + cli.Pipeline, + "__init__", + lambda self, config, client_=None, transcriber=None: real_init( + self, config, client=client, transcriber=FakeTranscriber(fail_on={"vid001"}) + ), + ) + assert cli.main(["run"]) == 1 + err = capsys.readouterr().err + assert "failed: vid001" in err + assert "--retry-failed" in err diff --git a/tests/test_config.py b/tests/test_config.py index 3d15022..bdd3c35 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -125,3 +125,15 @@ def test_members_only_reads_from_file_and_env(tmp_path: Path) -> None: off = load_config(path=path, env={"YTSCRIPT_INCLUDE_MEMBERS_ONLY": "false"}) assert off.include_members_only is False + + +def test_retry_settings_are_checked() -> None: + with pytest.raises(ConfigError, match="download_retries"): + Config(channel="@c", download_retries=-1).validate() + with pytest.raises(ConfigError, match="retry_backoff"): + Config(channel="@c", retry_backoff=-1.0).validate() + with pytest.raises(ConfigError, match="socket_timeout"): + Config(channel="@c", socket_timeout=0).validate() + with pytest.raises(ConfigError, match="retry_max_attempts"): + Config(channel="@c", retry_max_attempts=0).validate() + Config(channel="@c", download_retries=0, retry_backoff=0.0).validate() diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index 7fadecb..805ea6b 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -122,6 +122,115 @@ def test_failed_video_is_retried_on_the_next_run(tmp_path: Path) -> None: assert len(report.written) == 1 +def test_a_failure_is_written_to_the_state_file(tmp_path: Path) -> None: + client = FakeYouTubeClient(make_videos(2)) + transcriber = FakeTranscriber(fail_on={"vid001"}) + config = make_config(tmp_path, initial_backfill=2) + Pipeline(config, client=client, transcriber=transcriber).run() + + entry = State.load(tmp_path / "state.json").failures["vid001"] + assert entry["attempts"] == 1 + assert entry["title"] == "Episode 1" + assert entry["url"].endswith("vid001") + assert "backend exploded" in entry["error"] + + +def test_retry_failed_picks_up_a_video_outside_the_window(tmp_path: Path) -> None: + videos = make_videos(6) + client = FakeYouTubeClient(videos) + config = make_config(tmp_path, initial_backfill=6, check_limit=2) + Pipeline(config, client=client, transcriber=FakeTranscriber(fail_on={"vid005"})).run() + + # vid005 is the oldest, so a later run's two-video window no longer covers it. + healthy = FakeTranscriber() + plain = Pipeline(config, client=client, transcriber=healthy).run() + assert plain.written == [] and plain.retried == [] + + report = Pipeline(config, client=client, transcriber=healthy).run(retry_failed=True) + assert report.retried == ["vid005"] + assert len(report.written) == 1 + state = State.load(tmp_path / "state.json") + assert state.seen("vid005") and state.failures == {} + + +def test_retry_failed_can_be_turned_on_in_the_config(tmp_path: Path) -> None: + client = FakeYouTubeClient(make_videos(3)) + config = make_config(tmp_path, initial_backfill=3, check_limit=1, retry_failed=True) + Pipeline(config, client=client, transcriber=FakeTranscriber(fail_on={"vid002"})).run() + + report = Pipeline(config, client=client, transcriber=FakeTranscriber()).run() + assert report.retried == ["vid002"] + + +def test_a_failure_inside_the_window_is_not_queued_twice(tmp_path: Path) -> None: + client = FakeYouTubeClient(make_videos(2)) + config = make_config(tmp_path, initial_backfill=2, check_limit=2, retry_failed=True) + Pipeline(config, client=client, transcriber=FakeTranscriber(fail_on={"vid001"})).run() + + healthy = FakeTranscriber() + report = Pipeline(config, client=client, transcriber=healthy).run() + assert report.retried == [] + assert [call[0].stem for call in healthy.calls] == ["vid001"] + + +def test_a_video_is_left_alone_after_retry_max_attempts(tmp_path: Path) -> None: + client = FakeYouTubeClient(make_videos(3)) + config = make_config( + tmp_path, initial_backfill=3, check_limit=1, retry_failed=True, retry_max_attempts=2 + ) + broken = FakeTranscriber(fail_on={"vid002"}) + for _ in range(2): + Pipeline(config, client=client, transcriber=broken).run() + assert State.load(tmp_path / "state.json").attempts("vid002") == 2 + + report = Pipeline(config, client=client, transcriber=broken).run() + assert report.retried == [] and report.given_up == ["vid002"] + + # An explicit --only-failed run overrules the cap. + forced = Pipeline(config, client=client, transcriber=broken).run(only_failed=True) + assert forced.retried == ["vid002"] + + +def test_only_failed_never_lists_the_channel(tmp_path: Path) -> None: + client = FakeYouTubeClient(make_videos(2)) + config = make_config(tmp_path, initial_backfill=2) + Pipeline(config, client=client, transcriber=FakeTranscriber(fail_on={"vid001"})).run() + listings = len(client.listed) + + healthy = FakeTranscriber() + report = Pipeline(config, client=client, transcriber=healthy).run(only_failed=True) + assert len(client.listed) == listings + assert report.checked == 1 and report.retried == ["vid001"] + assert [call[0].stem for call in healthy.calls] == ["vid001"] + + +def test_only_failed_with_nothing_on_the_list_does_nothing(tmp_path: Path) -> None: + pipeline, client, transcriber = build(tmp_path, make_videos(2)) + report = pipeline.run(only_failed=True) + assert client.listed == [] and transcriber.calls == [] + assert report.checked == 0 and report.written == [] + + +def test_a_retried_members_only_video_is_still_passed_over(tmp_path: Path) -> None: + videos = _with_members_only(make_videos(2), 1) + config = make_config( + tmp_path, + initial_backfill=2, + include_members_only=True, + cookies_file=tmp_path / "cookies.txt", + ) + client = FakeYouTubeClient(videos) + Pipeline(config, client=client, transcriber=FakeTranscriber(fail_on={"vid001"})).run() + assert State.load(tmp_path / "state.json").failures["vid001"]["members_only"] is True + + # The membership lapses: the failure is on record but must not be downloaded again. + signed_out = make_config(tmp_path, initial_backfill=2, retry_failed=True) + report = Pipeline(signed_out, client=client, transcriber=FakeTranscriber()).run( + only_failed=True + ) + assert report.retried == [] and report.members_only == ["vid001"] + + def test_dry_run_downloads_nothing(tmp_path: Path) -> None: pipeline, client, transcriber = build(tmp_path, make_videos(2), initial_backfill=2) report = pipeline.run(dry_run=True) diff --git a/tests/test_state.py b/tests/test_state.py index 6d13466..d97e119 100644 --- a/tests/test_state.py +++ b/tests/test_state.py @@ -34,3 +34,52 @@ def test_corrupt_state_raises(tmp_path: Path) -> None: path.write_text("{not json", encoding="utf-8") with pytest.raises(ValueError, match="not valid JSON"): State.load(path) + + +def test_failures_are_recorded_and_counted(tmp_path: Path) -> None: + path = tmp_path / "state.json" + state = State.load(path) + state.record_failure("abc", "connection reset", title="Hello", url="https://y/abc") + state.record_failure("abc", "connection reset again", title="Hello") + state.save() + + entry = State.load(path).failures["abc"] + assert entry["attempts"] == 2 + assert entry["error"] == "connection reset again" + assert entry["url"] == "https://y/abc" + assert entry["first_failed_at"] <= entry["last_failed_at"] + + +def test_a_finished_video_drops_off_the_failure_list(tmp_path: Path) -> None: + state = State.load(tmp_path / "state.json") + state.record_failure("abc", "boom") + state.record("abc", title="Hello") + assert state.failures == {} + assert state.failed_videos() == [] + + +def test_failed_videos_are_oldest_first_and_respect_the_cap(tmp_path: Path) -> None: + state = State.load(tmp_path / "state.json") + state.failures = { + "new": {"error": "b", "attempts": 1, "first_failed_at": "2024-05-02T00:00:00+00:00"}, + "old": {"error": "a", "attempts": 3, "first_failed_at": "2024-05-01T00:00:00+00:00"}, + } + assert [entry["id"] for entry in state.failed_videos()] == ["old", "new"] + assert [entry["id"] for entry in state.failed_videos(max_attempts=3)] == ["new"] + assert state.attempts("old") == 3 and state.attempts("missing") == 0 + + +def test_forget_failures_clears_all_or_some(tmp_path: Path) -> None: + state = State.load(tmp_path / "state.json") + state.record_failure("a", "boom") + state.record_failure("b", "boom") + assert state.forget_failures(["a", "never-failed"]) == ["a"] + assert state.forget_failures() == ["b"] + assert state.failures == {} + + +def test_a_version_one_file_loads_without_failures(tmp_path: Path) -> None: + path = tmp_path / "state.json" + path.write_text('{"version": 1, "videos": {"abc": {}}}', encoding="utf-8") + state = State.load(path) + assert state.seen("abc") and state.failures == {} diff --git a/tests/test_youtube.py b/tests/test_youtube.py index 9648c76..ef1c3fd 100644 --- a/tests/test_youtube.py +++ b/tests/test_youtube.py @@ -13,9 +13,17 @@ _mentions_membership, _to_video, channel_uploads_url, + is_transient, parse_browser_spec, ) +# What yt-dlp prints when YouTube drops the connection part-way through, here on a +# Chinese Windows: "the remote host forcibly closed an existing connection". +CONNECTION_RESET = ( + "[download] Got error: ('Connection aborted.', " + "ConnectionResetError(10054, '远程主机强迫关闭了一个现有的连接。', None, 10054, None))" +) + @pytest.mark.parametrize( ("value", "expected"), @@ -112,3 +120,104 @@ def test_members_only_download_without_cookies_says_what_is_missing(tmp_path: Pa video = Video(id="x", title="T", url="https://y/watch?v=x", members_only=True) with pytest.raises(YouTubeError, match="members-only"): YouTubeClient().download_audio(video, tmp_path) + + +def test_a_dropped_connection_counts_as_transient() -> None: + assert is_transient(RuntimeError(CONNECTION_RESET)) + assert is_transient(ConnectionResetError(10054, "远程主机强迫关闭了一个现有的连接。")) + assert is_transient(TimeoutError("The read operation timed out")) + assert is_transient(OSError("HTTP Error 503: Service Unavailable")) + + +def test_a_refusal_is_not_transient() -> None: + assert not is_transient(RuntimeError("Video unavailable")) + assert not is_transient(RuntimeError("Join this channel to get access to members-only content")) + + +def test_transient_causes_are_seen_through_the_wrapper() -> None: + try: + try: + raise ConnectionResetError(10054, "closed") + except ConnectionResetError as reset: + raise RuntimeError("could not download audio for x") from reset + except RuntimeError as exc: + assert is_transient(exc) + + +def test_a_dropped_download_is_tried_again(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + video = Video(id="x", title="T", url="https://y/watch?v=x") + waits: list[float] = [] + client = YouTubeClient(retries=2, retry_backoff=1.0, sleep=waits.append) + attempts: list[str] = [] + + def flaky(video: Video, dest_dir: Path) -> tuple[Path, Video]: + attempts.append(video.id) + if len(attempts) < 3: + raise YouTubeError( + f"could not download audio for x: {CONNECTION_RESET}", transient=True + ) + return dest_dir / "x.m4a", video + + monkeypatch.setattr(client, "_download_audio_once", flaky) + path, got = client.download_audio(video, tmp_path) + + assert path == tmp_path / "x.m4a" and got is video + assert len(attempts) == 3 + # The pause widens, so a blip costs a second and an outage is not hammered. + assert waits == [1.0, 2.0] + + +def test_retries_run_out_and_the_last_error_is_raised( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + video = Video(id="x", title="T", url="https://y/watch?v=x") + attempts: list[str] = [] + client = YouTubeClient(retries=1, retry_backoff=0.0, sleep=lambda _: None) + + def always_drops(video: Video, dest_dir: Path) -> tuple[Path, Video]: + attempts.append(video.id) + raise YouTubeError("could not download audio for x: connection reset", transient=True) + + monkeypatch.setattr(client, "_download_audio_once", always_drops) + with pytest.raises(YouTubeError, match="connection reset"): + client.download_audio(video, tmp_path) + assert len(attempts) == 2 + + +def test_a_refused_download_is_not_tried_again( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + video = Video(id="x", title="T", url="https://y/watch?v=x") + attempts: list[str] = [] + client = YouTubeClient(retries=3, retry_backoff=0.0, sleep=lambda _: None) + + def refused(video: Video, dest_dir: Path) -> tuple[Path, Video]: + attempts.append(video.id) + raise YouTubeError("x is members-only; sign in as a member") + + monkeypatch.setattr(client, "_download_audio_once", refused) + with pytest.raises(YouTubeError, match="members-only"): + client.download_audio(video, tmp_path) + assert len(attempts) == 1 + + +def test_a_dropped_listing_is_tried_again(monkeypatch: pytest.MonkeyPatch) -> None: + client = YouTubeClient(retries=1, retry_backoff=0.0, sleep=lambda _: None) + attempts: list[int] = [] + + def flaky(channel: str, limit: int) -> list[Video]: + attempts.append(limit) + if len(attempts) == 1: + raise YouTubeError("could not list videos: connection aborted", transient=True) + return [Video(id="a", title="A", url="https://y/watch?v=a")] + + monkeypatch.setattr(client, "_list_videos_once", flaky) + assert [video.id for video in client.latest_videos("@chan", 5)] == ["a"] + assert attempts == [5, 5] + + +def test_retry_settings_reach_yt_dlp() -> None: + opts = YouTubeClient(retries=7, socket_timeout=12.0)._base_opts() + assert opts["retries"] == 7 and opts["fragment_retries"] == 7 + assert opts["socket_timeout"] == 12.0 + assert opts["continuedl"] is True