The distribution, console script and import package are now audio-scribe / audio_scribe (src/audio_scribe). Everything named for the old project follows: - CcnError -> AudioScribeError, and its code "ccn_error" -> "audio_scribe_error" (the code is never persisted, so existing job state still loads) - CCN_LIVE -> AUDIO_SCRIBE_LIVE for the live-GPU tests - OpenVINO kernel cache moves to <cache>/audio-scribe/ov_cache; the first run after upgrading recompiles kernels, and the old directory is left in place - README, build-binary.sh, hatch/coverage config and uv.lock updated to match Breaking: the command is now `audio-scribe`; reinstall any tool install of the old name with `uv tool uninstall ccn-transcribe && uv tool install .`. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
125 lines
5.1 KiB
Python
125 lines
5.1 KiB
Python
from __future__ import annotations
|
|
|
|
from typing import TYPE_CHECKING
|
|
|
|
import pytest
|
|
|
|
from audio_scribe import paths
|
|
|
|
if TYPE_CHECKING:
|
|
from pathlib import Path
|
|
|
|
|
|
class TestJobId:
|
|
def test_combines_extractor_and_video_id(self) -> None:
|
|
assert paths.job_id("Youtube", "dQw4w9WgXcQ") == "youtube-dQw4w9WgXcQ"
|
|
|
|
def test_extractor_key_is_lowercased(self) -> None:
|
|
# Extractor-key casing has changed across yt-dlp releases; folding it
|
|
# keeps a version bump from orphaning existing job directories.
|
|
assert paths.job_id("YouTube", "x") == paths.job_id("youtube", "x")
|
|
|
|
def test_extractor_punctuation_is_stripped(self) -> None:
|
|
assert paths.job_id("Some:Site!", "x").startswith("somesite-")
|
|
|
|
def test_empty_extractor_falls_back_to_generic(self) -> None:
|
|
assert paths.job_id("", "x").startswith("generic-")
|
|
|
|
def test_unsafe_characters_are_replaced(self) -> None:
|
|
assert "/" not in paths.job_id("youtube", "a/b")
|
|
|
|
def test_ids_that_sanitize_alike_stay_distinct(self) -> None:
|
|
# "a/b" and "a:b" both sanitize to "a_b"; without a digest suffix they
|
|
# would share one job directory and corrupt each other's state.
|
|
assert paths.job_id("youtube", "a/b") != paths.job_id("youtube", "a:b")
|
|
|
|
def test_clean_ids_get_no_digest_suffix(self) -> None:
|
|
assert paths.job_id("youtube", "abc-123_x.y") == "youtube-abc-123_x.y"
|
|
|
|
def test_very_long_ids_are_truncated_but_stay_distinct(self) -> None:
|
|
long_a = "z" * 200 + "a"
|
|
long_b = "z" * 200 + "b"
|
|
assert len(paths.job_id("youtube", long_a)) < 120
|
|
assert paths.job_id("youtube", long_a) != paths.job_id("youtube", long_b)
|
|
|
|
def test_is_deterministic(self) -> None:
|
|
assert paths.job_id("youtube", "a/b") == paths.job_id("youtube", "a/b")
|
|
|
|
@pytest.mark.parametrize("hostile", ["..", "../..", "../../etc/passwd", "/etc/passwd"])
|
|
def test_cannot_escape_the_jobs_directory(self, hostile: str, tmp_path: Path) -> None:
|
|
jid = paths.job_id("youtube", hostile)
|
|
resolved = (tmp_path / "jobs" / jid).resolve()
|
|
assert resolved.parent == (tmp_path / "jobs").resolve()
|
|
|
|
|
|
class TestUrlJobId:
|
|
def test_is_stable(self) -> None:
|
|
url = "https://example.com/a.mp4"
|
|
assert paths.url_job_id(url) == paths.url_job_id(url)
|
|
|
|
def test_differs_between_urls(self) -> None:
|
|
assert paths.url_job_id("https://a.test/x") != paths.url_job_id("https://a.test/y")
|
|
|
|
def test_is_filesystem_safe(self) -> None:
|
|
jid = paths.url_job_id("https://a.test/x?q=1&r=2#frag")
|
|
assert jid.startswith("url-")
|
|
assert all(c.isalnum() or c in "-_." for c in jid)
|
|
|
|
|
|
class TestNormalizeUrl:
|
|
def test_strips_surrounding_whitespace(self) -> None:
|
|
assert paths.normalize_url(" https://a.test/x ") == "https://a.test/x"
|
|
|
|
def test_drops_the_fragment(self) -> None:
|
|
assert paths.normalize_url("https://a.test/x#t=30") == "https://a.test/x"
|
|
|
|
def test_keeps_the_query(self) -> None:
|
|
# YouTube identifies the video in the query string.
|
|
assert paths.normalize_url("https://y.test/watch?v=abc") == "https://y.test/watch?v=abc"
|
|
|
|
def test_lowercases_scheme_and_host_only(self) -> None:
|
|
assert paths.normalize_url("HTTPS://Example.COM/Path") == "https://example.com/Path"
|
|
|
|
def test_equivalent_urls_map_to_one_job(self) -> None:
|
|
a = paths.normalize_url("https://a.test/x#one")
|
|
b = paths.normalize_url(" https://a.test/x#two ")
|
|
assert paths.url_job_id(a) == paths.url_job_id(b)
|
|
|
|
|
|
class TestWorkspace:
|
|
def test_job_directory_lives_under_jobs(self, tmp_path: Path) -> None:
|
|
ws = paths.Workspace(tmp_path)
|
|
assert ws.job("youtube-x").root == tmp_path / "jobs" / "youtube-x"
|
|
|
|
def test_index_file_location(self, tmp_path: Path) -> None:
|
|
assert paths.Workspace(tmp_path).index_file == tmp_path / "index.json"
|
|
|
|
def test_ensure_creates_the_tree(self, tmp_path: Path) -> None:
|
|
job = paths.Workspace(tmp_path).job("youtube-x")
|
|
job.ensure()
|
|
for d in (job.media_dir, job.out_dir, job.logs_dir, job.tmp_dir):
|
|
assert d.is_dir()
|
|
|
|
def test_ensure_is_idempotent(self, tmp_path: Path) -> None:
|
|
job = paths.Workspace(tmp_path).job("youtube-x")
|
|
job.ensure()
|
|
job.ensure()
|
|
assert job.media_dir.is_dir()
|
|
|
|
def test_artifact_locations(self, tmp_path: Path) -> None:
|
|
job = paths.Workspace(tmp_path).job("youtube-x")
|
|
assert job.state_file.name == "state.json"
|
|
assert job.lock_file.name == ".lock"
|
|
assert job.events_file.name == "events.jsonl"
|
|
assert job.info_file.name == "info.json"
|
|
assert job.audio_file == job.media_dir / "audio.flac"
|
|
|
|
def test_tmp_is_inside_the_job_dir(self, tmp_path: Path) -> None:
|
|
# os.replace is only atomic within one filesystem.
|
|
job = paths.Workspace(tmp_path).job("youtube-x")
|
|
assert job.tmp_dir.parent == job.root
|
|
|
|
def test_transcript_paths_are_named_per_format(self, tmp_path: Path) -> None:
|
|
job = paths.Workspace(tmp_path).job("youtube-x")
|
|
assert job.transcript_file("srt") == job.out_dir / "transcript.srt"
|