From f32516d5d56fe8d8ad5f64799b8dac0f307ac727 Mon Sep 17 00:00:00 2001 From: Jason Ross Date: Sun, 13 Sep 2026 15:51:45 -0500 Subject: [PATCH] feat(outputs): implement --out-dir, which was accepted but did nothing The flag was parsed into RunConfig and never read, so it silently had no effect. Finished transcripts are now also copied into that folder as -<job_id>.<ext>, which is what makes a batch readable without digging through job directories. It copies rather than moves, and only the formats actually requested: the job directory stays canonical because resume depends on it, and transcript.json is written as the segment cache whether or not it was asked for. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/ccn_transcribe/jobs/runner.py | 1 + src/ccn_transcribe/paths.py | 7 +++ src/ccn_transcribe/stages/outputs.py | 33 +++++++++++++ tests/test_stages_pipeline.py | 71 ++++++++++++++++++++++++++++ 4 files changed, 112 insertions(+) diff --git a/src/ccn_transcribe/jobs/runner.py b/src/ccn_transcribe/jobs/runner.py index 9731e57..a840406 100644 --- a/src/ccn_transcribe/jobs/runner.py +++ b/src/ccn_transcribe/jobs/runner.py @@ -228,6 +228,7 @@ def _do_transcript( state.outputs = Outputs(status=ArtifactStatus.PRESENT, paths=written) store.save(job, state) + outputs_stage.publish(job, written, config.formats, out_dir=config.out_dir, title=state.title) if config.no_retain_audio: _record_deletion(job, state, "audio", "--no-retain-audio", job.audio_file) diff --git a/src/ccn_transcribe/paths.py b/src/ccn_transcribe/paths.py index 3f8dc76..671a1be 100644 --- a/src/ccn_transcribe/paths.py +++ b/src/ccn_transcribe/paths.py @@ -50,6 +50,13 @@ def url_job_id(url: str) -> str: return f"url-{hashlib.sha256(normalize_url(url).encode()).hexdigest()[:16]}" +def output_stem(title: str | None, identifier: str) -> str: + """A readable, filesystem-safe basename for a collected transcript.""" + cleaned = _UNSAFE.sub("_", (title or "").strip()).strip("_.") + cleaned = re.sub(r"_{2,}", "_", cleaned)[:80].strip("_.") + return f"{cleaned}-{identifier}" if cleaned else identifier + + @dataclass(frozen=True, slots=True) class JobPaths: root: Path diff --git a/src/ccn_transcribe/stages/outputs.py b/src/ccn_transcribe/stages/outputs.py index 3388a5c..10abd53 100644 --- a/src/ccn_transcribe/stages/outputs.py +++ b/src/ccn_transcribe/stages/outputs.py @@ -4,12 +4,15 @@ from __future__ import annotations import logging import os +import shutil from typing import TYPE_CHECKING from ccn_transcribe import formats +from ccn_transcribe.paths import output_stem if TYPE_CHECKING: from collections.abc import Mapping, Sequence + from pathlib import Path from ccn_transcribe.paths import JobPaths from ccn_transcribe.transcript import TranscriptResult @@ -54,3 +57,33 @@ def write( written[name] = str(path.relative_to(job.root)) log.info("wrote %s", ", ".join(sorted(written))) return written + + +def publish( + job: JobPaths, + written: Mapping[str, str], + wanted: Sequence[str], + *, + out_dir: Path | None, + title: str | None, +) -> list[Path]: + """Additionally collect transcripts into one folder, named by title. + + The job directory stays canonical -- resume depends on it -- so this only + copies, and only the formats the user actually asked for. + """ + if out_dir is None: + return [] + out_dir.mkdir(parents=True, exist_ok=True) + stem = output_stem(title, job.root.name) + copied: list[Path] = [] + for name in wanted: + relative = written.get(name) + if relative is None: + continue + destination = out_dir / f"{stem}.{name}" + shutil.copy2(job.root / relative, destination) + copied.append(destination) + if copied: + log.info("collected %d file(s) into %s", len(copied), out_dir) + return copied diff --git a/tests/test_stages_pipeline.py b/tests/test_stages_pipeline.py index 4e48024..f857349 100644 --- a/tests/test_stages_pipeline.py +++ b/tests/test_stages_pipeline.py @@ -199,3 +199,74 @@ class TestOutputsStage: ) outputs_stage.write(job, shorter, ("txt",)) assert job.transcript_file("txt").read_text().strip() == "hi" + + +class TestPublish: + def test_does_nothing_without_an_out_dir(self, job: paths.JobPaths) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + assert outputs_stage.publish(job, written, ("txt",), out_dir=None, title="T") == [] + + def test_copies_requested_formats_named_by_title( + self, job: paths.JobPaths, tmp_path: Path + ) -> None: + written = outputs_stage.write(job, RESULT, ("txt", "srt")) + collected = tmp_path / "collected" + got = outputs_stage.publish( + job, written, ("txt", "srt"), out_dir=collected, title="My Talk" + ) + assert {p.name for p in got} == { + "My_Talk-youtube-abc.txt", + "My_Talk-youtube-abc.srt", + } + assert all(p.read_text() for p in got) + + def test_the_job_directory_is_left_canonical(self, job: paths.JobPaths, tmp_path: Path) -> None: + # Resume depends on the job dir, so publishing copies rather than moves. + written = outputs_stage.write(job, RESULT, ("txt",)) + outputs_stage.publish(job, written, ("txt",), out_dir=tmp_path / "c", title="T") + assert job.transcript_file("txt").exists() + + def test_only_requested_formats_are_collected( + self, job: paths.JobPaths, tmp_path: Path + ) -> None: + # json is always written as the segment cache, but need not be collected. + written = outputs_stage.write(job, RESULT, ("txt",)) + collected = tmp_path / "collected" + got = outputs_stage.publish(job, written, ("txt",), out_dir=collected, title="T") + assert [p.suffix for p in got] == [".txt"] + + def test_a_missing_format_is_skipped(self, job: paths.JobPaths, tmp_path: Path) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + got = outputs_stage.publish(job, written, ("txt", "srt"), out_dir=tmp_path / "c", title="T") + assert len(got) == 1 + + def test_creates_the_directory(self, job: paths.JobPaths, tmp_path: Path) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + target = tmp_path / "deep" / "nested" + outputs_stage.publish(job, written, ("txt",), out_dir=target, title="T") + assert target.is_dir() + + +class TestOutputStem: + def test_combines_title_and_id(self) -> None: + assert paths.output_stem("My Talk", "youtube-abc") == "My_Talk-youtube-abc" + + def test_strips_path_separators(self) -> None: + assert "/" not in paths.output_stem("a/b", "j") + + def test_collapses_runs_of_underscores(self) -> None: + assert paths.output_stem("a b", "j") == "a_b-j" + + def test_falls_back_to_the_id_when_there_is_no_title(self) -> None: + assert paths.output_stem(None, "youtube-abc") == "youtube-abc" + + def test_falls_back_when_the_title_is_all_punctuation(self) -> None: + assert paths.output_stem("///", "youtube-abc") == "youtube-abc" + + def test_truncates_a_very_long_title(self) -> None: + assert len(paths.output_stem("z" * 300, "j")) < 120 + + +def test_publish_with_nothing_to_copy_returns_empty(job: paths.JobPaths, tmp_path: Path) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + assert outputs_stage.publish(job, written, ("srt",), out_dir=tmp_path / "c", title="T") == []