diff --git a/src/ccn_transcribe/jobs/runner.py b/src/ccn_transcribe/jobs/runner.py index 9731e57..a840406 100644 --- a/src/ccn_transcribe/jobs/runner.py +++ b/src/ccn_transcribe/jobs/runner.py @@ -228,6 +228,7 @@ def _do_transcript( state.outputs = Outputs(status=ArtifactStatus.PRESENT, paths=written) store.save(job, state) + outputs_stage.publish(job, written, config.formats, out_dir=config.out_dir, title=state.title) if config.no_retain_audio: _record_deletion(job, state, "audio", "--no-retain-audio", job.audio_file) diff --git a/src/ccn_transcribe/paths.py b/src/ccn_transcribe/paths.py index 3f8dc76..671a1be 100644 --- a/src/ccn_transcribe/paths.py +++ b/src/ccn_transcribe/paths.py @@ -50,6 +50,13 @@ def url_job_id(url: str) -> str: return f"url-{hashlib.sha256(normalize_url(url).encode()).hexdigest()[:16]}" +def output_stem(title: str | None, identifier: str) -> str: + """A readable, filesystem-safe basename for a collected transcript.""" + cleaned = _UNSAFE.sub("_", (title or "").strip()).strip("_.") + cleaned = re.sub(r"_{2,}", "_", cleaned)[:80].strip("_.") + return f"{cleaned}-{identifier}" if cleaned else identifier + + @dataclass(frozen=True, slots=True) class JobPaths: root: Path diff --git a/src/ccn_transcribe/stages/outputs.py b/src/ccn_transcribe/stages/outputs.py index 3388a5c..10abd53 100644 --- a/src/ccn_transcribe/stages/outputs.py +++ b/src/ccn_transcribe/stages/outputs.py @@ -4,12 +4,15 @@ from __future__ import annotations import logging import os +import shutil from typing import TYPE_CHECKING from ccn_transcribe import formats +from ccn_transcribe.paths import output_stem if TYPE_CHECKING: from collections.abc import Mapping, Sequence + from pathlib import Path from ccn_transcribe.paths import JobPaths from ccn_transcribe.transcript import TranscriptResult @@ -54,3 +57,33 @@ def write( written[name] = str(path.relative_to(job.root)) log.info("wrote %s", ", ".join(sorted(written))) return written + + +def publish( + job: JobPaths, + written: Mapping[str, str], + wanted: Sequence[str], + *, + out_dir: Path | None, + title: str | None, +) -> list[Path]: + """Additionally collect transcripts into one folder, named by title. + + The job directory stays canonical -- resume depends on it -- so this only + copies, and only the formats the user actually asked for. + """ + if out_dir is None: + return [] + out_dir.mkdir(parents=True, exist_ok=True) + stem = output_stem(title, job.root.name) + copied: list[Path] = [] + for name in wanted: + relative = written.get(name) + if relative is None: + continue + destination = out_dir / f"{stem}.{name}" + shutil.copy2(job.root / relative, destination) + copied.append(destination) + if copied: + log.info("collected %d file(s) into %s", len(copied), out_dir) + return copied diff --git a/tests/test_stages_pipeline.py b/tests/test_stages_pipeline.py index 4e48024..f857349 100644 --- a/tests/test_stages_pipeline.py +++ b/tests/test_stages_pipeline.py @@ -199,3 +199,74 @@ class TestOutputsStage: ) outputs_stage.write(job, shorter, ("txt",)) assert job.transcript_file("txt").read_text().strip() == "hi" + + +class TestPublish: + def test_does_nothing_without_an_out_dir(self, job: paths.JobPaths) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + assert outputs_stage.publish(job, written, ("txt",), out_dir=None, title="T") == [] + + def test_copies_requested_formats_named_by_title( + self, job: paths.JobPaths, tmp_path: Path + ) -> None: + written = outputs_stage.write(job, RESULT, ("txt", "srt")) + collected = tmp_path / "collected" + got = outputs_stage.publish( + job, written, ("txt", "srt"), out_dir=collected, title="My Talk" + ) + assert {p.name for p in got} == { + "My_Talk-youtube-abc.txt", + "My_Talk-youtube-abc.srt", + } + assert all(p.read_text() for p in got) + + def test_the_job_directory_is_left_canonical(self, job: paths.JobPaths, tmp_path: Path) -> None: + # Resume depends on the job dir, so publishing copies rather than moves. + written = outputs_stage.write(job, RESULT, ("txt",)) + outputs_stage.publish(job, written, ("txt",), out_dir=tmp_path / "c", title="T") + assert job.transcript_file("txt").exists() + + def test_only_requested_formats_are_collected( + self, job: paths.JobPaths, tmp_path: Path + ) -> None: + # json is always written as the segment cache, but need not be collected. + written = outputs_stage.write(job, RESULT, ("txt",)) + collected = tmp_path / "collected" + got = outputs_stage.publish(job, written, ("txt",), out_dir=collected, title="T") + assert [p.suffix for p in got] == [".txt"] + + def test_a_missing_format_is_skipped(self, job: paths.JobPaths, tmp_path: Path) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + got = outputs_stage.publish(job, written, ("txt", "srt"), out_dir=tmp_path / "c", title="T") + assert len(got) == 1 + + def test_creates_the_directory(self, job: paths.JobPaths, tmp_path: Path) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + target = tmp_path / "deep" / "nested" + outputs_stage.publish(job, written, ("txt",), out_dir=target, title="T") + assert target.is_dir() + + +class TestOutputStem: + def test_combines_title_and_id(self) -> None: + assert paths.output_stem("My Talk", "youtube-abc") == "My_Talk-youtube-abc" + + def test_strips_path_separators(self) -> None: + assert "/" not in paths.output_stem("a/b", "j") + + def test_collapses_runs_of_underscores(self) -> None: + assert paths.output_stem("a b", "j") == "a_b-j" + + def test_falls_back_to_the_id_when_there_is_no_title(self) -> None: + assert paths.output_stem(None, "youtube-abc") == "youtube-abc" + + def test_falls_back_when_the_title_is_all_punctuation(self) -> None: + assert paths.output_stem("///", "youtube-abc") == "youtube-abc" + + def test_truncates_a_very_long_title(self) -> None: + assert len(paths.output_stem("z" * 300, "j")) < 120 + + +def test_publish_with_nothing_to_copy_returns_empty(job: paths.JobPaths, tmp_path: Path) -> None: + written = outputs_stage.write(job, RESULT, ("txt",)) + assert outputs_stage.publish(job, written, ("srt",), out_dir=tmp_path / "c", title="T") == []