feat(outputs): implement --out-dir, which was accepted but did nothing

The flag was parsed into RunConfig and never read, so it silently had no
effect. Finished transcripts are now also copied into that folder as
<title>-<job_id>.<ext>, which is what makes a batch readable without
digging through job directories.

It copies rather than moves, and only the formats actually requested: the
job directory stays canonical because resume depends on it, and
transcript.json is written as the segment cache whether or not it was asked
for.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-13 15:51:45 -05:00
co-authored by Claude Opus 5
parent 3ac070468a
commit f32516d5d5
4 changed files with 112 additions and 0 deletions
+1
View File
@@ -228,6 +228,7 @@ def _do_transcript(
state.outputs = Outputs(status=ArtifactStatus.PRESENT, paths=written)
store.save(job, state)
outputs_stage.publish(job, written, config.formats, out_dir=config.out_dir, title=state.title)
if config.no_retain_audio:
_record_deletion(job, state, "audio", "--no-retain-audio", job.audio_file)
+7
View File
@@ -50,6 +50,13 @@ def url_job_id(url: str) -> str:
return f"url-{hashlib.sha256(normalize_url(url).encode()).hexdigest()[:16]}"
def output_stem(title: str | None, identifier: str) -> str:
"""A readable, filesystem-safe basename for a collected transcript."""
cleaned = _UNSAFE.sub("_", (title or "").strip()).strip("_.")
cleaned = re.sub(r"_{2,}", "_", cleaned)[:80].strip("_.")
return f"{cleaned}-{identifier}" if cleaned else identifier
@dataclass(frozen=True, slots=True)
class JobPaths:
root: Path
+33
View File
@@ -4,12 +4,15 @@ from __future__ import annotations
import logging
import os
import shutil
from typing import TYPE_CHECKING
from ccn_transcribe import formats
from ccn_transcribe.paths import output_stem
if TYPE_CHECKING:
from collections.abc import Mapping, Sequence
from pathlib import Path
from ccn_transcribe.paths import JobPaths
from ccn_transcribe.transcript import TranscriptResult
@@ -54,3 +57,33 @@ def write(
written[name] = str(path.relative_to(job.root))
log.info("wrote %s", ", ".join(sorted(written)))
return written
def publish(
job: JobPaths,
written: Mapping[str, str],
wanted: Sequence[str],
*,
out_dir: Path | None,
title: str | None,
) -> list[Path]:
"""Additionally collect transcripts into one folder, named by title.
The job directory stays canonical -- resume depends on it -- so this only
copies, and only the formats the user actually asked for.
"""
if out_dir is None:
return []
out_dir.mkdir(parents=True, exist_ok=True)
stem = output_stem(title, job.root.name)
copied: list[Path] = []
for name in wanted:
relative = written.get(name)
if relative is None:
continue
destination = out_dir / f"{stem}.{name}"
shutil.copy2(job.root / relative, destination)
copied.append(destination)
if copied:
log.info("collected %d file(s) into %s", len(copied), out_dir)
return copied
+71
View File
@@ -199,3 +199,74 @@ class TestOutputsStage:
)
outputs_stage.write(job, shorter, ("txt",))
assert job.transcript_file("txt").read_text().strip() == "hi"
class TestPublish:
def test_does_nothing_without_an_out_dir(self, job: paths.JobPaths) -> None:
written = outputs_stage.write(job, RESULT, ("txt",))
assert outputs_stage.publish(job, written, ("txt",), out_dir=None, title="T") == []
def test_copies_requested_formats_named_by_title(
self, job: paths.JobPaths, tmp_path: Path
) -> None:
written = outputs_stage.write(job, RESULT, ("txt", "srt"))
collected = tmp_path / "collected"
got = outputs_stage.publish(
job, written, ("txt", "srt"), out_dir=collected, title="My Talk"
)
assert {p.name for p in got} == {
"My_Talk-youtube-abc.txt",
"My_Talk-youtube-abc.srt",
}
assert all(p.read_text() for p in got)
def test_the_job_directory_is_left_canonical(self, job: paths.JobPaths, tmp_path: Path) -> None:
# Resume depends on the job dir, so publishing copies rather than moves.
written = outputs_stage.write(job, RESULT, ("txt",))
outputs_stage.publish(job, written, ("txt",), out_dir=tmp_path / "c", title="T")
assert job.transcript_file("txt").exists()
def test_only_requested_formats_are_collected(
self, job: paths.JobPaths, tmp_path: Path
) -> None:
# json is always written as the segment cache, but need not be collected.
written = outputs_stage.write(job, RESULT, ("txt",))
collected = tmp_path / "collected"
got = outputs_stage.publish(job, written, ("txt",), out_dir=collected, title="T")
assert [p.suffix for p in got] == [".txt"]
def test_a_missing_format_is_skipped(self, job: paths.JobPaths, tmp_path: Path) -> None:
written = outputs_stage.write(job, RESULT, ("txt",))
got = outputs_stage.publish(job, written, ("txt", "srt"), out_dir=tmp_path / "c", title="T")
assert len(got) == 1
def test_creates_the_directory(self, job: paths.JobPaths, tmp_path: Path) -> None:
written = outputs_stage.write(job, RESULT, ("txt",))
target = tmp_path / "deep" / "nested"
outputs_stage.publish(job, written, ("txt",), out_dir=target, title="T")
assert target.is_dir()
class TestOutputStem:
def test_combines_title_and_id(self) -> None:
assert paths.output_stem("My Talk", "youtube-abc") == "My_Talk-youtube-abc"
def test_strips_path_separators(self) -> None:
assert "/" not in paths.output_stem("a/b", "j")
def test_collapses_runs_of_underscores(self) -> None:
assert paths.output_stem("a b", "j") == "a_b-j"
def test_falls_back_to_the_id_when_there_is_no_title(self) -> None:
assert paths.output_stem(None, "youtube-abc") == "youtube-abc"
def test_falls_back_when_the_title_is_all_punctuation(self) -> None:
assert paths.output_stem("///", "youtube-abc") == "youtube-abc"
def test_truncates_a_very_long_title(self) -> None:
assert len(paths.output_stem("z" * 300, "j")) < 120
def test_publish_with_nothing_to_copy_returns_empty(job: paths.JobPaths, tmp_path: Path) -> None:
written = outputs_stage.write(job, RESULT, ("txt",))
assert outputs_stage.publish(job, written, ("srt",), out_dir=tmp_path / "c", title="T") == []