feat(outputs): implement --out-dir, which was accepted but did nothing
The flag was parsed into RunConfig and never read, so it silently had no effect. Finished transcripts are now also copied into that folder as <title>-<job_id>.<ext>, which is what makes a batch readable without digging through job directories. It copies rather than moves, and only the formats actually requested: the job directory stays canonical because resume depends on it, and transcript.json is written as the segment cache whether or not it was asked for. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -228,6 +228,7 @@ def _do_transcript(
|
||||
|
||||
state.outputs = Outputs(status=ArtifactStatus.PRESENT, paths=written)
|
||||
store.save(job, state)
|
||||
outputs_stage.publish(job, written, config.formats, out_dir=config.out_dir, title=state.title)
|
||||
if config.no_retain_audio:
|
||||
_record_deletion(job, state, "audio", "--no-retain-audio", job.audio_file)
|
||||
|
||||
|
||||
@@ -50,6 +50,13 @@ def url_job_id(url: str) -> str:
|
||||
return f"url-{hashlib.sha256(normalize_url(url).encode()).hexdigest()[:16]}"
|
||||
|
||||
|
||||
def output_stem(title: str | None, identifier: str) -> str:
|
||||
"""A readable, filesystem-safe basename for a collected transcript."""
|
||||
cleaned = _UNSAFE.sub("_", (title or "").strip()).strip("_.")
|
||||
cleaned = re.sub(r"_{2,}", "_", cleaned)[:80].strip("_.")
|
||||
return f"{cleaned}-{identifier}" if cleaned else identifier
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class JobPaths:
|
||||
root: Path
|
||||
|
||||
@@ -4,12 +4,15 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from ccn_transcribe import formats
|
||||
from ccn_transcribe.paths import output_stem
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Mapping, Sequence
|
||||
from pathlib import Path
|
||||
|
||||
from ccn_transcribe.paths import JobPaths
|
||||
from ccn_transcribe.transcript import TranscriptResult
|
||||
@@ -54,3 +57,33 @@ def write(
|
||||
written[name] = str(path.relative_to(job.root))
|
||||
log.info("wrote %s", ", ".join(sorted(written)))
|
||||
return written
|
||||
|
||||
|
||||
def publish(
|
||||
job: JobPaths,
|
||||
written: Mapping[str, str],
|
||||
wanted: Sequence[str],
|
||||
*,
|
||||
out_dir: Path | None,
|
||||
title: str | None,
|
||||
) -> list[Path]:
|
||||
"""Additionally collect transcripts into one folder, named by title.
|
||||
|
||||
The job directory stays canonical -- resume depends on it -- so this only
|
||||
copies, and only the formats the user actually asked for.
|
||||
"""
|
||||
if out_dir is None:
|
||||
return []
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
stem = output_stem(title, job.root.name)
|
||||
copied: list[Path] = []
|
||||
for name in wanted:
|
||||
relative = written.get(name)
|
||||
if relative is None:
|
||||
continue
|
||||
destination = out_dir / f"{stem}.{name}"
|
||||
shutil.copy2(job.root / relative, destination)
|
||||
copied.append(destination)
|
||||
if copied:
|
||||
log.info("collected %d file(s) into %s", len(copied), out_dir)
|
||||
return copied
|
||||
|
||||
@@ -199,3 +199,74 @@ class TestOutputsStage:
|
||||
)
|
||||
outputs_stage.write(job, shorter, ("txt",))
|
||||
assert job.transcript_file("txt").read_text().strip() == "hi"
|
||||
|
||||
|
||||
class TestPublish:
|
||||
def test_does_nothing_without_an_out_dir(self, job: paths.JobPaths) -> None:
|
||||
written = outputs_stage.write(job, RESULT, ("txt",))
|
||||
assert outputs_stage.publish(job, written, ("txt",), out_dir=None, title="T") == []
|
||||
|
||||
def test_copies_requested_formats_named_by_title(
|
||||
self, job: paths.JobPaths, tmp_path: Path
|
||||
) -> None:
|
||||
written = outputs_stage.write(job, RESULT, ("txt", "srt"))
|
||||
collected = tmp_path / "collected"
|
||||
got = outputs_stage.publish(
|
||||
job, written, ("txt", "srt"), out_dir=collected, title="My Talk"
|
||||
)
|
||||
assert {p.name for p in got} == {
|
||||
"My_Talk-youtube-abc.txt",
|
||||
"My_Talk-youtube-abc.srt",
|
||||
}
|
||||
assert all(p.read_text() for p in got)
|
||||
|
||||
def test_the_job_directory_is_left_canonical(self, job: paths.JobPaths, tmp_path: Path) -> None:
|
||||
# Resume depends on the job dir, so publishing copies rather than moves.
|
||||
written = outputs_stage.write(job, RESULT, ("txt",))
|
||||
outputs_stage.publish(job, written, ("txt",), out_dir=tmp_path / "c", title="T")
|
||||
assert job.transcript_file("txt").exists()
|
||||
|
||||
def test_only_requested_formats_are_collected(
|
||||
self, job: paths.JobPaths, tmp_path: Path
|
||||
) -> None:
|
||||
# json is always written as the segment cache, but need not be collected.
|
||||
written = outputs_stage.write(job, RESULT, ("txt",))
|
||||
collected = tmp_path / "collected"
|
||||
got = outputs_stage.publish(job, written, ("txt",), out_dir=collected, title="T")
|
||||
assert [p.suffix for p in got] == [".txt"]
|
||||
|
||||
def test_a_missing_format_is_skipped(self, job: paths.JobPaths, tmp_path: Path) -> None:
|
||||
written = outputs_stage.write(job, RESULT, ("txt",))
|
||||
got = outputs_stage.publish(job, written, ("txt", "srt"), out_dir=tmp_path / "c", title="T")
|
||||
assert len(got) == 1
|
||||
|
||||
def test_creates_the_directory(self, job: paths.JobPaths, tmp_path: Path) -> None:
|
||||
written = outputs_stage.write(job, RESULT, ("txt",))
|
||||
target = tmp_path / "deep" / "nested"
|
||||
outputs_stage.publish(job, written, ("txt",), out_dir=target, title="T")
|
||||
assert target.is_dir()
|
||||
|
||||
|
||||
class TestOutputStem:
|
||||
def test_combines_title_and_id(self) -> None:
|
||||
assert paths.output_stem("My Talk", "youtube-abc") == "My_Talk-youtube-abc"
|
||||
|
||||
def test_strips_path_separators(self) -> None:
|
||||
assert "/" not in paths.output_stem("a/b", "j")
|
||||
|
||||
def test_collapses_runs_of_underscores(self) -> None:
|
||||
assert paths.output_stem("a b", "j") == "a_b-j"
|
||||
|
||||
def test_falls_back_to_the_id_when_there_is_no_title(self) -> None:
|
||||
assert paths.output_stem(None, "youtube-abc") == "youtube-abc"
|
||||
|
||||
def test_falls_back_when_the_title_is_all_punctuation(self) -> None:
|
||||
assert paths.output_stem("///", "youtube-abc") == "youtube-abc"
|
||||
|
||||
def test_truncates_a_very_long_title(self) -> None:
|
||||
assert len(paths.output_stem("z" * 300, "j")) < 120
|
||||
|
||||
|
||||
def test_publish_with_nothing_to_copy_returns_empty(job: paths.JobPaths, tmp_path: Path) -> None:
|
||||
written = outputs_stage.write(job, RESULT, ("txt",))
|
||||
assert outputs_stage.publish(job, written, ("srt",), out_dir=tmp_path / "c", title="T") == []
|
||||
|
||||
Reference in New Issue
Block a user