Files
audio-scribe/scripts/smoke.py
T
JMR-devandClaude Opus 5 35b8adfa98 Add Intel GPU setup docs and transcription benchmark scripts
Set up this workstation to run Whisper on the Intel Iris Xe iGPU via
OpenVINO GenAI, and capture the setup steps and measured baseline.

- README-intel.md: end-to-end setup for Intel GPUs on Linux — compute
  runtime install, render-node permissions, venv, pre-converted models,
  verification, and troubleshooting.
- scripts/smoke.py: transcribes a clip on GPU and CPU, reports timings.
- scripts/bench.py: 3 passes per device in an isolated process, also
  reporting CPU-time consumed to quantify offload.

Measured on Iris Xe (80 EU) + i5-1145G7 with large-v3-turbo-int8 over
121s of audio: GPU ~23s (~5.2x realtime, 1.0 cores busy) vs CPU ~39s
(~3.1x, 3.8 cores busy) — ~1.7x faster using ~6x less CPU time.

Two findings recorded in the README because both silently mislead:
Python 3.14 defaults multiprocessing to forkserver, so an unguarded
script runs a second copy of itself concurrently and inflates timings;
and short clips are dominated by fixed overhead, where GPU and CPU tie.

No pipeline code yet — environment setup only.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-13 13:06:53 -05:00

91 lines
3.3 KiB
Python

"""Throwaway smoke test: prove Whisper runs on the Intel iGPU via OpenVINO GenAI."""
import subprocess, sys, time
import numpy as np
from huggingface_hub import snapshot_download
import openvino as ov
import openvino_genai as ov_genai
MODEL = "OpenVINO/whisper-large-v3-turbo-int8-ov"
AUDIO = sys.argv[1] if len(sys.argv) > 1 else "sample.wav"
SR = 16000
def decode(path):
"""ffmpeg -> 16 kHz mono float32, no librosa/numba needed."""
raw = subprocess.run(
["ffmpeg", "-nostdin", "-loglevel", "error", "-i", path,
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
capture_output=True, check=True).stdout
return np.frombuffer(raw, dtype=np.float32)
def make_pipe(model_dir, device, cache):
"""CACHE_DIR passthrough form is unconfirmed for WhisperPipeline; try each."""
try:
return ov_genai.WhisperPipeline(model_dir, device, CACHE_DIR=cache), "kwarg"
except Exception:
try:
return ov_genai.WhisperPipeline(model_dir, device, {"CACHE_DIR": cache}), "dict"
except Exception:
ov.Core().set_property(device, {"CACHE_DIR": cache})
return ov_genai.WhisperPipeline(model_dir, device), "set_property"
def run(model_dir, device, speech, audio_sec):
cache = f".ov_cache_{device.lower()}"
print(f"\n{'='*62}\n{device}\n{'='*62}", flush=True)
t = time.perf_counter(); pipe, form = make_pipe(model_dir, device, cache)
cold = time.perf_counter() - t
print(f" compile, cold cache : {cold:7.1f}s (CACHE_DIR form: {form})", flush=True)
t = time.perf_counter()
res = pipe.generate(speech, task="transcribe", return_timestamps=True)
gen = time.perf_counter() - t
print(f" generate : {gen:7.1f}s RTF {audio_sec/gen:.2f}x realtime", flush=True)
del pipe
t = time.perf_counter(); pipe, _ = make_pipe(model_dir, device, cache)
warm = time.perf_counter() - t
print(f" compile, warm cache : {warm:7.1f}s", flush=True)
del pipe
return {"device": device, "cold": cold, "warm": warm, "gen": gen,
"rtf": audio_sec / gen, "text": str(res)}
def main():
print("OpenVINO", ov.__version__)
devices = ov.Core().available_devices
print("available_devices:", devices)
model_dir = snapshot_download(MODEL)
print("model:", model_dir)
speech = decode(AUDIO)
audio_sec = len(speech) / SR
print(f"audio: {AUDIO} {audio_sec:.1f}s ({len(speech):,} samples)")
targets = [d for d in ("GPU", "CPU") if d in devices]
if "GPU" not in devices:
print("\n!! GPU NOT PRESENT -- intel-compute-runtime not installed yet")
results = [run(model_dir, d, speech, audio_sec) for d in targets]
print(f"\n{'='*62}\nTRANSCRIPT ({results[0]['device']}), first 400 chars\n{'='*62}")
print(results[0]["text"][:400])
print(f"\n{'='*62}\nSUMMARY ({audio_sec:.0f}s of audio)\n{'='*62}")
print(f"{'device':<8}{'cold':>9}{'warm':>9}{'generate':>11}{'RTF':>9}")
for r in results:
print(f"{r['device']:<8}{r['cold']:>8.1f}s{r['warm']:>8.1f}s"
f"{r['gen']:>10.1f}s{r['rtf']:>8.2f}x")
if len(results) == 2:
g, c = results[0]["gen"], results[1]["gen"]
verdict = f"GPU {c/g:.2f}x faster" if g < c else f"GPU {g/c:.2f}x SLOWER"
print(f"\n -> {verdict} than CPU on generate")
if __name__ == "__main__":
main()