Add Intel GPU setup docs and transcription benchmark scripts

Set up this workstation to run Whisper on the Intel Iris Xe iGPU via
OpenVINO GenAI, and capture the setup steps and measured baseline.

- README-intel.md: end-to-end setup for Intel GPUs on Linux — compute
  runtime install, render-node permissions, venv, pre-converted models,
  verification, and troubleshooting.
- scripts/smoke.py: transcribes a clip on GPU and CPU, reports timings.
- scripts/bench.py: 3 passes per device in an isolated process, also
  reporting CPU-time consumed to quantify offload.

Measured on Iris Xe (80 EU) + i5-1145G7 with large-v3-turbo-int8 over
121s of audio: GPU ~23s (~5.2x realtime, 1.0 cores busy) vs CPU ~39s
(~3.1x, 3.8 cores busy) — ~1.7x faster using ~6x less CPU time.

Two findings recorded in the README because both silently mislead:
Python 3.14 defaults multiprocessing to forkserver, so an unguarded
script runs a second copy of itself concurrently and inflates timings;
and short clips are dominated by fixed overhead, where GPU and CPU tie.

No pipeline code yet — environment setup only.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-13 13:06:53 -05:00
co-authored by Claude Opus 5
commit 35b8adfa98
4 changed files with 344 additions and 0 deletions
+35
View File
@@ -0,0 +1,35 @@
"""One device per process: wall time + CPU time consumed (offload evidence)."""
import subprocess, sys, time, resource
import numpy as np
from huggingface_hub import snapshot_download
import openvino_genai as ov_genai
SR = 16000
def main():
dev = sys.argv[1]
audio = sys.argv[2] if len(sys.argv) > 2 else "sample.wav"
raw = subprocess.run(["ffmpeg", "-nostdin", "-loglevel", "error", "-i", audio,
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
capture_output=True, check=True).stdout
speech = np.frombuffer(raw, dtype=np.float32)
secs = len(speech) / SR
pipe = ov_genai.WhisperPipeline(
snapshot_download("OpenVINO/whisper-large-v3-turbo-int8-ov"),
dev, CACHE_DIR=f".ov_cache_{dev.lower()}")
for i in range(3):
r0 = resource.getrusage(resource.RUSAGE_SELF)
t = time.perf_counter()
pipe.generate(speech, task="transcribe", return_timestamps=True)
wall = time.perf_counter() - t
r1 = resource.getrusage(resource.RUSAGE_SELF)
cpu = (r1.ru_utime - r0.ru_utime) + (r1.ru_stime - r0.ru_stime)
print(f"{dev} pass{i+1}: wall {wall:6.1f}s RTF {secs/wall:.2f}x "
f"cpu-time {cpu:6.1f}s ({cpu/wall:4.1f} cores busy)", flush=True)
if __name__ == "__main__": # forkserver re-imports __main__; guard is required
main()
+90
View File
@@ -0,0 +1,90 @@
"""Throwaway smoke test: prove Whisper runs on the Intel iGPU via OpenVINO GenAI."""
import subprocess, sys, time
import numpy as np
from huggingface_hub import snapshot_download
import openvino as ov
import openvino_genai as ov_genai
MODEL = "OpenVINO/whisper-large-v3-turbo-int8-ov"
AUDIO = sys.argv[1] if len(sys.argv) > 1 else "sample.wav"
SR = 16000
def decode(path):
"""ffmpeg -> 16 kHz mono float32, no librosa/numba needed."""
raw = subprocess.run(
["ffmpeg", "-nostdin", "-loglevel", "error", "-i", path,
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
capture_output=True, check=True).stdout
return np.frombuffer(raw, dtype=np.float32)
def make_pipe(model_dir, device, cache):
"""CACHE_DIR passthrough form is unconfirmed for WhisperPipeline; try each."""
try:
return ov_genai.WhisperPipeline(model_dir, device, CACHE_DIR=cache), "kwarg"
except Exception:
try:
return ov_genai.WhisperPipeline(model_dir, device, {"CACHE_DIR": cache}), "dict"
except Exception:
ov.Core().set_property(device, {"CACHE_DIR": cache})
return ov_genai.WhisperPipeline(model_dir, device), "set_property"
def run(model_dir, device, speech, audio_sec):
cache = f".ov_cache_{device.lower()}"
print(f"\n{'='*62}\n{device}\n{'='*62}", flush=True)
t = time.perf_counter(); pipe, form = make_pipe(model_dir, device, cache)
cold = time.perf_counter() - t
print(f" compile, cold cache : {cold:7.1f}s (CACHE_DIR form: {form})", flush=True)
t = time.perf_counter()
res = pipe.generate(speech, task="transcribe", return_timestamps=True)
gen = time.perf_counter() - t
print(f" generate : {gen:7.1f}s RTF {audio_sec/gen:.2f}x realtime", flush=True)
del pipe
t = time.perf_counter(); pipe, _ = make_pipe(model_dir, device, cache)
warm = time.perf_counter() - t
print(f" compile, warm cache : {warm:7.1f}s", flush=True)
del pipe
return {"device": device, "cold": cold, "warm": warm, "gen": gen,
"rtf": audio_sec / gen, "text": str(res)}
def main():
print("OpenVINO", ov.__version__)
devices = ov.Core().available_devices
print("available_devices:", devices)
model_dir = snapshot_download(MODEL)
print("model:", model_dir)
speech = decode(AUDIO)
audio_sec = len(speech) / SR
print(f"audio: {AUDIO} {audio_sec:.1f}s ({len(speech):,} samples)")
targets = [d for d in ("GPU", "CPU") if d in devices]
if "GPU" not in devices:
print("\n!! GPU NOT PRESENT -- intel-compute-runtime not installed yet")
results = [run(model_dir, d, speech, audio_sec) for d in targets]
print(f"\n{'='*62}\nTRANSCRIPT ({results[0]['device']}), first 400 chars\n{'='*62}")
print(results[0]["text"][:400])
print(f"\n{'='*62}\nSUMMARY ({audio_sec:.0f}s of audio)\n{'='*62}")
print(f"{'device':<8}{'cold':>9}{'warm':>9}{'generate':>11}{'RTF':>9}")
for r in results:
print(f"{r['device']:<8}{r['cold']:>8.1f}s{r['warm']:>8.1f}s"
f"{r['gen']:>10.1f}s{r['rtf']:>8.2f}x")
if len(results) == 2:
g, c = results[0]["gen"], results[1]["gen"]
verdict = f"GPU {c/g:.2f}x faster" if g < c else f"GPU {g/c:.2f}x SLOWER"
print(f"\n -> {verdict} than CPU on generate")
if __name__ == "__main__":
main()