Set up this workstation to run Whisper on the Intel Iris Xe iGPU via OpenVINO GenAI, and capture the setup steps and measured baseline. - README-intel.md: end-to-end setup for Intel GPUs on Linux — compute runtime install, render-node permissions, venv, pre-converted models, verification, and troubleshooting. - scripts/smoke.py: transcribes a clip on GPU and CPU, reports timings. - scripts/bench.py: 3 passes per device in an isolated process, also reporting CPU-time consumed to quantify offload. Measured on Iris Xe (80 EU) + i5-1145G7 with large-v3-turbo-int8 over 121s of audio: GPU ~23s (~5.2x realtime, 1.0 cores busy) vs CPU ~39s (~3.1x, 3.8 cores busy) — ~1.7x faster using ~6x less CPU time. Two findings recorded in the README because both silently mislead: Python 3.14 defaults multiprocessing to forkserver, so an unguarded script runs a second copy of itself concurrently and inflates timings; and short clips are dominated by fixed overhead, where GPU and CPU tie. No pipeline code yet — environment setup only. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
91 lines
3.3 KiB
Python
91 lines
3.3 KiB
Python
"""Throwaway smoke test: prove Whisper runs on the Intel iGPU via OpenVINO GenAI."""
|
|
import subprocess, sys, time
|
|
import numpy as np
|
|
from huggingface_hub import snapshot_download
|
|
import openvino as ov
|
|
import openvino_genai as ov_genai
|
|
|
|
MODEL = "OpenVINO/whisper-large-v3-turbo-int8-ov"
|
|
AUDIO = sys.argv[1] if len(sys.argv) > 1 else "sample.wav"
|
|
SR = 16000
|
|
|
|
|
|
def decode(path):
|
|
"""ffmpeg -> 16 kHz mono float32, no librosa/numba needed."""
|
|
raw = subprocess.run(
|
|
["ffmpeg", "-nostdin", "-loglevel", "error", "-i", path,
|
|
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
|
|
capture_output=True, check=True).stdout
|
|
return np.frombuffer(raw, dtype=np.float32)
|
|
|
|
|
|
def make_pipe(model_dir, device, cache):
|
|
"""CACHE_DIR passthrough form is unconfirmed for WhisperPipeline; try each."""
|
|
try:
|
|
return ov_genai.WhisperPipeline(model_dir, device, CACHE_DIR=cache), "kwarg"
|
|
except Exception:
|
|
try:
|
|
return ov_genai.WhisperPipeline(model_dir, device, {"CACHE_DIR": cache}), "dict"
|
|
except Exception:
|
|
ov.Core().set_property(device, {"CACHE_DIR": cache})
|
|
return ov_genai.WhisperPipeline(model_dir, device), "set_property"
|
|
|
|
|
|
def run(model_dir, device, speech, audio_sec):
|
|
cache = f".ov_cache_{device.lower()}"
|
|
print(f"\n{'='*62}\n{device}\n{'='*62}", flush=True)
|
|
|
|
t = time.perf_counter(); pipe, form = make_pipe(model_dir, device, cache)
|
|
cold = time.perf_counter() - t
|
|
print(f" compile, cold cache : {cold:7.1f}s (CACHE_DIR form: {form})", flush=True)
|
|
|
|
t = time.perf_counter()
|
|
res = pipe.generate(speech, task="transcribe", return_timestamps=True)
|
|
gen = time.perf_counter() - t
|
|
print(f" generate : {gen:7.1f}s RTF {audio_sec/gen:.2f}x realtime", flush=True)
|
|
|
|
del pipe
|
|
t = time.perf_counter(); pipe, _ = make_pipe(model_dir, device, cache)
|
|
warm = time.perf_counter() - t
|
|
print(f" compile, warm cache : {warm:7.1f}s", flush=True)
|
|
del pipe
|
|
|
|
return {"device": device, "cold": cold, "warm": warm, "gen": gen,
|
|
"rtf": audio_sec / gen, "text": str(res)}
|
|
|
|
|
|
def main():
|
|
print("OpenVINO", ov.__version__)
|
|
devices = ov.Core().available_devices
|
|
print("available_devices:", devices)
|
|
|
|
model_dir = snapshot_download(MODEL)
|
|
print("model:", model_dir)
|
|
|
|
speech = decode(AUDIO)
|
|
audio_sec = len(speech) / SR
|
|
print(f"audio: {AUDIO} {audio_sec:.1f}s ({len(speech):,} samples)")
|
|
|
|
targets = [d for d in ("GPU", "CPU") if d in devices]
|
|
if "GPU" not in devices:
|
|
print("\n!! GPU NOT PRESENT -- intel-compute-runtime not installed yet")
|
|
|
|
results = [run(model_dir, d, speech, audio_sec) for d in targets]
|
|
|
|
print(f"\n{'='*62}\nTRANSCRIPT ({results[0]['device']}), first 400 chars\n{'='*62}")
|
|
print(results[0]["text"][:400])
|
|
|
|
print(f"\n{'='*62}\nSUMMARY ({audio_sec:.0f}s of audio)\n{'='*62}")
|
|
print(f"{'device':<8}{'cold':>9}{'warm':>9}{'generate':>11}{'RTF':>9}")
|
|
for r in results:
|
|
print(f"{r['device']:<8}{r['cold']:>8.1f}s{r['warm']:>8.1f}s"
|
|
f"{r['gen']:>10.1f}s{r['rtf']:>8.2f}x")
|
|
if len(results) == 2:
|
|
g, c = results[0]["gen"], results[1]["gen"]
|
|
verdict = f"GPU {c/g:.2f}x faster" if g < c else f"GPU {g/c:.2f}x SLOWER"
|
|
print(f"\n -> {verdict} than CPU on generate")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|