From 35b8adfa9879616742ab92f823ebe2e1220dbfcb Mon Sep 17 00:00:00 2001 From: Jason Ross Date: Sun, 13 Sep 2026 13:06:53 -0500 Subject: [PATCH] Add Intel GPU setup docs and transcription benchmark scripts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Set up this workstation to run Whisper on the Intel Iris Xe iGPU via OpenVINO GenAI, and capture the setup steps and measured baseline. - README-intel.md: end-to-end setup for Intel GPUs on Linux — compute runtime install, render-node permissions, venv, pre-converted models, verification, and troubleshooting. - scripts/smoke.py: transcribes a clip on GPU and CPU, reports timings. - scripts/bench.py: 3 passes per device in an isolated process, also reporting CPU-time consumed to quantify offload. Measured on Iris Xe (80 EU) + i5-1145G7 with large-v3-turbo-int8 over 121s of audio: GPU ~23s (~5.2x realtime, 1.0 cores busy) vs CPU ~39s (~3.1x, 3.8 cores busy) — ~1.7x faster using ~6x less CPU time. Two findings recorded in the README because both silently mislead: Python 3.14 defaults multiprocessing to forkserver, so an unguarded script runs a second copy of itself concurrently and inflates timings; and short clips are dominated by fixed overhead, where GPU and CPU tie. No pipeline code yet — environment setup only. Co-Authored-By: Claude Opus 5 (1M context) --- .gitignore | 10 +++ README-intel.md | 209 +++++++++++++++++++++++++++++++++++++++++++++++ scripts/bench.py | 35 ++++++++ scripts/smoke.py | 90 ++++++++++++++++++++ 4 files changed, 344 insertions(+) create mode 100644 .gitignore create mode 100644 README-intel.md create mode 100644 scripts/bench.py create mode 100644 scripts/smoke.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..00599be --- /dev/null +++ b/.gitignore @@ -0,0 +1,10 @@ +# Python environment +.venv/ +__pycache__/ +*.py[cod] + +# OpenVINO compiled-kernel caches (scripts create .ov_cache_gpu / .ov_cache_cpu) +.ov_cache*/ + +# Test audio fixtures - fetched by scripts, not committed +*.wav diff --git a/README-intel.md b/README-intel.md new file mode 100644 index 0000000..bc453e8 --- /dev/null +++ b/README-intel.md @@ -0,0 +1,209 @@ +# Running Whisper on an Intel GPU (Linux) + +Setup for GPU-accelerated Whisper transcription on Intel integrated or discrete +graphics, using [OpenVINO GenAI](https://github.com/openvinotoolkit/openvino.genai). + +Verified on: Fedora 44, kernel 7.1, Intel Iris Xe (Tiger Lake GT2, Gen12LP, 80 EU), +i5-1145G7, Python 3.14.7, OpenVINO 2026.3.1. + +--- + +## 1. Prerequisites + +| Requirement | Notes | +|---|---| +| Intel GPU, Gen9 or newer | OpenVINO's GPU plugin targets Gen9-Gen12LP and Arc. Gen8/9/11 need the `legacy1` runtime packages instead of the mainline ones. | +| Python 3.10+ | 3.14 works; `openvino-genai` ships cp310-cp314 wheels. | +| `ffmpeg` | Used to decode audio. Avoids a `librosa`/`numba` dependency. | + +Check which GPU you have and that a kernel driver is bound: + +```bash +lspci -nn | grep -Ei 'vga|display' # e.g. Intel ... [8086:9a49] +lspci -k -s 00:02.0 | grep 'Kernel driver' # expect i915 (older) or xe (Lunar Lake+) +``` + +## 2. Install the Intel compute runtime + +This is the piece that's usually missing. OpenVINO's GPU plugin is OpenCL-based, so +without it OpenVINO sees only the CPU. The Mesa/Vulkan driver that ships with your +desktop is **not** sufficient. + +**Fedora / RHEL:** +```bash +sudo dnf install intel-compute-runtime oneapi-level-zero clinfo +``` + +**Ubuntu / Debian:** +```bash +sudo apt install intel-opencl-icd libze1 clinfo +``` + +**Arch:** +```bash +sudo pacman -S intel-compute-runtime level-zero-loader clinfo +``` + +> Package names other than the Fedora set are the usual spellings but were not +> verified here — check your distro's repos if they don't resolve. On older Intel +> hardware (Gen8/9/11) look for `-legacy1` variants. + +**Verify — this must list your GPU before anything else will work:** + +```bash +clinfo -l +# Platform #0: Intel(R) OpenCL Graphics +# `-- Device #0: Intel(R) Iris(R) Xe Graphics +``` + +### Device permissions + +Your user needs access to the render node. Check it: + +```bash +ls -l /dev/dri/renderD128 +``` + +If the mode is `0660` (rather than `0666`), add yourself to the owning group and +log out and back in: + +```bash +sudo usermod -aG render $USER # some distros use 'video' instead +``` + +## 3. Python environment + +```bash +python3 -m venv .venv && source .venv/bin/activate +pip install openvino-genai huggingface-hub numpy +``` + +Install from PyPI, **not** your distro's `python3-openvino` package — distro builds +tend to lag several releases behind and will conflict with the pip install. + +**Verify the plugin sees the GPU:** + +```bash +python -c "import openvino as ov; print(ov.Core().available_devices)" +# ['CPU', 'GPU'] +``` + +If `clinfo -l` succeeded but `GPU` is missing here, the problem is the OpenVINO +plugin rather than the driver — the two checks are deliberately separate. + +## 4. Get a model + +Use Intel's pre-converted models; this avoids pulling `torch`, `optimum-intel` and +`nncf` (several GB) just to convert weights: + +```bash +hf download OpenVINO/whisper-large-v3-turbo-int8-ov +``` + +Other sizes exist under the same org — `whisper-{tiny,base,small,medium,large-v3}` +and `distil-whisper-large-v3`, each in `fp16`, `int8` and `int4`. The `int8` builds +are weight-only compressed; on an iGPU the win is memory bandwidth, which is the +binding constraint when the GPU shares system RAM with the CPU. + +## 5. Verify end to end + +Fetch a test clip and loop it so long-form (>30 s) chunking is exercised: + +```bash +curl -sLO https://raw.githubusercontent.com/ggml-org/whisper.cpp/master/samples/jfk.wav +ffmpeg -stream_loop 10 -i jfk.wav -ar 16000 -ac 1 sample.wav # ~2 min +``` + +```bash +python scripts/smoke.py sample.wav # transcript + GPU/CPU timings +python scripts/bench.py GPU sample.wav # 3 passes, one device per process +python scripts/bench.py CPU sample.wav +``` + +### Expected results + +Measured on Iris Xe (80 EU) + i5-1145G7, `large-v3-turbo-int8`, 121 s of audio: + +| device | wall | realtime factor | CPU-time | cores busy | +|---|---|---|---|---| +| GPU | ~23 s | ~5.2x | ~24 s | 1.0 | +| CPU | ~39 s | ~3.1x | ~147 s | 3.8 | + +The iGPU is ~1.7x faster and uses ~6x less CPU time. Note it still occupies **one +full core** — the plugin busy-waits — so the offload is 3.8 cores down to 1.0, not +to zero. Budget for that if transcription runs alongside other CPU work. + +Don't expect Arc-class numbers from an integrated part: Xe-LP has no XMX matrix +engines, so the gain comes from bandwidth and offload rather than raw matrix +throughput. + +**Benchmark with at least a minute of audio.** On a short clip the fixed per-call +overhead dominates and the GPU advantage disappears entirely - on an 11 s sample the +same hardware measured 7.9 s on GPU vs 7.8 s on CPU, a dead heat, versus the 1.7x +gap on 121 s. Short-clip numbers will tell you the GPU is pointless when it isn't. + +## 6. Usage notes + +**Enable the kernel cache.** The GPU plugin JIT-compiles OpenCL kernels; `CACHE_DIR` +works as a plain keyword argument: + +```python +pipe = ov_genai.WhisperPipeline(model_dir, "GPU", CACHE_DIR=".ov_cache_gpu") +result = pipe.generate(speech, task="transcribe", return_timestamps=True) +for c in result.chunks: + print(c.start_ts, c.end_ts, c.text) +``` + +Model load is ~7 s cold and ~0.7 s cached. Separately, the **first transcription** +on a fresh cache costs ~31 s vs ~23 s steady-state, because shape-specific kernels +are compiled lazily at first inference rather than at load. A slow first run is +expected, not a fault. + +**Audio must be 16 kHz mono float32.** Decode with ffmpeg rather than a Python audio +library: + +```python +raw = subprocess.run(["ffmpeg", "-nostdin", "-loglevel", "error", "-i", path, + "-f", "f32le", "-ac", "1", "-ar", "16000", "-"], + capture_output=True, check=True).stdout +speech = np.frombuffer(raw, dtype=np.float32) +``` + +Long-form audio needs no chunking code of your own — `WhisperPipeline` applies a +sequential 30-second sliding window internally. + +**Always guard your entrypoints:** + +```python +if __name__ == "__main__": + main() +``` + +Python 3.14 changed the default multiprocessing start method on Linux from `fork` to +`forkserver`, which re-imports `__main__`. Without the guard, a script whose +dependencies touch multiprocessing runs a **second copy of itself concurrently**. It +fails silently — the symptom is duplicated output lines and timings inflated by +self-contention. This cost us a benchmark that reported the GPU 3x slower than it is. + +## 7. Troubleshooting + +| Symptom | Cause | +|---|---| +| `clinfo -l` lists no platform | Compute runtime not installed, or too old for your GPU. | +| `clinfo` works, OpenVINO shows only `['CPU']` | Plugin problem, not a driver problem. Check you're in the venv and not shadowed by a distro `python3-openvino`. | +| `Permission denied` on `/dev/dri/renderD128` | Not in the `render` (or `video`) group; needs a re-login. | +| First run very slow, later runs fine | Kernel JIT populating `CACHE_DIR`. Expected. | +| Output lines appear twice; everything slow | Missing `if __name__ == "__main__"` guard — see above. | + +## Alternative: whisper.cpp + Vulkan + +If the OpenCL stack won't cooperate, whisper.cpp's Vulkan backend needs no Intel +runtime at all — the Mesa driver most desktops already have is enough: + +```bash +git clone https://github.com/ggml-org/whisper.cpp && cd whisper.cpp +cmake -B build -DGGML_VULKAN=ON && cmake --build build -j$(nproc) +``` + +Generally slower than OpenVINO on Intel hardware, but a useful fallback and an +independent check on your numbers. diff --git a/scripts/bench.py b/scripts/bench.py new file mode 100644 index 0000000..ec13bbb --- /dev/null +++ b/scripts/bench.py @@ -0,0 +1,35 @@ +"""One device per process: wall time + CPU time consumed (offload evidence).""" +import subprocess, sys, time, resource +import numpy as np +from huggingface_hub import snapshot_download +import openvino_genai as ov_genai + +SR = 16000 + + +def main(): + dev = sys.argv[1] + audio = sys.argv[2] if len(sys.argv) > 2 else "sample.wav" + raw = subprocess.run(["ffmpeg", "-nostdin", "-loglevel", "error", "-i", audio, + "-f", "f32le", "-ac", "1", "-ar", str(SR), "-"], + capture_output=True, check=True).stdout + speech = np.frombuffer(raw, dtype=np.float32) + secs = len(speech) / SR + + pipe = ov_genai.WhisperPipeline( + snapshot_download("OpenVINO/whisper-large-v3-turbo-int8-ov"), + dev, CACHE_DIR=f".ov_cache_{dev.lower()}") + + for i in range(3): + r0 = resource.getrusage(resource.RUSAGE_SELF) + t = time.perf_counter() + pipe.generate(speech, task="transcribe", return_timestamps=True) + wall = time.perf_counter() - t + r1 = resource.getrusage(resource.RUSAGE_SELF) + cpu = (r1.ru_utime - r0.ru_utime) + (r1.ru_stime - r0.ru_stime) + print(f"{dev} pass{i+1}: wall {wall:6.1f}s RTF {secs/wall:.2f}x " + f"cpu-time {cpu:6.1f}s ({cpu/wall:4.1f} cores busy)", flush=True) + + +if __name__ == "__main__": # forkserver re-imports __main__; guard is required + main() diff --git a/scripts/smoke.py b/scripts/smoke.py new file mode 100644 index 0000000..a6d373f --- /dev/null +++ b/scripts/smoke.py @@ -0,0 +1,90 @@ +"""Throwaway smoke test: prove Whisper runs on the Intel iGPU via OpenVINO GenAI.""" +import subprocess, sys, time +import numpy as np +from huggingface_hub import snapshot_download +import openvino as ov +import openvino_genai as ov_genai + +MODEL = "OpenVINO/whisper-large-v3-turbo-int8-ov" +AUDIO = sys.argv[1] if len(sys.argv) > 1 else "sample.wav" +SR = 16000 + + +def decode(path): + """ffmpeg -> 16 kHz mono float32, no librosa/numba needed.""" + raw = subprocess.run( + ["ffmpeg", "-nostdin", "-loglevel", "error", "-i", path, + "-f", "f32le", "-ac", "1", "-ar", str(SR), "-"], + capture_output=True, check=True).stdout + return np.frombuffer(raw, dtype=np.float32) + + +def make_pipe(model_dir, device, cache): + """CACHE_DIR passthrough form is unconfirmed for WhisperPipeline; try each.""" + try: + return ov_genai.WhisperPipeline(model_dir, device, CACHE_DIR=cache), "kwarg" + except Exception: + try: + return ov_genai.WhisperPipeline(model_dir, device, {"CACHE_DIR": cache}), "dict" + except Exception: + ov.Core().set_property(device, {"CACHE_DIR": cache}) + return ov_genai.WhisperPipeline(model_dir, device), "set_property" + + +def run(model_dir, device, speech, audio_sec): + cache = f".ov_cache_{device.lower()}" + print(f"\n{'='*62}\n{device}\n{'='*62}", flush=True) + + t = time.perf_counter(); pipe, form = make_pipe(model_dir, device, cache) + cold = time.perf_counter() - t + print(f" compile, cold cache : {cold:7.1f}s (CACHE_DIR form: {form})", flush=True) + + t = time.perf_counter() + res = pipe.generate(speech, task="transcribe", return_timestamps=True) + gen = time.perf_counter() - t + print(f" generate : {gen:7.1f}s RTF {audio_sec/gen:.2f}x realtime", flush=True) + + del pipe + t = time.perf_counter(); pipe, _ = make_pipe(model_dir, device, cache) + warm = time.perf_counter() - t + print(f" compile, warm cache : {warm:7.1f}s", flush=True) + del pipe + + return {"device": device, "cold": cold, "warm": warm, "gen": gen, + "rtf": audio_sec / gen, "text": str(res)} + + +def main(): + print("OpenVINO", ov.__version__) + devices = ov.Core().available_devices + print("available_devices:", devices) + + model_dir = snapshot_download(MODEL) + print("model:", model_dir) + + speech = decode(AUDIO) + audio_sec = len(speech) / SR + print(f"audio: {AUDIO} {audio_sec:.1f}s ({len(speech):,} samples)") + + targets = [d for d in ("GPU", "CPU") if d in devices] + if "GPU" not in devices: + print("\n!! GPU NOT PRESENT -- intel-compute-runtime not installed yet") + + results = [run(model_dir, d, speech, audio_sec) for d in targets] + + print(f"\n{'='*62}\nTRANSCRIPT ({results[0]['device']}), first 400 chars\n{'='*62}") + print(results[0]["text"][:400]) + + print(f"\n{'='*62}\nSUMMARY ({audio_sec:.0f}s of audio)\n{'='*62}") + print(f"{'device':<8}{'cold':>9}{'warm':>9}{'generate':>11}{'RTF':>9}") + for r in results: + print(f"{r['device']:<8}{r['cold']:>8.1f}s{r['warm']:>8.1f}s" + f"{r['gen']:>10.1f}s{r['rtf']:>8.2f}x") + if len(results) == 2: + g, c = results[0]["gen"], results[1]["gen"] + verdict = f"GPU {c/g:.2f}x faster" if g < c else f"GPU {g/c:.2f}x SLOWER" + print(f"\n -> {verdict} than CPU on generate") + + +if __name__ == "__main__": + main()