Add Intel GPU setup docs and transcription benchmark scripts
Set up this workstation to run Whisper on the Intel Iris Xe iGPU via OpenVINO GenAI, and capture the setup steps and measured baseline. - README-intel.md: end-to-end setup for Intel GPUs on Linux — compute runtime install, render-node permissions, venv, pre-converted models, verification, and troubleshooting. - scripts/smoke.py: transcribes a clip on GPU and CPU, reports timings. - scripts/bench.py: 3 passes per device in an isolated process, also reporting CPU-time consumed to quantify offload. Measured on Iris Xe (80 EU) + i5-1145G7 with large-v3-turbo-int8 over 121s of audio: GPU ~23s (~5.2x realtime, 1.0 cores busy) vs CPU ~39s (~3.1x, 3.8 cores busy) — ~1.7x faster using ~6x less CPU time. Two findings recorded in the README because both silently mislead: Python 3.14 defaults multiprocessing to forkserver, so an unguarded script runs a second copy of itself concurrently and inflates timings; and short clips are dominated by fixed overhead, where GPU and CPU tie. No pipeline code yet — environment setup only. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,35 @@
|
||||
"""One device per process: wall time + CPU time consumed (offload evidence)."""
|
||||
import subprocess, sys, time, resource
|
||||
import numpy as np
|
||||
from huggingface_hub import snapshot_download
|
||||
import openvino_genai as ov_genai
|
||||
|
||||
SR = 16000
|
||||
|
||||
|
||||
def main():
|
||||
dev = sys.argv[1]
|
||||
audio = sys.argv[2] if len(sys.argv) > 2 else "sample.wav"
|
||||
raw = subprocess.run(["ffmpeg", "-nostdin", "-loglevel", "error", "-i", audio,
|
||||
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
|
||||
capture_output=True, check=True).stdout
|
||||
speech = np.frombuffer(raw, dtype=np.float32)
|
||||
secs = len(speech) / SR
|
||||
|
||||
pipe = ov_genai.WhisperPipeline(
|
||||
snapshot_download("OpenVINO/whisper-large-v3-turbo-int8-ov"),
|
||||
dev, CACHE_DIR=f".ov_cache_{dev.lower()}")
|
||||
|
||||
for i in range(3):
|
||||
r0 = resource.getrusage(resource.RUSAGE_SELF)
|
||||
t = time.perf_counter()
|
||||
pipe.generate(speech, task="transcribe", return_timestamps=True)
|
||||
wall = time.perf_counter() - t
|
||||
r1 = resource.getrusage(resource.RUSAGE_SELF)
|
||||
cpu = (r1.ru_utime - r0.ru_utime) + (r1.ru_stime - r0.ru_stime)
|
||||
print(f"{dev} pass{i+1}: wall {wall:6.1f}s RTF {secs/wall:.2f}x "
|
||||
f"cpu-time {cpu:6.1f}s ({cpu/wall:4.1f} cores busy)", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__": # forkserver re-imports __main__; guard is required
|
||||
main()
|
||||
@@ -0,0 +1,90 @@
|
||||
"""Throwaway smoke test: prove Whisper runs on the Intel iGPU via OpenVINO GenAI."""
|
||||
import subprocess, sys, time
|
||||
import numpy as np
|
||||
from huggingface_hub import snapshot_download
|
||||
import openvino as ov
|
||||
import openvino_genai as ov_genai
|
||||
|
||||
MODEL = "OpenVINO/whisper-large-v3-turbo-int8-ov"
|
||||
AUDIO = sys.argv[1] if len(sys.argv) > 1 else "sample.wav"
|
||||
SR = 16000
|
||||
|
||||
|
||||
def decode(path):
|
||||
"""ffmpeg -> 16 kHz mono float32, no librosa/numba needed."""
|
||||
raw = subprocess.run(
|
||||
["ffmpeg", "-nostdin", "-loglevel", "error", "-i", path,
|
||||
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
|
||||
capture_output=True, check=True).stdout
|
||||
return np.frombuffer(raw, dtype=np.float32)
|
||||
|
||||
|
||||
def make_pipe(model_dir, device, cache):
|
||||
"""CACHE_DIR passthrough form is unconfirmed for WhisperPipeline; try each."""
|
||||
try:
|
||||
return ov_genai.WhisperPipeline(model_dir, device, CACHE_DIR=cache), "kwarg"
|
||||
except Exception:
|
||||
try:
|
||||
return ov_genai.WhisperPipeline(model_dir, device, {"CACHE_DIR": cache}), "dict"
|
||||
except Exception:
|
||||
ov.Core().set_property(device, {"CACHE_DIR": cache})
|
||||
return ov_genai.WhisperPipeline(model_dir, device), "set_property"
|
||||
|
||||
|
||||
def run(model_dir, device, speech, audio_sec):
|
||||
cache = f".ov_cache_{device.lower()}"
|
||||
print(f"\n{'='*62}\n{device}\n{'='*62}", flush=True)
|
||||
|
||||
t = time.perf_counter(); pipe, form = make_pipe(model_dir, device, cache)
|
||||
cold = time.perf_counter() - t
|
||||
print(f" compile, cold cache : {cold:7.1f}s (CACHE_DIR form: {form})", flush=True)
|
||||
|
||||
t = time.perf_counter()
|
||||
res = pipe.generate(speech, task="transcribe", return_timestamps=True)
|
||||
gen = time.perf_counter() - t
|
||||
print(f" generate : {gen:7.1f}s RTF {audio_sec/gen:.2f}x realtime", flush=True)
|
||||
|
||||
del pipe
|
||||
t = time.perf_counter(); pipe, _ = make_pipe(model_dir, device, cache)
|
||||
warm = time.perf_counter() - t
|
||||
print(f" compile, warm cache : {warm:7.1f}s", flush=True)
|
||||
del pipe
|
||||
|
||||
return {"device": device, "cold": cold, "warm": warm, "gen": gen,
|
||||
"rtf": audio_sec / gen, "text": str(res)}
|
||||
|
||||
|
||||
def main():
|
||||
print("OpenVINO", ov.__version__)
|
||||
devices = ov.Core().available_devices
|
||||
print("available_devices:", devices)
|
||||
|
||||
model_dir = snapshot_download(MODEL)
|
||||
print("model:", model_dir)
|
||||
|
||||
speech = decode(AUDIO)
|
||||
audio_sec = len(speech) / SR
|
||||
print(f"audio: {AUDIO} {audio_sec:.1f}s ({len(speech):,} samples)")
|
||||
|
||||
targets = [d for d in ("GPU", "CPU") if d in devices]
|
||||
if "GPU" not in devices:
|
||||
print("\n!! GPU NOT PRESENT -- intel-compute-runtime not installed yet")
|
||||
|
||||
results = [run(model_dir, d, speech, audio_sec) for d in targets]
|
||||
|
||||
print(f"\n{'='*62}\nTRANSCRIPT ({results[0]['device']}), first 400 chars\n{'='*62}")
|
||||
print(results[0]["text"][:400])
|
||||
|
||||
print(f"\n{'='*62}\nSUMMARY ({audio_sec:.0f}s of audio)\n{'='*62}")
|
||||
print(f"{'device':<8}{'cold':>9}{'warm':>9}{'generate':>11}{'RTF':>9}")
|
||||
for r in results:
|
||||
print(f"{r['device']:<8}{r['cold']:>8.1f}s{r['warm']:>8.1f}s"
|
||||
f"{r['gen']:>10.1f}s{r['rtf']:>8.2f}x")
|
||||
if len(results) == 2:
|
||||
g, c = results[0]["gen"], results[1]["gen"]
|
||||
verdict = f"GPU {c/g:.2f}x faster" if g < c else f"GPU {g/c:.2f}x SLOWER"
|
||||
print(f"\n -> {verdict} than CPU on generate")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user