style(scripts): apply ruff formatting to the benchmark scripts

Formatting only: split combined imports and semicolon statements, sort
imports, wrap subprocess argument lists. No behaviour change; the
one-device-per-process structure that makes bench.py's numbers
trustworthy is untouched.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-13 14:07:58 -05:00
co-authored by Claude Opus 5
parent b2d648e532
commit c86749973c
2 changed files with 80 additions and 25 deletions
+35 -9
View File
@@ -1,8 +1,13 @@
"""One device per process: wall time + CPU time consumed (offload evidence)."""
import subprocess, sys, time, resource
import resource
import subprocess
import sys
import time
import numpy as np
from huggingface_hub import snapshot_download
import openvino_genai as ov_genai
from huggingface_hub import snapshot_download
SR = 16000
@@ -10,15 +15,33 @@ SR = 16000
def main():
dev = sys.argv[1]
audio = sys.argv[2] if len(sys.argv) > 2 else "sample.wav"
raw = subprocess.run(["ffmpeg", "-nostdin", "-loglevel", "error", "-i", audio,
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
capture_output=True, check=True).stdout
raw = subprocess.run(
[
"ffmpeg",
"-nostdin",
"-loglevel",
"error",
"-i",
audio,
"-f",
"f32le",
"-ac",
"1",
"-ar",
str(SR),
"-",
],
capture_output=True,
check=True,
).stdout
speech = np.frombuffer(raw, dtype=np.float32)
secs = len(speech) / SR
pipe = ov_genai.WhisperPipeline(
snapshot_download("OpenVINO/whisper-large-v3-turbo-int8-ov"),
dev, CACHE_DIR=f".ov_cache_{dev.lower()}")
dev,
CACHE_DIR=f".ov_cache_{dev.lower()}",
)
for i in range(3):
r0 = resource.getrusage(resource.RUSAGE_SELF)
@@ -27,9 +50,12 @@ def main():
wall = time.perf_counter() - t
r1 = resource.getrusage(resource.RUSAGE_SELF)
cpu = (r1.ru_utime - r0.ru_utime) + (r1.ru_stime - r0.ru_stime)
print(f"{dev} pass{i+1}: wall {wall:6.1f}s RTF {secs/wall:.2f}x "
f"cpu-time {cpu:6.1f}s ({cpu/wall:4.1f} cores busy)", flush=True)
print(
f"{dev} pass{i + 1}: wall {wall:6.1f}s RTF {secs / wall:.2f}x "
f"cpu-time {cpu:6.1f}s ({cpu / wall:4.1f} cores busy)",
flush=True,
)
if __name__ == "__main__": # forkserver re-imports __main__; guard is required
if __name__ == "__main__": # forkserver re-imports __main__; guard is required
main()
+45 -16
View File
@@ -1,9 +1,13 @@
"""Throwaway smoke test: prove Whisper runs on the Intel iGPU via OpenVINO GenAI."""
import subprocess, sys, time
import subprocess
import sys
import time
import numpy as np
from huggingface_hub import snapshot_download
import openvino as ov
import openvino_genai as ov_genai
from huggingface_hub import snapshot_download
MODEL = "OpenVINO/whisper-large-v3-turbo-int8-ov"
AUDIO = sys.argv[1] if len(sys.argv) > 1 else "sample.wav"
@@ -13,9 +17,24 @@ SR = 16000
def decode(path):
"""ffmpeg -> 16 kHz mono float32, no librosa/numba needed."""
raw = subprocess.run(
["ffmpeg", "-nostdin", "-loglevel", "error", "-i", path,
"-f", "f32le", "-ac", "1", "-ar", str(SR), "-"],
capture_output=True, check=True).stdout
[
"ffmpeg",
"-nostdin",
"-loglevel",
"error",
"-i",
path,
"-f",
"f32le",
"-ac",
"1",
"-ar",
str(SR),
"-",
],
capture_output=True,
check=True,
).stdout
return np.frombuffer(raw, dtype=np.float32)
@@ -33,25 +52,33 @@ def make_pipe(model_dir, device, cache):
def run(model_dir, device, speech, audio_sec):
cache = f".ov_cache_{device.lower()}"
print(f"\n{'='*62}\n{device}\n{'='*62}", flush=True)
print(f"\n{'=' * 62}\n{device}\n{'=' * 62}", flush=True)
t = time.perf_counter(); pipe, form = make_pipe(model_dir, device, cache)
t = time.perf_counter()
pipe, form = make_pipe(model_dir, device, cache)
cold = time.perf_counter() - t
print(f" compile, cold cache : {cold:7.1f}s (CACHE_DIR form: {form})", flush=True)
t = time.perf_counter()
res = pipe.generate(speech, task="transcribe", return_timestamps=True)
gen = time.perf_counter() - t
print(f" generate : {gen:7.1f}s RTF {audio_sec/gen:.2f}x realtime", flush=True)
print(f" generate : {gen:7.1f}s RTF {audio_sec / gen:.2f}x realtime", flush=True)
del pipe
t = time.perf_counter(); pipe, _ = make_pipe(model_dir, device, cache)
t = time.perf_counter()
pipe, _ = make_pipe(model_dir, device, cache)
warm = time.perf_counter() - t
print(f" compile, warm cache : {warm:7.1f}s", flush=True)
del pipe
return {"device": device, "cold": cold, "warm": warm, "gen": gen,
"rtf": audio_sec / gen, "text": str(res)}
return {
"device": device,
"cold": cold,
"warm": warm,
"gen": gen,
"rtf": audio_sec / gen,
"text": str(res),
}
def main():
@@ -72,17 +99,19 @@ def main():
results = [run(model_dir, d, speech, audio_sec) for d in targets]
print(f"\n{'='*62}\nTRANSCRIPT ({results[0]['device']}), first 400 chars\n{'='*62}")
print(f"\n{'=' * 62}\nTRANSCRIPT ({results[0]['device']}), first 400 chars\n{'=' * 62}")
print(results[0]["text"][:400])
print(f"\n{'='*62}\nSUMMARY ({audio_sec:.0f}s of audio)\n{'='*62}")
print(f"\n{'=' * 62}\nSUMMARY ({audio_sec:.0f}s of audio)\n{'=' * 62}")
print(f"{'device':<8}{'cold':>9}{'warm':>9}{'generate':>11}{'RTF':>9}")
for r in results:
print(f"{r['device']:<8}{r['cold']:>8.1f}s{r['warm']:>8.1f}s"
f"{r['gen']:>10.1f}s{r['rtf']:>8.2f}x")
print(
f"{r['device']:<8}{r['cold']:>8.1f}s{r['warm']:>8.1f}s"
f"{r['gen']:>10.1f}s{r['rtf']:>8.2f}x"
)
if len(results) == 2:
g, c = results[0]["gen"], results[1]["gen"]
verdict = f"GPU {c/g:.2f}x faster" if g < c else f"GPU {g/c:.2f}x SLOWER"
verdict = f"GPU {c / g:.2f}x faster" if g < c else f"GPU {g / c:.2f}x SLOWER"
print(f"\n -> {verdict} than CPU on generate")