chore: OpenVINO MetricGAN GPU spike (do not merge)

Arc B580 lost the kill test: batch-14 mask p50 103 ms vs Torch CPU
70 ms enhance / OpenVINO CPU 5 ms. Production stays on known-good
CPU workers (tag known-good-cpu-2026-08-28). Hop stays 200 ms.
This commit is contained in:
root
2026-08-28 07:18:54 +00:00
parent 736eccb932
commit 42443420be
6 changed files with 307 additions and 0 deletions
+2
View File
@@ -1,4 +1,6 @@
.venv/
.venv-ov/
scripts/*.onnx
__pycache__/
*.pyc
.env
+55
View File
@@ -0,0 +1,55 @@
# Spike: OpenVINO MetricGAN on Arc B580
Date: 2026-08-28
Known-good tag: `known-good-cpu-2026-08-28` (736eccb)
Fleet was **stopped** (cameras + displays). LiveKit server stayed up.
## What ran
- Export `enhance_model` (EnhancementGenerator BLSTM) to ONNX.
Input for 1 s @ 16 kHz: `(batch, 63, 257)` log-magnitude.
STFT / ISTFT stay on Torch CPU (not in the ONNX).
- OpenVINO 2026.3.1 in `.venv-ov` (not the CUDA project venv).
- Devices: CPU i7-14700, GPU.0 UHD 770, GPU.1 Arc B580.
## Numbers (batch=14, 1 s windows)
| path | p50 batch | p95 batch | p99 batch | p50 / cam |
|---|---:|---:|---:|---:|
| Torch CPU `enhance_batch` (1 thread) | 69.7 ms | 69.9 ms | 69.9 ms | 4.98 ms |
| Torch CPU mask only | 63.6 ms | 63.7 ms | 63.7 ms | 4.54 ms |
| OpenVINO CPU mask | 5.13 ms | 5.49 ms | 5.65 ms | 0.37 ms |
| OpenVINO GPU.0 iGPU mask | 75.0 ms | 75.4 ms | 75.4 ms | 5.36 ms |
| OpenVINO GPU.1 Arc B580 mask | 103.0 ms | 103.1 ms | 103.4 ms | 7.35 ms |
## Kill criterion (plan)
OpenVINO GPU ms/camera must be ≥2× thread-capped CPU batch, and p99
including copies < hop/2.
- Arc vs Torch CPU enhance: **slower** (7.35 vs 4.98 ms/cam). Fail.
- Arc vs OpenVINO CPU mask: **20× slower**. Fail.
- Arc p99 103 ms is already half of a 200 ms hop for the **mask only**,
before STFT/ISTFT. Fail for hop cut.
**Do not put MetricGAN on the B580. Do not merge a GPU enhance daemon.**
## What would actually cut hop
OpenVINO **CPU** mask is ~12× Torch. A single batched CPU enhance
process (STFT + OV-CPU mask + ISTFT) is the only path that looks fast
enough for `ENHANCE_HOP_S=0.1`. That is a new architecture (workers
DelaySum locally, one enhance server), not a CUDA/`device=xpu` switch.
This spike does **not** implement that server. Production stays on the
known-good per-worker Torch path (hop 200 ms).
## Reproduce
```bash
systemctl stop livekit-cameras livekit-displays
.venv/bin/python scripts/spike_export_metricgan_onnx.py
.venv/bin/python scripts/spike_bench_cpu_metricgan.py --batch 14
.venv-ov/bin/python scripts/spike_bench_openvino.py --batch 14
systemctl start livekit-cameras livekit-displays
```
+76
View File
@@ -0,0 +1,76 @@
#!/usr/bin/env python3
"""CPU baseline: thread-capped MetricGAN enhance_batch (project .venv)."""
from __future__ import annotations
import argparse
import os
import statistics
import time
os.environ.setdefault("OMP_NUM_THREADS", "1")
os.environ.setdefault("MKL_NUM_THREADS", "1")
os.environ.setdefault("OPENBLAS_NUM_THREADS", "1")
import torch
from speechbrain.inference.enhancement import SpectralMaskEnhancement
SAVEDIR = "/root/.cache/speechbrain-enhancement"
def pct(xs, p):
xs = sorted(xs)
i = int(round((p / 100.0) * (len(xs) - 1)))
return xs[max(0, min(i, len(xs) - 1))]
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--batch", type=int, default=14)
ap.add_argument("--samples", type=int, default=16000)
ap.add_argument("--iters", type=int, default=30)
ap.add_argument("--warmup", type=int, default=5)
args = ap.parse_args()
torch.set_num_threads(1)
torch.set_num_interop_threads(1)
m = SpectralMaskEnhancement.from_hparams(
source="speechbrain/metricgan-plus-voicebank",
savedir=SAVEDIR,
run_opts={"device": "cpu"},
)
wav = torch.randn(args.batch, args.samples)
lengths = torch.ones(args.batch)
with torch.no_grad():
for _ in range(args.warmup):
m.enhance_batch(wav, lengths=lengths)
times = []
for _ in range(args.iters):
t0 = time.perf_counter()
m.enhance_batch(wav, lengths=lengths)
times.append((time.perf_counter() - t0) * 1000.0)
feats = m.compute_features(wav)
net = m.mods.enhance_model
for _ in range(args.warmup):
net(feats, lengths=lengths)
mask_times = []
for _ in range(args.iters):
t0 = time.perf_counter()
net(feats, lengths=lengths)
mask_times.append((time.perf_counter() - t0) * 1000.0)
def report(name, xs):
per = [x / args.batch for x in xs]
print(
f"{name} batch={args.batch} n={len(xs)} "
f"p50={statistics.median(xs):.2f}ms p95={pct(xs,95):.2f}ms "
f"p99={pct(xs,99):.2f}ms "
f"per_cam_p50={statistics.median(per):.2f}ms "
f"per_cam_p95={pct(per,95):.2f}ms"
)
report("cpu_enhance_batch", times)
report("cpu_mask_only", mask_times)
if __name__ == "__main__":
main()
+78
View File
@@ -0,0 +1,78 @@
#!/usr/bin/env python3
"""OpenVINO GPU/CPU bench of MetricGAN mask ONNX (.venv-ov)."""
from __future__ import annotations
import argparse
import statistics
import time
from pathlib import Path
import numpy as np
import openvino as ov
def pct(xs, p):
xs = sorted(xs)
i = int(round((p / 100.0) * (len(xs) - 1)))
return xs[max(0, min(i, len(xs) - 1))]
def bench(compiled, feats, warmup, iters):
infer = compiled.create_infer_request()
inp = compiled.inputs[0]
for _ in range(warmup):
infer.infer({inp: feats})
xs = []
for _ in range(iters):
t0 = time.perf_counter()
infer.infer({inp: feats})
xs.append((time.perf_counter() - t0) * 1000.0)
return xs
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--onnx", default="scripts/metricgan_enhance.onnx")
ap.add_argument("--batch", type=int, default=14)
ap.add_argument("--time", type=int, default=63)
ap.add_argument("--feat", type=int, default=257)
ap.add_argument("--iters", type=int, default=40)
ap.add_argument("--warmup", type=int, default=10)
args = ap.parse_args()
core = ov.Core()
print("openvino", ov.__version__)
print("devices", core.available_devices)
for d in core.available_devices:
try:
print(" ", d, core.get_property(d, "FULL_DEVICE_NAME"))
except Exception as e:
print(" ", d, e)
onnx = Path(args.onnx)
model = ov.convert_model(str(onnx))
feats = np.random.randn(args.batch, args.time, args.feat).astype(np.float32)
for device in core.available_devices:
try:
compiled = core.compile_model(model, device)
except Exception as e:
print(f"compile {device} FAILED: {type(e).__name__}: {e}")
continue
try:
xs = bench(compiled, feats, args.warmup, args.iters)
except Exception as e:
print(f"infer {device} FAILED: {type(e).__name__}: {e}")
continue
per = [x / args.batch for x in xs]
print(
f"ov_{device} batch={args.batch} "
f"p50={statistics.median(xs):.2f}ms p95={pct(xs,95):.2f}ms "
f"p99={pct(xs,99):.2f}ms "
f"per_cam_p50={statistics.median(per):.2f}ms "
f"per_cam_p95={pct(per,95):.2f}ms"
)
if __name__ == "__main__":
main()
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env python3
"""Dump MetricGAN enhance_model I/O so the OpenVINO spike can convert it."""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
import torch
from speechbrain.inference.enhancement import SpectralMaskEnhancement
SAVEDIR = "/root/.cache/speechbrain-enhancement"
def main() -> None:
m = SpectralMaskEnhancement.from_hparams(
source="speechbrain/metricgan-plus-voicebank",
savedir=SAVEDIR,
run_opts={"device": "cpu"},
)
net = m.mods.enhance_model
print("enhance_model", type(net).__name__)
print("device", m.device)
wav = torch.randn(1, 16000)
feats = m.compute_features(wav)
print("wav", tuple(wav.shape), "feats", tuple(feats.shape), feats.dtype)
lengths = torch.tensor([1.0])
with torch.no_grad():
mask = net(feats, lengths=lengths)
print("mask", tuple(mask.shape), mask.dtype)
print("modules:")
for name, mod in net.named_children():
print(f" {name}: {type(mod).__name__}")
if __name__ == "__main__":
main()
+58
View File
@@ -0,0 +1,58 @@
#!/usr/bin/env python3
"""Export MetricGAN enhance_model (BLSTM masker) to ONNX. Uses project .venv."""
from __future__ import annotations
import argparse
from pathlib import Path
import torch
import torch.nn as nn
from speechbrain.inference.enhancement import SpectralMaskEnhancement
SAVEDIR = "/root/.cache/speechbrain-enhancement"
class MaskNet(nn.Module):
def __init__(self, net: nn.Module):
super().__init__()
self.net = net
def forward(self, feats: torch.Tensor) -> torch.Tensor:
# lengths=1.0 relative, one per batch row
lengths = torch.ones(feats.shape[0], dtype=feats.dtype)
return self.net(feats, lengths=lengths)
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--out", default="scripts/metricgan_enhance.onnx")
args = ap.parse_args()
out = Path(args.out)
out.parent.mkdir(parents=True, exist_ok=True)
m = SpectralMaskEnhancement.from_hparams(
source="speechbrain/metricgan-plus-voicebank",
savedir=SAVEDIR,
run_opts={"device": "cpu"},
)
wrap = MaskNet(m.mods.enhance_model).eval()
feats = m.compute_features(torch.randn(1, 16000))
with torch.no_grad():
torch.onnx.export(
wrap,
feats,
str(out),
input_names=["feats"],
output_names=["mask"],
dynamic_axes={
"feats": {0: "batch", 1: "time"},
"mask": {0: "batch", 1: "time"},
},
opset_version=17,
dynamo=False,
)
print("wrote", out, "bytes", out.stat().st_size, "example_feats", tuple(feats.shape))
if __name__ == "__main__":
main()