Publish and subscribe over wss://livekit.uni-wh.de:7800 and refuse cleartext ws://. Conference room is uwh-telhai. Includes the uncommitted encoded H.264 publish path, Rally hairpin, and KMS wall overlay.
101 lines
3.3 KiB
Python
101 lines
3.3 KiB
Python
"""OpenVINO CPU compile of the MetricGAN mask net.
|
|
|
|
STFT / ISTFT stay in SpeechBrain; only EnhancementGenerator runs here.
|
|
Device is CPU — the Arc B580 lost the GPU kill test.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import fcntl
|
|
import logging
|
|
import os
|
|
from pathlib import Path
|
|
|
|
import numpy as np
|
|
|
|
log = logging.getLogger("audio.cleanup")
|
|
|
|
CACHE_DIR = Path("/root/.cache/speechbrain-enhancement")
|
|
ONNX_NAME = "metricgan_enhance.onnx"
|
|
LOCK_NAME = "metricgan_enhance.lock"
|
|
SHIPPED_ONNX = Path(__file__).resolve().parent / "scripts" / "metricgan_enhance.onnx"
|
|
|
|
|
|
def ensure_onnx(enhancer, torch) -> Path:
|
|
"""Export or reuse the BLSTM mask ONNX. One process writes; others wait."""
|
|
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
|
onnx_path = CACHE_DIR / ONNX_NAME
|
|
lock_path = CACHE_DIR / LOCK_NAME
|
|
with open(lock_path, "w") as lockf:
|
|
fcntl.flock(lockf.fileno(), fcntl.LOCK_EX)
|
|
if onnx_path.is_file() and onnx_path.stat().st_size > 1000:
|
|
return onnx_path
|
|
if SHIPPED_ONNX.is_file() and SHIPPED_ONNX.stat().st_size > 1000:
|
|
onnx_path.write_bytes(SHIPPED_ONNX.read_bytes())
|
|
log.info("copied shipped MetricGAN ONNX to %s", onnx_path)
|
|
return onnx_path
|
|
log.info("exporting MetricGAN enhance_model to ONNX %s", onnx_path)
|
|
import torch.nn as nn
|
|
|
|
class MaskNet(nn.Module):
|
|
def __init__(self, net):
|
|
super().__init__()
|
|
self.net = net
|
|
|
|
def forward(self, feats):
|
|
lengths = torch.ones(feats.shape[0], dtype=feats.dtype)
|
|
return self.net(feats, lengths=lengths)
|
|
|
|
wrap = MaskNet(enhancer.mods.enhance_model).eval()
|
|
feats = enhancer.compute_features(torch.randn(1, 16000))
|
|
tmp = onnx_path.with_suffix(".onnx.tmp")
|
|
with torch.no_grad():
|
|
torch.onnx.export(
|
|
wrap,
|
|
feats,
|
|
str(tmp),
|
|
input_names=["feats"],
|
|
output_names=["mask"],
|
|
dynamic_axes={
|
|
"feats": {0: "batch", 1: "time"},
|
|
"mask": {0: "batch", 1: "time"},
|
|
},
|
|
opset_version=17,
|
|
dynamo=False,
|
|
)
|
|
tmp.replace(onnx_path)
|
|
log.info("wrote %s (%d bytes)", onnx_path, onnx_path.stat().st_size)
|
|
return onnx_path
|
|
|
|
|
|
_COMPILED = None
|
|
|
|
|
|
def compile_cpu(onnx_path: Path, num_threads: int = 1):
|
|
"""Compile ONNX for OpenVINO CPU with a 1-thread infer request."""
|
|
global _COMPILED
|
|
if _COMPILED is not None:
|
|
compiled, _, ov_input = _COMPILED
|
|
request = compiled.create_infer_request()
|
|
return compiled, request, ov_input
|
|
import openvino as ov
|
|
|
|
core = ov.Core()
|
|
model = ov.convert_model(str(onnx_path))
|
|
compiled = core.compile_model(
|
|
model,
|
|
"CPU",
|
|
{
|
|
"INFERENCE_NUM_THREADS": num_threads,
|
|
"NUM_STREAMS": max(1, int(os.environ.get("OV_NUM_STREAMS", "8"))),
|
|
},
|
|
)
|
|
request = compiled.create_infer_request()
|
|
log.info("OpenVINO CPU MetricGAN mask ready (%s)", ov.__version__)
|
|
_COMPILED = (compiled, None, compiled.inputs[0])
|
|
return compiled, request, compiled.inputs[0]
|
|
|
|
|
|
def infer_mask(request, ov_input, feats: np.ndarray) -> np.ndarray:
|
|
result = request.infer({ov_input: feats})
|
|
return np.asarray(next(iter(result.values())))
|