Publish and subscribe over wss://livekit.uni-wh.de:7800 and refuse cleartext ws://. Conference room is uwh-telhai. Includes the uncommitted encoded H.264 publish path, Rally hairpin, and KMS wall overlay.
404 lines
13 KiB
Python
404 lines
13 KiB
Python
"""Person-aware background blur for C920 I420 SHM.
|
|
|
|
OpenVINO CPU only. iGPU stays H.264; Arc stays kmssink.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
import numpy as np
|
|
|
|
MODELS_DIR = Path(__file__).resolve().parent / "models"
|
|
SELFIE_ONNX = MODELS_DIR / "selfie_segmentation.onnx"
|
|
MOVENET_LIGHTNING = MODELS_DIR / "movenet_singlepose_lightning.onnx"
|
|
POSE_SIZE = 192
|
|
YUNET_CANDIDATES = (
|
|
"face_detection_yunet_2026may.onnx",
|
|
"face_detection_yunet_2023mar.onnx",
|
|
)
|
|
RALLY_ID = "rally"
|
|
|
|
|
|
def i420_source_path(identity: str, raw_dir: str, portrait_dir: str = "") -> str:
|
|
"""C920s read portrait SHM when enabled; rally always stays on capture SHM."""
|
|
ident = (identity or "").strip()
|
|
raw = (raw_dir or "").rstrip("/")
|
|
por = (portrait_dir or "").rstrip("/")
|
|
if por and ident and ident != RALLY_ID:
|
|
return f"{por}/{ident}.i420"
|
|
return f"{raw}/{ident}.i420"
|
|
|
|
|
|
def i420_nbytes(width: int, height: int) -> int:
|
|
return int(width) * int(height) * 3 // 2
|
|
|
|
|
|
def i420_to_bgr(buf, width: int, height: int) -> np.ndarray:
|
|
import cv2
|
|
|
|
w, h = int(width), int(height)
|
|
need = i420_nbytes(w, h)
|
|
raw = np.frombuffer(buf, dtype=np.uint8, count=need)
|
|
yuv = raw.reshape((h * 3 // 2, w))
|
|
return cv2.cvtColor(yuv, cv2.COLOR_YUV2BGR_I420)
|
|
|
|
|
|
def bgr_to_i420(bgr: np.ndarray) -> bytes:
|
|
import cv2
|
|
|
|
yuv = cv2.cvtColor(bgr, cv2.COLOR_BGR2YUV_I420)
|
|
return np.ascontiguousarray(yuv).tobytes()
|
|
|
|
|
|
def soft_blur_bgr(img: np.ndarray, radius: int = 7) -> np.ndarray:
|
|
"""Cheap bokeh: half-res box blur, then upsample."""
|
|
import cv2
|
|
|
|
k = max(3, int(radius) | 1)
|
|
h, w = img.shape[:2]
|
|
sw, sh = max(2, w // 2), max(2, h // 2)
|
|
small = cv2.resize(img, (sw, sh), interpolation=cv2.INTER_AREA)
|
|
small = cv2.blur(small, (k, k))
|
|
return cv2.resize(small, (w, h), interpolation=cv2.INTER_LINEAR)
|
|
|
|
|
|
def feather_mask(mask: np.ndarray, ksize: int = 7) -> np.ndarray:
|
|
import cv2
|
|
|
|
k = max(3, int(ksize) | 1)
|
|
m = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
|
|
return cv2.GaussianBlur(m, (k, k), 0)
|
|
|
|
|
|
def composite_bgr(sharp: np.ndarray, blurred: np.ndarray, mask: np.ndarray) -> np.ndarray:
|
|
a = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
|
|
if a.ndim == 2:
|
|
a = a[:, :, None]
|
|
out = sharp.astype(np.float32) * a + blurred.astype(np.float32) * (1.0 - a)
|
|
return np.clip(out, 0, 255).astype(np.uint8)
|
|
|
|
|
|
def person_present(
|
|
mask: np.ndarray,
|
|
min_peak: float = 0.35,
|
|
min_area: float = 0.04,
|
|
) -> bool:
|
|
m = np.asarray(mask, dtype=np.float32)
|
|
if m.size == 0:
|
|
return False
|
|
if float(m.max()) < float(min_peak):
|
|
return False
|
|
return float((m >= min_peak).mean()) >= float(min_area)
|
|
|
|
|
|
def union_masks(*masks: Optional[np.ndarray]) -> np.ndarray:
|
|
acc: Optional[np.ndarray] = None
|
|
for m in masks:
|
|
if m is None:
|
|
continue
|
|
a = np.clip(np.asarray(m, dtype=np.float32), 0.0, 1.0)
|
|
acc = a if acc is None else np.maximum(acc, a)
|
|
if acc is None:
|
|
return np.zeros((1, 1), dtype=np.float32)
|
|
return acc
|
|
|
|
|
|
# MoveNet COCO-17: nose, eyes, ears, shoulders, elbows, wrists.
|
|
_POSE_R = {
|
|
0: 0.14, # nose / face
|
|
1: 0.08, 2: 0.08, 3: 0.08, 4: 0.08, # eyes, ears
|
|
5: 0.10, 6: 0.10, # shoulders
|
|
7: 0.09, 8: 0.09, # elbows
|
|
9: 0.11, 10: 0.11, # wrists / hands
|
|
}
|
|
_POSE_MIN_SCORE = 0.25
|
|
|
|
|
|
def pose_protect_mask(
|
|
height: int,
|
|
width: int,
|
|
kpts: np.ndarray,
|
|
min_score: float = _POSE_MIN_SCORE,
|
|
) -> np.ndarray:
|
|
"""Soft disks on face + arms so raised hands stay sharp."""
|
|
h, w = int(height), int(width)
|
|
out = np.zeros((h, w), dtype=np.float32)
|
|
if kpts is None or kpts.shape != (17, 3) or h < 2 or w < 2:
|
|
return out
|
|
scale = float(min(h, w))
|
|
yy, xx = np.ogrid[:h, :w]
|
|
for idx, frac in _POSE_R.items():
|
|
y, x, s = float(kpts[idx, 0]), float(kpts[idx, 1]), float(kpts[idx, 2])
|
|
if s < min_score:
|
|
continue
|
|
cy, cx = y * h, x * w
|
|
r = max(6.0, frac * scale)
|
|
dist = ((xx - cx) / r) ** 2 + ((yy - cy) / r) ** 2
|
|
out = np.maximum(out, np.clip(1.0 - dist, 0.0, 1.0).astype(np.float32))
|
|
return out
|
|
|
|
|
|
def apply_portrait(
|
|
bgr: np.ndarray,
|
|
mask: np.ndarray,
|
|
radius: int = 7,
|
|
feather: int = 7,
|
|
) -> np.ndarray:
|
|
"""Passthrough never: empty seats are full-frame blur."""
|
|
if bgr is None:
|
|
return bgr
|
|
if mask is None or not person_present(mask):
|
|
z = np.zeros(bgr.shape[:2], dtype=np.float32)
|
|
return composite_bgr(bgr, soft_blur_bgr(bgr, radius), z)
|
|
soft = feather_mask(mask, feather)
|
|
return composite_bgr(bgr, soft_blur_bgr(bgr, radius), soft)
|
|
|
|
|
|
class MaskHold:
|
|
"""EMA + hold so a missed infer does not shimmer or drop the person."""
|
|
|
|
def __init__(self, hold_s: float = 0.8, ema: float = 0.4) -> None:
|
|
self.hold_s = max(0.0, float(hold_s))
|
|
self.ema = min(1.0, max(0.05, float(ema)))
|
|
self.mask: Optional[np.ndarray] = None
|
|
self.last_true = 0.0
|
|
|
|
def update(
|
|
self,
|
|
mask: Optional[np.ndarray],
|
|
present: bool,
|
|
now: float,
|
|
) -> Optional[np.ndarray]:
|
|
if present and mask is not None:
|
|
m = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
|
|
if self.mask is None or self.ema >= 1.0:
|
|
self.mask = m
|
|
else:
|
|
a = self.ema
|
|
self.mask = a * m + (1.0 - a) * self.mask
|
|
self.last_true = float(now)
|
|
return self.mask
|
|
if self.mask is not None and (float(now) - self.last_true) < self.hold_s:
|
|
return self.mask
|
|
self.mask = None
|
|
return None
|
|
|
|
|
|
def apply_i420(
|
|
payload,
|
|
width: int,
|
|
height: int,
|
|
mask: Optional[np.ndarray],
|
|
radius: int = 7,
|
|
feather: int = 7,
|
|
) -> bytes:
|
|
"""I420 in/out. Same WxH so encode MCU crop is unchanged."""
|
|
n = i420_nbytes(width, height)
|
|
raw = payload[:n]
|
|
if mask is None or not person_present(mask):
|
|
z = np.zeros((int(height), int(width)), dtype=np.float32)
|
|
return _composite_i420(raw, width, height, z, radius, feather=0)
|
|
return _composite_i420(raw, width, height, mask, radius, feather)
|
|
|
|
|
|
def _composite_i420(
|
|
payload,
|
|
width: int,
|
|
height: int,
|
|
mask: np.ndarray,
|
|
radius: int,
|
|
feather: int,
|
|
) -> bytes:
|
|
"""Blur background luma in I420. UV stays; avoids BGR round-trip."""
|
|
import cv2
|
|
|
|
w, h = int(width), int(height)
|
|
n = i420_nbytes(w, h)
|
|
arr = np.frombuffer(memoryview(payload)[:n], dtype=np.uint8).copy()
|
|
y = arr[: w * h].reshape(h, w)
|
|
a = np.asarray(mask, dtype=np.float32)
|
|
if a.shape != (h, w):
|
|
a = cv2.resize(a, (w, h), interpolation=cv2.INTER_LINEAR)
|
|
if feather and feather > 1:
|
|
a = feather_mask(a, feather)
|
|
k = max(3, int(radius) | 1)
|
|
sw, sh = max(2, w // 2), max(2, h // 2)
|
|
y_s = cv2.resize(y, (sw, sh), interpolation=cv2.INTER_AREA)
|
|
y_s = cv2.blur(y_s, (k, k))
|
|
yb = cv2.resize(y_s, (w, h), interpolation=cv2.INTER_LINEAR)
|
|
af = a
|
|
y[:] = (y.astype(np.float32) * af + yb.astype(np.float32) * (1.0 - af)).clip(0, 255).astype(np.uint8)
|
|
return arr.tobytes()
|
|
|
|
|
|
def letterbox_bgr(bgr: np.ndarray, size: int = 256) -> tuple[np.ndarray, tuple[int, int, int, int]]:
|
|
"""Pad 16:9 into a square. Returns canvas and (x0, y0, nw, nh) content box."""
|
|
import cv2
|
|
|
|
size = int(size)
|
|
h, w = bgr.shape[:2]
|
|
if h < 1 or w < 1:
|
|
return np.zeros((size, size, 3), dtype=np.uint8), (0, 0, size, size)
|
|
scale = size / float(max(h, w))
|
|
nw = max(1, int(round(w * scale)))
|
|
nh = max(1, int(round(h * scale)))
|
|
resized = cv2.resize(bgr, (nw, nh), interpolation=cv2.INTER_AREA)
|
|
canvas = np.zeros((size, size, 3), dtype=np.uint8)
|
|
y0 = (size - nh) // 2
|
|
x0 = (size - nw) // 2
|
|
canvas[y0:y0 + nh, x0:x0 + nw] = resized
|
|
return canvas, (x0, y0, nw, nh)
|
|
|
|
|
|
def unletterbox_mask(
|
|
mask: np.ndarray,
|
|
box: tuple[int, int, int, int],
|
|
height: int,
|
|
width: int,
|
|
) -> np.ndarray:
|
|
import cv2
|
|
|
|
x0, y0, nw, nh = (int(v) for v in box)
|
|
crop = np.asarray(mask, dtype=np.float32)[y0:y0 + nh, x0:x0 + nw]
|
|
if crop.size == 0:
|
|
return np.zeros((int(height), int(width)), dtype=np.float32)
|
|
return cv2.resize(crop, (int(width), int(height)), interpolation=cv2.INTER_LINEAR)
|
|
|
|
|
|
def resolve_yunet(models_dir: Path | None = None) -> Path:
|
|
root = Path(models_dir) if models_dir is not None else MODELS_DIR
|
|
for name in YUNET_CANDIDATES:
|
|
path = root / name
|
|
if path.is_file():
|
|
return path
|
|
return root / YUNET_CANDIDATES[0]
|
|
|
|
|
|
def _face_soft_mask(bgr: np.ndarray, faces) -> Optional[np.ndarray]:
|
|
"""Soft ellipse per YuNet face so close-up heads stay unblurred."""
|
|
if faces is None or len(faces) == 0:
|
|
return None
|
|
h, w = bgr.shape[:2]
|
|
acc = None
|
|
yy, xx = np.ogrid[:h, :w]
|
|
for best in faces:
|
|
x, y, fw, fh = (float(best[0]), float(best[1]), float(best[2]), float(best[3]))
|
|
if fw < 8 or fh < 8:
|
|
continue
|
|
cx = x + fw * 0.5
|
|
cy = y + fh * 0.42
|
|
rx = max(fw * 1.15, w * 0.12)
|
|
ry = max(fh * 1.35, h * 0.16)
|
|
dist = ((xx - cx) / rx) ** 2 + ((yy - cy) / ry) ** 2
|
|
m = np.clip(1.0 - dist, 0.0, 1.0).astype(np.float32)
|
|
acc = m if acc is None else np.maximum(acc, m)
|
|
return acc
|
|
|
|
|
|
class PersonSeg:
|
|
"""MediaPipe selfie ONNX on OpenVINO CPU. Never GPU.0/GPU.1."""
|
|
|
|
def __init__(
|
|
self,
|
|
model_path: Path | None = None,
|
|
yunet_path: Path | None = None,
|
|
num_threads: int = 2,
|
|
) -> None:
|
|
self.device = "CPU"
|
|
self._request = None
|
|
self._ov_input = None
|
|
self._size = 256
|
|
self._yunet = None
|
|
self._cv2 = None
|
|
self._pose_request = None
|
|
self._pose_input = None
|
|
self._pose_size = POSE_SIZE
|
|
path = Path(model_path) if model_path is not None else SELFIE_ONNX
|
|
import cv2
|
|
import openvino as ov
|
|
|
|
self._cv2 = cv2
|
|
try:
|
|
cv2.setNumThreads(1)
|
|
except Exception:
|
|
pass
|
|
if not path.is_file():
|
|
raise FileNotFoundError(path)
|
|
core = ov.Core()
|
|
model = core.read_model(str(path))
|
|
compiled = core.compile_model(
|
|
model,
|
|
"CPU",
|
|
{"INFERENCE_NUM_THREADS": max(1, int(num_threads)), "NUM_STREAMS": 1},
|
|
)
|
|
self._request = compiled.create_infer_request()
|
|
self._ov_input = compiled.input(0)
|
|
ypath = Path(yunet_path) if yunet_path is not None else resolve_yunet()
|
|
if ypath.is_file():
|
|
self._yunet = cv2.FaceDetectorYN_create(
|
|
str(ypath), "", (320, 180), 0.45, 0.3, 5000)
|
|
if MOVENET_LIGHTNING.is_file():
|
|
pose = core.read_model(str(MOVENET_LIGHTNING))
|
|
pose_c = core.compile_model(
|
|
pose,
|
|
"CPU",
|
|
{"INFERENCE_NUM_THREADS": 1, "NUM_STREAMS": 1},
|
|
)
|
|
self._pose_request = pose_c.create_infer_request()
|
|
self._pose_input = pose_c.input(0)
|
|
|
|
def infer_bgr(self, bgr: np.ndarray) -> np.ndarray:
|
|
h, w = bgr.shape[:2]
|
|
canvas, box = letterbox_bgr(bgr, self._size)
|
|
rgb = canvas[:, :, ::-1].astype(np.float32) * (1.0 / 255.0)
|
|
inp = np.transpose(rgb, (2, 0, 1))[None]
|
|
self._request.infer({self._ov_input: inp})
|
|
raw = np.asarray(self._request.get_output_tensor(0).data, dtype=np.float32)
|
|
raw = np.squeeze(raw)
|
|
mask = unletterbox_mask(raw, box, h, w)
|
|
face = self._yunet_mask(bgr)
|
|
if face is not None:
|
|
mask = union_masks(mask, face)
|
|
if person_present(mask) or face is not None:
|
|
pose = self._pose_mask(bgr)
|
|
if pose is not None and float(pose.max()) > 0:
|
|
mask = union_masks(mask, pose)
|
|
return mask
|
|
|
|
def _pose_mask(self, bgr: np.ndarray) -> Optional[np.ndarray]:
|
|
if self._pose_request is None or self._pose_input is None:
|
|
return None
|
|
h, w = bgr.shape[:2]
|
|
canvas, box = letterbox_bgr(bgr, self._pose_size)
|
|
rgb = np.ascontiguousarray(canvas[:, :, ::-1], dtype=np.int32)
|
|
inp = rgb[None]
|
|
self._pose_request.infer({self._pose_input: inp})
|
|
kpts = np.asarray(
|
|
self._pose_request.get_output_tensor(0).data, dtype=np.float32
|
|
).reshape(17, 3)
|
|
x0, y0, nw, nh = box
|
|
size = float(self._pose_size)
|
|
mapped = kpts.copy()
|
|
mapped[:, 0] = (kpts[:, 0] * size - y0) / max(1.0, float(nh))
|
|
mapped[:, 1] = (kpts[:, 1] * size - x0) / max(1.0, float(nw))
|
|
return pose_protect_mask(h, w, mapped)
|
|
|
|
def _yunet_mask(self, bgr: np.ndarray) -> Optional[np.ndarray]:
|
|
if self._yunet is None or self._cv2 is None:
|
|
return None
|
|
h, w = bgr.shape[:2]
|
|
dw, dh = 320, 180
|
|
small = self._cv2.resize(bgr, (dw, dh), interpolation=self._cv2.INTER_AREA)
|
|
self._yunet.setInputSize((dw, dh))
|
|
_ok, faces = self._yunet.detect(small)
|
|
if faces is None or len(faces) == 0:
|
|
return None
|
|
sx = w / float(dw)
|
|
sy = h / float(dh)
|
|
scaled = []
|
|
for f in faces:
|
|
scaled.append([f[0] * sx, f[1] * sy, f[2] * sx, f[3] * sy])
|
|
return _face_soft_mask(bgr, scaled)
|