feat: point the fleet at wss LiveKit room uwh-telhai

Publish and subscribe over wss://livekit.uni-wh.de:7800 and refuse
cleartext ws://. Conference room is uwh-telhai. Includes the uncommitted
encoded H.264 publish path, Rally hairpin, and KMS wall overlay.
This commit is contained in:
root
2026-10-11 00:48:50 +00:00
parent ef9573c294
commit 95bb7c50ee
89 changed files with 9883 additions and 243 deletions
+64 -8
View File
@@ -1,13 +1,19 @@
# LiveKit server connection
# Local TEST server (autostarted). Switch this URL to the remote
# LiveKit later; then: systemctl disable --now livekit-server
LIVEKIT_URL=ws://127.0.0.1:7880
# Remote LiveKit. Local livekit-server.service is disabled.
# Signaling must be wss://. The client verifies the TLS cert (do not use ws://).
LIVEKIT_URL=wss://livekit.uni-wh.de:7800
LIVEKIT_API_KEY=APIKey_xxx
LIVEKIT_API_SECRET=APIsecret_xxx
# Room / naming
LIVEKIT_ROOM=cameras
# Conference room. Publishers and the KMS wall must use this same name.
LIVEKIT_ROOM=uwh-telhai
PARTICIPANT_PREFIX=cam
# Published on local camera participants as LiveKit attributes tag/tags.
PARTICIPANT_TAGS=uwh
# Display wall: empty = show local cameras (filter OFF). Set to uwh to hide
# local tagged participants so remotes fill the wall instead.
DISPLAY_HIDE_TAGS=
# Streams
# "all" = every discovered camera, "1,3,5" or "0-19" = subset
@@ -16,7 +22,9 @@ VIDEO_WIDTH=640
VIDEO_HEIGHT=360
VIDEO_FPS=30
VIDEO_MIN_FPS=5
VIDEO_BITRATE=800000
VIDEO_BITRATE=1200000
# Rally 1080p encode (bps). Rollback if encode/gst dies: 12000000 then restart cameras.
VIDEO_RALLY_BITRATE=20000000
VIDEO_CODEC=h264
VIDEO_ENCODER=vaapi
LIBVA_DRM_DEVICE=/dev/dri/renderD129
@@ -43,8 +51,33 @@ ENHANCE_INFER=auto
BEAMFORM_MODE=auto
# One BLAS/Torch thread per worker (20 workers vs 28 cores)
TORCH_NUM_THREADS=1
# Recompute GCC-PHAT TDOA every N hops (5 * 200 ms = 1 s)
# Recompute GCC-PHAT TDOA every N hops (5 * hop)
BEAMFORM_TDOA_EVERY=5
# One MetricGAN process for all cameras (workers send hops over a unix socket).
# ENHANCE_DAEMON=0 reverts to per-worker model load (~10 GiB extra RAM).
ENHANCE_DAEMON=1
ENHANCE_SOCKET=/tmp/livekit-enhance.sock
# --- optional latency knobs (defaults keep current behavior) ---
# AUDIO_QUEUE_MS=60 # LiveKit AudioSource queue; unset = max(150, hop_ms+50)
# VIDEO_HOLD_S=0 # hold published video this many seconds; unset = match hop; 0 = no hold
# WORKER_SPLIT_EXECUTOR=1 # dedicated thread pools: ffmpeg reads vs enhance
# ENHANCE_SPEAKER_ONLY=1 # MetricGAN+beamform only on DISPLAY_SPEAKER_CAMERA (rally)
# AUDIO_GATE=1 # optional hop mute; off by default so two talkers both publish
# AUDIO_GATE_OPEN_DB=-28
# AUDIO_GATE_CLOSE_DB=-34
# AUDIO_GATE_HOLD_S=0.3
# DISPLAY_VIDEO_CAPACITY=1 # VideoStream ring; 0 = unbounded, 1 = latest frame
# DISPLAY_VIDEO_FORMAT=bgra # request BGRA from SDK (skip extra convert)
# DISPLAY_GST_QUEUE_BUFFERS=1
# DISPLAY_SPEAKER_PUSH=arrival # timer | arrival (push speaker pane on decode)
#
# Suggested A/B (wall, ~hop 50 ms, speaker-only enhance):
# ENHANCE_HOP_S=0.05 AUDIO_QUEUE_MS=60 WORKER_SPLIT_EXECUTOR=1
# ENHANCE_SPEAKER_ONLY=1 DISPLAY_VIDEO_CAPACITY=1 DISPLAY_VIDEO_FORMAT=bgra
# DISPLAY_GST_QUEUE_BUFFERS=1 DISPLAY_SPEAKER_PUSH=arrival
# Aggressive (no neural enhance, 20 ms hops):
# ENHANCE_MODE=off BEAMFORM_MODE=off ENHANCE_HOP_S=0.02 AUDIO_QUEUE_MS=40 VIDEO_HOLD_S=0
# Misc
PUBLISH_TIMEOUT_S=60
@@ -57,12 +90,35 @@ DISPLAY_ROLES=grid,speaker,screenshare
# Empty = auto (connected heads first, prefer Arc HDMI-A-*)
DISPLAY_CONNECTORS=
DISPLAY_IDENTITY=display-wall
# Speaker camera pane: cam-NN or "active"
DISPLAY_SPEAKER_CAMERA=cam-01
# Speaker HDMI pane: LiveKit identity of the Logitech Rally (not a C920).
# The Rally is not on this host yet — pane shows a placeholder until `rally` joins.
DISPLAY_SPEAKER_CAMERA=rally
DISPLAY_GRID_COLS=5
DISPLAY_GRID_ROWS=4
DISPLAY_GRID_WIDTH=1920
DISPLAY_GRID_HEIGHT=1080
DISPLAY_GRID_FPS=30
# Yellow tile border when a participant raises a hand (MoveNet on the wall).
# DISPLAY_HAND_RAISE=0 to disable. Displays-only restart.
DISPLAY_HAND_RAISE=1
DISPLAY_HAND_RAISE_HZ=4
DISPLAY_HAND_RAISE_HOLD_S=1.0
# DISPLAY_HAND_RAISE_MODEL=thunder # thunder (256) or lightning (192); displays-only
# Green speaker chrome: LiveKit candidates, hop RMS picks louder vs two talkers.
# DISPLAY_SPEAK_MIN_DB=-40
# DISPLAY_SPEAK_RISE_DB=3
# DISPLAY_SPEAK_MARGIN_DB=6
# DISPLAY_SPEAK_HOLD_S=0.6
# DISPLAY_SPEAK_CORR=0.6 # hop xcorr; similar waveforms = bleed, keep louder
# DISPLAY_SPEAK_CONFIRM=2 # consecutive hops before a new seat greens
# Screen share is a stub until production LiveKit is online.
LIVEKIT_PUBLIC_URL=
# Person-aware background blur on C920 publish (OpenVINO CPU, not iGPU/Arc).
# Empty seats passthrough. Rally is not blurred. Rollback: PORTRAIT_BLUR=0 then
# cameras-only restart.
PORTRAIT_BLUR=1
# PORTRAIT_HZ=8
# PORTRAIT_BLUR_PX=7
# PORTRAIT_HOLD_S=0.8
# PORTRAIT_SHM_DIR=/run/livekit-cameras/portrait
+25 -16
View File
@@ -5,9 +5,13 @@ LiveKit room.
Each camera becomes one LiveKit participant (`cam-01` .. `cam-20`) that
publishes:
- a camera video track (640x360 @ 30 fps MJPG, H.264 VAAPI — sized
for low-latency wall decode, not max quality)
- a microphone audio track (16 kHz mono, cleaned with speechbrain)
- a camera video track (640x360 @ 30 fps MJPG, H.264 VAAPI on Arc).
1280x720@30 and 1920x1080@30 negotiate, but UVC on the 6–7 camera
USB 2.0 hubs fails buffer-pool allocation for most devices. Keep 30
fps; drop resolution if USB isoc is exhausted. YUYV 1080p is 5 fps
only — always MJPEG.
- a microphone audio track (16 kHz mono, cleaned with one shared
speechbrain MetricGAN process; workers do not each load the model)
Identity is the USB serial, pinned by udev (`/dev/camNN` + `slots.json`).
`--cameras 1,5` means participants `cam-01` and `cam-05`, not discovery order.
@@ -96,7 +100,10 @@ journalctl -u livekit-displays -f
```
`.env`: `DISPLAY_ROLES`, `DISPLAY_CONNECTORS`, `DISPLAY_SPEAKER_CAMERA`,
`DISPLAY_GRID_*`. kmssink takes the DRM master (framebuffer console on
`DISPLAY_GRID_*`. Local cameras publish LiveKit attributes `tag=uwh`
(`PARTICIPANT_TAGS`). The wall can hide those with `DISPLAY_HIDE_TAGS=uwh`
so remotes fill the mosaic instead; leave `DISPLAY_HIDE_TAGS` empty to keep
showing the local fleet (current default). kmssink takes the DRM master (framebuffer console on
HDMI-A-7 is replaced by the grid).
## Audio cleanup
@@ -116,6 +123,13 @@ stereo through a SpeechBrain preprocessor, then published as mono:
The 1 s window is **past context**, not +1 s delay: only the latest hop
is published. Algorithmic delay ≈ hop + compute.
Optional latency flags (all default to current behavior). Suggested wall
A/B: `ENHANCE_HOP_S=0.05 AUDIO_QUEUE_MS=60 WORKER_SPLIT_EXECUTOR=1
ENHANCE_SPEAKER_ONLY=1 DISPLAY_VIDEO_CAPACITY=1 DISPLAY_VIDEO_FORMAT=bgra
DISPLAY_GST_QUEUE_BUFFERS=1 DISPLAY_SPEAKER_PUSH=arrival`. Aggressive:
`ENHANCE_MODE=off BEAMFORM_MODE=off ENHANCE_HOP_S=0.02 AUDIO_QUEUE_MS=40
VIDEO_HOLD_S=0`. See `.env.example`.
Each worker caps Torch/BLAS/OpenVINO to `TORCH_NUM_THREADS=1` (default)
so 14–20 processes do not oversubscribe the i7-14700. This host has an
Intel Arc B580; the venv Torch is NVIDIA CUDA and **cannot** use it. Do
@@ -144,20 +158,15 @@ This box is meant to behave like an appliance:
## systemd (enabled)
- `livekit-server.service` — **local TEST** LiveKit on :7880 (autostart for now)
- `livekit-cameras.service` — publishers. `Wants=` the test server, does not
`Require` it. Destination is `LIVEKIT_URL` in `.env` (currently
`ws://127.0.0.1:7880`).
- `livekit-displays.service` — optional. Subscribes to `cam-01`/`cam-02`/`cam-03`
- `livekit-server.service` — **local TEST** LiveKit on :7880 (**disabled**;
cameras no longer `Wants=` it). Do not re-enable unless falling back
to localhost.
- `livekit-cameras.service` — publishers. Destination is `LIVEKIT_URL` in
`.env` (`wss://livekit.uni-wh.de:7800`, room `uwh-telhai`). Signaling
is WSS only; the client verifies the TLS certificate.
- `livekit-displays.service` — optional. Subscribes to the camera room
and drives 3 DRM connectors with GStreamer `kmssink` (no X11/Wayland).
When the real/remote LiveKit exists, change `.env` `LIVEKIT_URL` and:
```bash
systemctl disable --now livekit-server
systemctl restart livekit-cameras
```
```bash
systemctl status livekit-cameras livekit-server
journalctl -u livekit-cameras -f
+72
View File
@@ -0,0 +1,72 @@
"""Split Annex-B H.264 into access units."""
from __future__ import annotations
def _start_at(buf: bytes, i: int) -> int:
n = len(buf)
while i + 2 < n:
if buf[i] == 0 and buf[i + 1] == 0:
if buf[i + 2] == 1:
return i
if i + 3 < n and buf[i + 2] == 0 and buf[i + 3] == 1:
return i
i += 1
return -1
def nal_unit_type(nal: bytes) -> int:
"""First NAL type in an AU (5 = IDR)."""
i = _start_at(nal, 0)
if i < 0:
return 0
hdr = i + (4 if i + 3 < len(nal) and nal[i + 2] == 0 else 3)
if hdr >= len(nal):
return 0
return nal[hdr] & 0x1F
def is_keyframe(au: bytes) -> bool:
i = 0
n = len(au)
while True:
s = _start_at(au, i)
if s < 0:
return False
hdr = s + (4 if s + 3 < n and au[s + 2] == 0 else 3)
if hdr >= n:
return False
ntype = au[hdr] & 0x1F
if ntype == 5:
return True
i = hdr + 1
class AnnexBSplitter:
"""Feed byte-stream; yield complete AUs (including start codes)."""
def __init__(self) -> None:
self._buf = bytearray()
def push(self, data: bytes) -> list[bytes]:
if data:
self._buf.extend(data)
out: list[bytes] = []
buf = self._buf
first = _start_at(bytes(buf), 0)
if first < 0:
if len(buf) > 1_000_000:
del buf[: len(buf) - 8]
return out
if first:
del buf[:first]
pos = 0
while True:
nxt = _start_at(bytes(buf), pos + 3)
if nxt < 0:
break
au = bytes(buf[:nxt])
if au:
out.append(au)
del buf[:nxt]
pos = 0
return out
+52 -1
View File
@@ -32,6 +32,8 @@ import time
import numpy as np
from audio_gate import hop_mono_int16
log = logging.getLogger("audio.cleanup")
SAMPLE_RATE = 16000
@@ -66,7 +68,10 @@ class AudioCleaner:
beamform: str = "auto",
torch_num_threads: int = 1,
tdoa_every: int = 5,
enhance_infer: str = "auto"):
enhance_infer: str = "auto",
enhance_socket: str = "",
identity: str = "",
gate=None):
self.mode = mode
self.beamform_mode = beamform
self.sample_rate = sample_rate
@@ -93,6 +98,19 @@ class AudioCleaner:
self._ov_compiled = None
self._ov_request = None
self._ov_input = None
self.enhance_socket = (enhance_socket or "").strip()
self.identity = identity or ""
self._client = None
self._gate = gate
if self.enhance_socket and mode != "off":
self._connect_daemon()
if self._client is not None:
return
if mode == "force":
raise RuntimeError(
f"ENHANCE_MODE=force but enhance daemon "
f"{self.enhance_socket!r} is unreachable")
self._load_torch()
if beamform == "force":
@@ -121,6 +139,19 @@ class AudioCleaner:
self._refresh_backend()
def _connect_daemon(self) -> None:
from enhance_ipc import EnhanceClient
try:
self._client = EnhanceClient(self.enhance_socket)
except Exception as e: # noqa: BLE001
log.warning("enhance daemon %s unreachable (%s)", self.enhance_socket, e)
self._client = None
return
self.backend = "daemon"
log.info("audio enhance via daemon %s identity=%s",
self.enhance_socket, self.identity or "-")
def _refresh_backend(self) -> None:
parts = []
if self._enhancer is not None:
@@ -323,6 +354,11 @@ class AudioCleaner:
pcm = np.asarray(pcm)
t0 = time.perf_counter()
try:
if self._gate is not None:
n = pcm.shape[0]
self._gate.apply(hop_mono_int16(pcm))
if not self._gate.open:
return np.zeros(n, dtype=np.int16)
return self._process_inner(pcm)
finally:
self.last_dt_ms = (time.perf_counter() - t0) * 1000.0
@@ -340,6 +376,21 @@ class AudioCleaner:
n = frames.shape[0]
x = frames.astype(np.float32) / 32768.0
if self._client is not None:
try:
out = self._client.process(self.identity, frames.astype(np.int16))
out = np.asarray(out, dtype=np.int16)
if out.shape[0] != n:
if out.shape[0] > n:
out = out[:n]
else:
out = np.pad(out, (0, n - out.shape[0]))
return out
except Exception as e: # noqa: BLE001
log.warning("enhance daemon failed (%s); using fallback", e)
mono_hop = frames.astype(np.float32).reshape(n, -1).mean(axis=1) / 32768.0
return self._to_int16(self._simple_clean(mono_hop))
# Delay-and-Sum the *hop* (200 ms is ~60 ms here). Beamforming the
# full 1 s window every hop is ~330 ms — not 20-camera real-time.
# Same tutorial modules; TDOA is estimated per hop via GCC-PHAT.
+158
View File
@@ -0,0 +1,158 @@
"""Shared C920 mic capture: small gst-launch pool, PCM hops in mmap rings."""
from __future__ import annotations
import argparse
import json
import logging
import os
import select
import signal
import subprocess
import sys
import time
from audio_gst import GST_GROUP_SIZE, grouped_audio, shared_audio_cmd
from hop_shm import HopWriter, hop_bytes
log = logging.getLogger("cameras.audio_daemon")
READY_NAME = "ready"
def _open_pipes(n: int) -> tuple[list[int], list[int]]:
reads: list[int] = []
writes: list[int] = []
for _ in range(n):
r, w = os.pipe()
try:
import fcntl
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 65536)
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 65536)
except OSError:
pass
os.set_inheritable(w, True)
os.set_inheritable(r, False)
reads.append(r)
writes.append(w)
return reads, writes
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(description="Shared ALSA capture daemon")
ap.add_argument("--cameras-json", required=True)
ap.add_argument("--rate", type=int, default=16000)
ap.add_argument("--hop-s", type=float, default=0.1)
ap.add_argument("--shm-dir", default="/run/livekit-cameras/pcm")
ap.add_argument("--group-size", type=int, default=GST_GROUP_SIZE)
args = ap.parse_args(argv)
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [audio-daemon] %(levelname)s %(message)s")
cams = json.loads(args.cameras_json)
pairs = [(c["identity"], int(c["audio_card"]))
for c in cams if c.get("audio_card") is not None]
if not pairs:
log.error("no audio cards")
return 2
need = hop_bytes(args.rate, 2, args.hop_s)
os.makedirs(args.shm_dir, exist_ok=True)
writers = [
HopWriter(os.path.join(args.shm_dir, f"{ident}.pcm"), need)
for ident, _card in pairs
]
reads, writes = _open_pipes(len(pairs))
env = dict(os.environ)
groups = grouped_audio(pairs, writes, args.group_size)
log.info("starting %d gst-launch for %d mics rate=%d hop=%.3f (group=%d)",
len(groups), len(pairs), args.rate, args.hop_s, args.group_size)
procs: list[subprocess.Popen] = []
for g_cams, g_fds in groups:
cmd = shared_audio_cmd(g_cams, args.rate, g_fds)
procs.append(subprocess.Popen(
cmd,
stdin=subprocess.DEVNULL,
stdout=subprocess.DEVNULL,
stderr=None,
pass_fds=tuple(g_fds),
env=env,
close_fds=True,
))
for w in writes:
os.close(w)
for r in reads:
os.set_blocking(r, False)
bufs = [bytearray() for _ in pairs]
seqs = [0] * len(pairs)
stop = False
def _stop(_signum=None, _frame=None):
nonlocal stop
stop = True
signal.signal(signal.SIGTERM, _stop)
signal.signal(signal.SIGINT, _stop)
ready_path = os.path.join(args.shm_dir, READY_NAME)
if os.path.exists(ready_path):
os.unlink(ready_path)
marked = False
last_log = time.monotonic()
fd_index = {r: i for i, r in enumerate(reads)}
try:
while not stop:
dead = [p for p in procs if p.poll() is not None]
if dead:
log.error("gst exited rc=%s", [p.returncode for p in dead])
break
ready, _, _ = select.select(reads, [], [], 0.2)
for r in ready:
i = fd_index[r]
while True:
try:
chunk = os.read(r, max(need, need - len(bufs[i])))
except BlockingIOError:
break
if not chunk:
break
bufs[i].extend(chunk)
while len(bufs[i]) >= need:
hop = bytes(bufs[i][:need])
del bufs[i][:need]
seqs[i] = writers[i].write(hop)
if not marked and any(s > 0 for s in seqs):
open(ready_path, "w").write("ok\n")
marked = True
log.info("ready first hops %s",
",".join(f"{pairs[i][0]}={seqs[i]}" for i in range(len(pairs))))
now = time.monotonic()
if now - last_log >= 5.0:
last_log = now
log.info("seqs %s",
" ".join(f"{pairs[i][0]}:{seqs[i]}" for i in range(len(pairs))))
finally:
for proc in procs:
if proc.poll() is None:
proc.terminate()
try:
proc.wait(timeout=5)
except subprocess.TimeoutExpired:
proc.kill()
for r in reads:
try:
os.close(r)
except OSError:
pass
for w in writers:
w.close()
if os.path.exists(ready_path):
os.unlink(ready_path)
return 0
if __name__ == "__main__":
sys.exit(main())
+91
View File
@@ -0,0 +1,91 @@
"""Per-hop noise gate for C920 mics (LiveKit active-speaker needs distinct energy)."""
from __future__ import annotations
import numpy as np
FLOOR_DB = -120.0
def rms_dbfs(pcm: np.ndarray) -> float:
"""RMS of int16 PCM in dBFS. Empty/zero -> FLOOR_DB."""
x = np.asarray(pcm)
if x.size == 0:
return FLOOR_DB
rms = float(np.sqrt(np.mean(np.square(x.astype(np.float64) / 32768.0))))
if rms <= 1e-12:
return FLOOR_DB
return 20.0 * float(np.log10(rms))
def hop_mono_int16(pcm: np.ndarray) -> np.ndarray:
x = np.asarray(pcm)
if x.ndim == 2:
return np.mean(x.astype(np.float32), axis=1).astype(np.int16)
return np.asarray(x, dtype=np.int16)
def gate_enabled_for(identity: str, speaker_camera: str, enabled: bool) -> bool:
"""Dante/rally stays ungated; C920 seats are gated when enabled."""
if not enabled:
return False
ident = (identity or "").strip()
skip = (speaker_camera or "").strip()
if ident and skip and ident == skip:
return False
return True
def hold_hops(hold_s: float, hop_s: float) -> int:
hop = max(1e-6, float(hop_s))
return max(1, int(round(float(hold_s) / hop)))
def make_gate(
identity: str,
speaker_camera: str,
enabled: bool,
open_db: float,
close_db: float,
hold_s: float,
hop_s: float,
) -> HopGate | None:
if not gate_enabled_for(identity, speaker_camera, enabled):
return None
return HopGate(
open_db=open_db,
close_db=close_db,
hold_hops=hold_hops(hold_s, hop_s),
)
class HopGate:
"""Open on close-talk energy; hangover then mute so neighbors stay silent."""
def __init__(
self,
open_db: float = -32.0,
close_db: float = -40.0,
hold_hops: int = 3,
) -> None:
self.open_db = float(open_db)
self.close_db = min(float(close_db), self.open_db)
self.hold_hops = max(0, int(hold_hops))
self.open = False
self.hold = 0
def apply(self, pcm: np.ndarray) -> np.ndarray:
pcm = np.asarray(pcm, dtype=np.int16)
db = rms_dbfs(pcm)
if db >= self.open_db:
self.open = True
self.hold = self.hold_hops
return pcm
if self.open:
if db >= self.close_db:
self.hold = self.hold_hops
return pcm
if self.hold > 0:
self.hold -= 1
return pcm
self.open = False
return np.zeros_like(pcm)
+40
View File
@@ -0,0 +1,40 @@
"""GStreamer ALSA capture graphs for the shared audio daemon."""
from __future__ import annotations
import shutil
GST_LAUNCH = shutil.which("gst-launch-1.0") or "gst-launch-1.0"
GST_GROUP_SIZE = 5
def shared_audio_cmd(
cameras: list[tuple[str, int]],
rate: int,
fds: list[int],
) -> list[str]:
if len(cameras) != len(fds):
raise ValueError("cameras and fds length mismatch")
cmd: list[str] = [GST_LAUNCH, "-q"]
caps = f"audio/x-raw,format=S16LE,rate={int(rate)},channels=2"
for (_ident, card), fd in zip(cameras, fds):
cmd += [
"alsasrc", f"device=hw:{int(card)},0", "do-timestamp=false",
"!", caps,
"!", "queue", "max-size-buffers=2", "max-size-bytes=0",
"max-size-time=0", "leaky=downstream",
"!", "fdsink", f"fd={int(fd)}", "sync=false",
]
return cmd
def grouped_audio(
cameras: list[tuple[str, int]],
fds: list[int],
group_size: int = GST_GROUP_SIZE,
) -> list[tuple[list[tuple[str, int]], list[int]]]:
if group_size < 1:
raise ValueError("group_size must be >= 1")
out = []
for i in range(0, len(cameras), group_size):
out.append((cameras[i:i + group_size], fds[i:i + group_size]))
return out
+123
View File
@@ -0,0 +1,123 @@
"""Split AV1 low-overhead OBU streams into temporal units."""
from __future__ import annotations
OBU_SEQUENCE_HEADER = 1
OBU_TEMPORAL_DELIMITER = 2
def _leb128(data: bytes, i: int) -> tuple[int, int] | None:
n = len(data)
result = 0
shift = 0
while i < n:
byte = data[i]
i += 1
result |= (byte & 0x7F) << shift
if (byte & 0x80) == 0:
return result, i
shift += 7
if shift > 56:
return None
return None
def parse_obus(data: bytes) -> list[tuple[int, int, int]]:
"""Return (type, start, end) for each complete OBU. Empty if malformed."""
out: list[tuple[int, int, int]] = []
i = 0
n = len(data)
while i < n:
start = i
hdr = data[i]
i += 1
if hdr & 0x80:
return []
obu_type = (hdr >> 3) & 0x0F
has_ext = (hdr & 0x04) != 0
has_size = (hdr & 0x02) != 0
if has_ext:
if i >= n:
return []
i += 1
if has_size:
leb = _leb128(data, i)
if leb is None:
return []
size, i = leb
end = i + size
if end > n:
return []
i = end
else:
i = n
out.append((obu_type, start, i))
return out
def is_keyframe(tu: bytes) -> bool:
return any(t == OBU_SEQUENCE_HEADER for t, _s, _e in parse_obus(tu))
class Av1TuSplitter:
"""Feed obu-stream bytes; yield complete temporal units."""
def __init__(self) -> None:
self._buf = bytearray()
def push(self, data: bytes) -> list[bytes]:
if data:
self._buf.extend(data)
out: list[bytes] = []
buf = self._buf
while True:
n = len(buf)
if n < 1:
break
i = 0
tu_end = -1
saw_coded = False
ok = True
while i < n:
start = i
hdr = buf[i]
i += 1
if hdr & 0x80:
ok = False
break
obu_type = (hdr >> 3) & 0x0F
has_ext = (hdr & 0x04) != 0
has_size = (hdr & 0x02) != 0
if has_ext:
if i >= n:
ok = False
break
i += 1
if has_size:
leb = _leb128(bytes(buf), i)
if leb is None:
ok = False
break
size, i = leb
if i + size > n:
ok = False
break
i = i + size
else:
ok = False
break
if obu_type == OBU_TEMPORAL_DELIMITER and saw_coded:
tu_end = start
break
if obu_type != OBU_TEMPORAL_DELIMITER:
saw_coded = True
if not ok:
if n > 2_000_000:
del buf[: n - 16]
break
if tu_end < 0:
break
tu = bytes(buf[:tu_end])
if tu:
out.append(tu)
del buf[:tu_end]
return out
+176
View File
@@ -0,0 +1,176 @@
"""Shared C920 capture: small gst-launch pool, latest I420 frames in mmap."""
from __future__ import annotations
import argparse
import json
import logging
import os
import select
import signal
import subprocess
import sys
import time
from capture_va import (
GST_GROUP_SIZE,
grouped_cameras,
shared_capture_cmd,
)
from frame_shm import LatestFrameWriter, frame_bytes
from uvc_exposure import pin_devices
log = logging.getLogger("cameras.capture_daemon")
READY_NAME = "ready"
def _open_pipes(n: int) -> tuple[list[int], list[int]]:
reads: list[int] = []
writes: list[int] = []
for _ in range(n):
r, w = os.pipe()
try:
import fcntl
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 1048576)
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 1048576)
except OSError:
pass
os.set_inheritable(w, True)
os.set_inheritable(r, False)
reads.append(r)
writes.append(w)
return reads, writes
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(description="Shared JPEG capture daemon")
ap.add_argument("--cameras-json", required=True,
help="JSON list of {identity, video_device}")
ap.add_argument("--width", type=int, default=640)
ap.add_argument("--height", type=int, default=360)
ap.add_argument("--fps", type=int, default=30)
ap.add_argument("--shm-dir", default="/run/livekit-cameras/raw")
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
ap.add_argument("--group-size", type=int, default=GST_GROUP_SIZE)
args = ap.parse_args(argv)
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [capture-daemon] %(levelname)s %(message)s")
cams = json.loads(args.cameras_json)
pairs = [(c["identity"], c["video_device"]) for c in cams]
if not pairs:
log.error("no cameras")
return 2
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
need = frame_bytes(args.width, args.height)
os.makedirs(args.shm_dir, exist_ok=True)
writers = [
LatestFrameWriter(os.path.join(args.shm_dir, f"{ident}.i420"),
args.width, args.height)
for ident, _dev in pairs
]
reads, writes = _open_pipes(len(pairs))
env = dict(os.environ)
env["LIBVA_DRM_DEVICE"] = args.vaapi_device
env["LIBVA_DRIVER_NAME"] = "iHD"
env["GST_REGISTRY_UPDATE"] = "no"
groups = grouped_cameras(pairs, writes, args.group_size)
log.info("starting %d gst-launch for %d cameras %dx%d@%d (group=%d)",
len(groups), len(pairs), args.width, args.height, args.fps,
args.group_size)
procs: list[subprocess.Popen] = []
for g_cams, g_fds in groups:
cmd = shared_capture_cmd(g_cams, args.width, args.height, args.fps, g_fds)
procs.append(subprocess.Popen(
cmd,
stdin=subprocess.DEVNULL,
stdout=subprocess.DEVNULL,
stderr=None,
pass_fds=tuple(g_fds),
env=env,
close_fds=True,
))
for w in writes:
os.close(w)
for r in reads:
os.set_blocking(r, False)
pin_devices([dev for _ident, dev in pairs])
bufs = [bytearray() for _ in pairs]
seqs = [0] * len(pairs)
stop = False
def _stop(_signum=None, _frame=None):
nonlocal stop
stop = True
signal.signal(signal.SIGTERM, _stop)
signal.signal(signal.SIGINT, _stop)
ready_path = os.path.join(args.shm_dir, READY_NAME)
if os.path.exists(ready_path):
os.unlink(ready_path)
marked = False
last_log = time.monotonic()
fd_index = {r: i for i, r in enumerate(reads)}
try:
while not stop:
dead = [p for p in procs if p.poll() is not None]
if dead:
log.error("gst exited rc=%s", [p.returncode for p in dead])
break
ready, _, _ = select.select(reads, [], [], 0.2)
now_ns = time.time_ns()
for r in ready:
i = fd_index[r]
while True:
try:
chunk = os.read(r, max(need, need - len(bufs[i])))
except BlockingIOError:
break
if not chunk:
break
bufs[i].extend(chunk)
while len(bufs[i]) >= need:
frame = bytearray(bufs[i][:need])
del bufs[i][:need]
seqs[i] = writers[i].write(frame, now_ns)
if not marked and any(s > 0 for s in seqs):
pin_devices([dev for _ident, dev in pairs])
open(ready_path, "w").write("ok\n")
marked = True
log.info("ready first frames %s",
",".join(f"{pairs[i][0]}={seqs[i]}" for i in range(len(pairs))))
now = time.monotonic()
if now - last_log >= 5.0:
last_log = now
log.info("seqs %s",
" ".join(f"{pairs[i][0]}:{seqs[i]}" for i in range(len(pairs))))
finally:
for proc in procs:
if proc.poll() is None:
proc.terminate()
try:
proc.wait(timeout=5)
except subprocess.TimeoutExpired:
proc.kill()
for r in reads:
try:
os.close(r)
except OSError:
pass
for w in writers:
w.close()
if os.path.exists(ready_path):
os.unlink(ready_path)
return 0
if __name__ == "__main__":
sys.exit(main())
+175
View File
@@ -0,0 +1,175 @@
"""GStreamer capture: C920 MJPEG -> jpegdec -> I420.
Shared capture daemon hosts a small pool of gst-launch processes (not
one per worker). No sidecar H.264: LiveKit VAAPI is the only encoder.
"""
from __future__ import annotations
import shutil
GST_LAUNCH = shutil.which("gst-launch-1.0") or "gst-launch-1.0"
# 5 cameras per gst-launch. group=1 does not raise fps (UVC still 15);
# a small pool still cuts 20 worker pipelines.
GST_GROUP_SIZE = 5
_QUEUE = (
"queue",
"max-size-buffers=1",
"max-size-bytes=0",
"max-size-time=0",
"leaky=downstream",
)
def _caps(width: int, height: int, fps: int) -> str:
fps = max(1, int(fps))
return f"image/jpeg,width={int(width)},height={int(height)},framerate={fps}/1"
JPEG_MCU_FLICKER_CAMS = frozenset()
def jpeg_mcu_bottom_crop(height: int) -> int:
"""Rows in the last 16-tall 4:2:0 JPEG MCU (0 if height is aligned).
C920 MJPEG 640x360 leaves 8 leftover rows. Some sensors fill that
pad with garbage that flickers on the wall (cam-07/14/15).
"""
h = int(height)
if h < 32:
return 0
rem = h % 16
if rem == 0:
return 0
return rem + 16
def needs_jpeg_mcu_fix(ident: str) -> bool:
return ident in JPEG_MCU_FLICKER_CAMS
def crop_scale_i420_drop_bottom(
buf: bytearray | bytes, width: int, height: int, crop: int
) -> bytearray:
"""Drop the bottom `crop` I420 rows; fill with studio black (Y=16).
Unused on the live path: Y=16 is a raised bar once encode is 16-aligned.
Clone is the live fix on JPEG_MCU_FLICKER_CAMS. Kept for tests.
"""
crop = int(crop)
width, height = int(width), int(height)
need = width * height * 3 // 2
out = bytearray(buf[:need] if len(buf) >= need else buf)
if crop <= 0 or height <= crop or width < 2 or height % 2 or crop % 2:
return out
y_stride = width
y_size = width * height
uv_w = width // 2
uv_h = height // 2
for r in range(height - crop, height):
dst = r * y_stride
out[dst:dst + y_stride] = bytes([16]) * y_stride
uv_crop = crop // 2
u_base = y_size
v_base = y_size + uv_w * uv_h
z = bytes([128]) * uv_w
for r in range(uv_h - uv_crop, uv_h):
out[u_base + r * uv_w:u_base + (r + 1) * uv_w] = z
out[v_base + r * uv_w:v_base + (r + 1) * uv_w] = z
return out
def repeat_i420_bottom_from_last_good(
buf: bytearray, width: int, height: int, crop: int
) -> None:
"""Overwrite the bottom `crop` I420 rows by cloning the last good line.
Used after videocrop+videobox (studio-black pad). Cloning beats a
visible 8px bar and beats videoscale (which re-edged cam-07).
"""
crop = int(crop)
if crop <= 0:
return
width, height = int(width), int(height)
if width <= 0 or height <= crop or width % 2 or height % 2 or crop % 2:
return
y_stride = width
y_size = width * height
uv_w = width // 2
uv_h = height // 2
uv_crop = crop // 2
last_y = height - crop - 1
src_y = last_y * y_stride
src_row = buf[src_y:src_y + y_stride]
for r in range(height - crop, height):
dst = r * y_stride
buf[dst:dst + y_stride] = src_row
last_uv = uv_h - uv_crop - 1
u_base = y_size
v_base = y_size + uv_w * uv_h
src_u = buf[u_base + last_uv * uv_w:u_base + (last_uv + 1) * uv_w]
src_v = buf[v_base + last_uv * uv_w:v_base + (last_uv + 1) * uv_w]
for r in range(uv_h - uv_crop, uv_h):
buf[u_base + r * uv_w:u_base + (r + 1) * uv_w] = src_u
buf[v_base + r * uv_w:v_base + (r + 1) * uv_w] = src_v
def _one_cam(device: str, width: int, height: int, fps: int, fd: int) -> list[str]:
width, height = int(width), int(height)
tail: list[str] = [
"videoconvert", "qos=false", "n-threads=2",
"!", "video/x-raw,format=I420",
"!", *_QUEUE,
"!", "fdsink", f"fd={int(fd)}", "sync=false",
]
return [
"v4l2src", f"device={device}", "do-timestamp=false",
"!", _caps(width, height, fps),
"!", "jpegparse",
"!", "jpegdec", "qos=false",
"!", *tail,
]
def video_capture_cmd(
device: str,
width: int,
height: int,
fps: int,
h264_fifo: str | None = None, # ignored: no sidecar encode
bitrate_kbps: int = 800,
fd: int = 1,
) -> list[str]:
"""Single-camera pipeline: jpegdec -> I420, latest-frame queue, fdsink."""
del h264_fifo, bitrate_kbps
return [GST_LAUNCH, "-q", *_one_cam(device, width, height, fps, fd)]
def shared_capture_cmd(
cameras: list[tuple[str, str]],
width: int,
height: int,
fps: int,
fds: list[int],
) -> list[str]:
"""One gst-launch with N jpegdec->I420->fdsink graphs (no h264enc)."""
if len(cameras) != len(fds):
raise ValueError("cameras and fds length mismatch")
cmd: list[str] = [GST_LAUNCH, "-q"]
for (_ident, device), fd in zip(cameras, fds):
cmd += _one_cam(device, width, height, fps, fd)
return cmd
def grouped_cameras(
cameras: list[tuple[str, str]],
fds: list[int],
group_size: int = GST_GROUP_SIZE,
) -> list[tuple[list[tuple[str, str]], list[int]]]:
if group_size < 1:
raise ValueError("group_size must be >= 1")
out: list[tuple[list[tuple[str, str]], list[int]]] = []
for i in range(0, len(cameras), group_size):
out.append((cameras[i:i + group_size], fds[i:i + group_size]))
return out
+170 -11
View File
@@ -26,11 +26,12 @@ class Config:
room: str = "cameras"
participant_prefix: str = "cam"
# video
width: int = 640
height: int = 360
fps: int = 30
width: int = 320
height: int = 180
fps: int = 15
min_fps: int = 5
video_bitrate: int = 800_000
video_bitrate: int = 400_000
rally_video_bitrate: int = 20_000_000
video_codec: str = "h264"
video_encoder: str = "vaapi" # auto|software|hardware|nvenc|vaapi
vaapi_device: str = "/dev/dri/renderD129"
@@ -47,6 +48,29 @@ class Config:
beamform_mode: str = "auto" # auto | force | off
torch_num_threads: int = 1
beamform_tdoa_every: int = 5
# latency knobs (unset / False = legacy behavior)
audio_queue_ms: Optional[int] = None # AUDIO_QUEUE_MS; None = max(150, hop_ms+50)
video_hold_s: Optional[float] = None # VIDEO_HOLD_S; None = match hop
worker_split_executor: bool = False
enhance_speaker_only: bool = False
audio_gate: bool = False
audio_gate_open_db: float = -28.0
audio_gate_close_db: float = -34.0
audio_gate_hold_s: float = 0.3
enhance_daemon: bool = True
enhance_socket: str = "/tmp/livekit-enhance.sock"
capture_daemon: bool = True
capture_shm_dir: str = "/run/livekit-cameras/raw"
audio_daemon: bool = True
audio_shm_dir: str = "/run/livekit-cameras/pcm"
encode_daemon: bool = True
encode_shm_dir: str = "/run/livekit-cameras/h264"
single_publisher: bool = True
display_video_capacity: int = 0 # 0 = unbounded VideoStream
display_video_format: str = "" # "" = SDK default, "bgra" | "i420"
display_gst_queue_buffers: int = 2
display_speaker_push: str = "timer" # timer | arrival
display_pump_workers: int = 4
# misc
publish_timeout_s: float = 60.0
log_level: str = "INFO"
@@ -59,7 +83,7 @@ class Config:
display_identity: str = "display-wall"
display_roles: list[str] = field(
default_factory=lambda: ["grid", "speaker", "screenshare"])
display_speaker_camera: str = "cam-01" # cam-NN or "active"
display_speaker_camera: str = "rally" # LiveKit identity; not a C920
display_speaker_participant: str = "speaker"
display_grid_cols: int = 5
display_grid_rows: int = 4
@@ -67,6 +91,28 @@ class Config:
display_grid_height: int = 1080
display_grid_fps: int = 30
livekit_public_url: str = ""
# Site tag published on local cameras (LiveKit participant attributes).
participant_tags: list[str] = field(default_factory=lambda: ["uwh"])
# Display wall: skip participants carrying these tags. Empty = off
# (show local cameras). Set to ["uwh"] to show remotes instead.
display_hide_tags: list[str] = field(default_factory=list)
# Yellow tile chrome when MoveNet sees a raised hand (display process only).
display_hand_raise: bool = True
display_hand_raise_hz: float = 4.0
display_hand_raise_hold_s: float = 1.0
display_hand_raise_model: str = "thunder"
display_speak_min_db: float = -40.0
display_speak_margin_db: float = 6.0
display_speak_hold_s: float = 0.6
display_speak_rise_db: float = 3.0
display_speak_corr: float = 0.4
display_speak_confirm: int = 2
# Person-aware background blur (C920s only; OpenVINO CPU). Rollback: PORTRAIT_BLUR=0.
portrait_blur: bool = True
portrait_shm_dir: str = "/run/livekit-cameras/portrait"
portrait_hz: float = 8.0
portrait_blur_px: int = 7
portrait_hold_s: float = 0.8
def _parse_cameras(spec: str) -> list[str]:
@@ -95,6 +141,31 @@ def _parse_list(spec: str) -> list[str]:
return [p.strip() for p in spec.split(",") if p.strip()]
def _parse_bool(spec: str, default: bool = False) -> bool:
raw = (spec or "").strip().lower()
if not raw:
return default
if raw in ("1", "true", "yes", "on"):
return True
if raw in ("0", "false", "no", "off"):
return False
raise ValueError(f"invalid bool {spec!r}")
def _opt_int(spec: str) -> Optional[int]:
raw = (spec or "").strip()
if raw == "":
return None
return int(raw)
def _opt_float(spec: str) -> Optional[float]:
raw = (spec or "").strip()
if raw == "":
return None
return float(raw)
def load_config(env_path: Optional[Path] = None) -> Config:
_load_dotenv(Path(env_path or Path(__file__).parent / ".env"))
@@ -102,13 +173,14 @@ def load_config(env_path: Optional[Path] = None) -> Config:
url=os.environ.get("LIVEKIT_URL", ""),
api_key=os.environ.get("LIVEKIT_API_KEY", ""),
api_secret=os.environ.get("LIVEKIT_API_SECRET", ""),
room=os.environ.get("LIVEKIT_ROOM", "cameras"),
room=os.environ.get("LIVEKIT_ROOM", "uwh-telhai"),
participant_prefix=os.environ.get("PARTICIPANT_PREFIX", "cam"),
width=int(os.environ.get("VIDEO_WIDTH", "640")),
height=int(os.environ.get("VIDEO_HEIGHT", "360")),
fps=int(os.environ.get("VIDEO_FPS", "30")),
width=int(os.environ.get("VIDEO_WIDTH", "320")),
height=int(os.environ.get("VIDEO_HEIGHT", "180")),
fps=int(os.environ.get("VIDEO_FPS", "15")),
min_fps=int(os.environ.get("VIDEO_MIN_FPS", "5")),
video_bitrate=int(os.environ.get("VIDEO_BITRATE", "800000")),
video_bitrate=int(os.environ.get("VIDEO_BITRATE", "400000")),
rally_video_bitrate=int(os.environ.get("VIDEO_RALLY_BITRATE", "20000000")),
video_codec=os.environ.get("VIDEO_CODEC", "h264").lower(),
video_encoder=os.environ.get("VIDEO_ENCODER", "vaapi").lower(),
vaapi_device=os.environ.get("LIBVA_DRM_DEVICE", "/dev/dri/renderD129"),
@@ -124,6 +196,32 @@ def load_config(env_path: Optional[Path] = None) -> Config:
beamform_mode=os.environ.get("BEAMFORM_MODE", "auto").lower(),
torch_num_threads=int(os.environ.get("TORCH_NUM_THREADS", "1")),
beamform_tdoa_every=int(os.environ.get("BEAMFORM_TDOA_EVERY", "5")),
audio_queue_ms=_opt_int(os.environ.get("AUDIO_QUEUE_MS", "")),
video_hold_s=_opt_float(os.environ.get("VIDEO_HOLD_S", "")),
worker_split_executor=_parse_bool(os.environ.get("WORKER_SPLIT_EXECUTOR", "")),
enhance_speaker_only=_parse_bool(os.environ.get("ENHANCE_SPEAKER_ONLY", "")),
audio_gate=_parse_bool(os.environ.get("AUDIO_GATE", ""), default=False),
audio_gate_open_db=float(os.environ.get("AUDIO_GATE_OPEN_DB", "-28")),
audio_gate_close_db=float(os.environ.get("AUDIO_GATE_CLOSE_DB", "-34")),
audio_gate_hold_s=float(os.environ.get("AUDIO_GATE_HOLD_S", "0.3")),
enhance_daemon=_parse_bool(os.environ.get("ENHANCE_DAEMON", ""), default=True),
enhance_socket=os.environ.get("ENHANCE_SOCKET", "/tmp/livekit-enhance.sock").strip()
or "/tmp/livekit-enhance.sock",
capture_daemon=_parse_bool(os.environ.get("CAPTURE_DAEMON", ""), default=True),
capture_shm_dir=os.environ.get("CAPTURE_SHM_DIR", "/run/livekit-cameras/raw").strip()
or "/run/livekit-cameras/raw",
audio_daemon=_parse_bool(os.environ.get("AUDIO_DAEMON", ""), default=True),
audio_shm_dir=os.environ.get("AUDIO_SHM_DIR", "/run/livekit-cameras/pcm").strip()
or "/run/livekit-cameras/pcm",
encode_daemon=_parse_bool(os.environ.get("ENCODE_DAEMON", ""), default=True),
encode_shm_dir=os.environ.get("ENCODE_SHM_DIR", "/run/livekit-cameras/h264").strip()
or "/run/livekit-cameras/h264",
single_publisher=_parse_bool(os.environ.get("SINGLE_PUBLISHER", ""), default=True),
display_video_capacity=int(os.environ.get("DISPLAY_VIDEO_CAPACITY", "0")),
display_video_format=os.environ.get("DISPLAY_VIDEO_FORMAT", "").strip().lower(),
display_gst_queue_buffers=int(os.environ.get("DISPLAY_GST_QUEUE_BUFFERS", "2")),
display_speaker_push=os.environ.get("DISPLAY_SPEAKER_PUSH", "timer").strip().lower(),
display_pump_workers=int(os.environ.get("DISPLAY_PUMP_WORKERS", "4")),
publish_timeout_s=float(os.environ.get("PUBLISH_TIMEOUT_S", "60")),
log_level=os.environ.get("LOG_LEVEL", "INFO").upper(),
cameras=_parse_cameras(os.environ.get("CAMERAS", "all")),
@@ -135,7 +233,8 @@ def load_config(env_path: Optional[Path] = None) -> Config:
display_roles=_parse_list(
os.environ.get("DISPLAY_ROLES", "grid,speaker,screenshare"))
or ["grid", "speaker", "screenshare"],
display_speaker_camera=os.environ.get("DISPLAY_SPEAKER_CAMERA", "cam-01"),
display_speaker_camera=os.environ.get("DISPLAY_SPEAKER_CAMERA", "rally").strip()
or "rally",
display_speaker_participant=os.environ.get(
"DISPLAY_SPEAKER_PARTICIPANT", "speaker"),
display_grid_cols=int(os.environ.get("DISPLAY_GRID_COLS", "5")),
@@ -144,6 +243,25 @@ def load_config(env_path: Optional[Path] = None) -> Config:
display_grid_height=int(os.environ.get("DISPLAY_GRID_HEIGHT", "1080")),
display_grid_fps=int(os.environ.get("DISPLAY_GRID_FPS", "30")),
livekit_public_url=os.environ.get("LIVEKIT_PUBLIC_URL", ""),
participant_tags=_parse_list(os.environ.get("PARTICIPANT_TAGS", "uwh")),
display_hide_tags=_parse_list(os.environ.get("DISPLAY_HIDE_TAGS", "")),
display_hand_raise=_parse_bool(os.environ.get("DISPLAY_HAND_RAISE", ""), default=True),
display_hand_raise_hz=float(os.environ.get("DISPLAY_HAND_RAISE_HZ", "4")),
display_hand_raise_hold_s=float(os.environ.get("DISPLAY_HAND_RAISE_HOLD_S", "1.0")),
display_hand_raise_model=os.environ.get("DISPLAY_HAND_RAISE_MODEL", "thunder").strip().lower(),
display_speak_min_db=float(os.environ.get("DISPLAY_SPEAK_MIN_DB", "-40")),
display_speak_margin_db=float(os.environ.get("DISPLAY_SPEAK_MARGIN_DB", "6")),
display_speak_hold_s=float(os.environ.get("DISPLAY_SPEAK_HOLD_S", "0.6")),
display_speak_rise_db=float(os.environ.get("DISPLAY_SPEAK_RISE_DB", "3")),
display_speak_corr=float(os.environ.get("DISPLAY_SPEAK_CORR", "0.4")),
display_speak_confirm=int(os.environ.get("DISPLAY_SPEAK_CONFIRM", "2")),
portrait_blur=_parse_bool(os.environ.get("PORTRAIT_BLUR", ""), default=True),
portrait_shm_dir=os.environ.get(
"PORTRAIT_SHM_DIR", "/run/livekit-cameras/portrait"
).strip() or "/run/livekit-cameras/portrait",
portrait_hz=float(os.environ.get("PORTRAIT_HZ", "8")),
portrait_blur_px=int(os.environ.get("PORTRAIT_BLUR_PX", "7")),
portrait_hold_s=float(os.environ.get("PORTRAIT_HOLD_S", "0.8")),
)
if cfg.enhance_mode not in ("auto", "force", "off"):
@@ -158,4 +276,45 @@ def load_config(env_path: Optional[Path] = None) -> Config:
raise ValueError("TORCH_NUM_THREADS must be >= 1")
if cfg.beamform_tdoa_every < 1:
raise ValueError("BEAMFORM_TDOA_EVERY must be >= 1")
if cfg.audio_queue_ms is not None and cfg.audio_queue_ms < 1:
raise ValueError("AUDIO_QUEUE_MS must be >= 1")
if cfg.audio_gate_hold_s < 0:
raise ValueError("AUDIO_GATE_HOLD_S must be >= 0")
if cfg.video_hold_s is not None and cfg.video_hold_s < 0:
raise ValueError("VIDEO_HOLD_S must be >= 0")
if cfg.display_video_capacity < 0:
raise ValueError("DISPLAY_VIDEO_CAPACITY must be >= 0")
if cfg.display_video_format not in ("", "bgra", "i420", "yuv420p"):
raise ValueError(
f"DISPLAY_VIDEO_FORMAT must be empty, bgra or i420, got {cfg.display_video_format}")
if cfg.display_gst_queue_buffers < 1:
raise ValueError("DISPLAY_GST_QUEUE_BUFFERS must be >= 1")
if cfg.display_speaker_push not in ("timer", "arrival"):
raise ValueError(
f"DISPLAY_SPEAKER_PUSH must be timer|arrival, got {cfg.display_speaker_push}")
if cfg.display_pump_workers < 1:
raise ValueError("DISPLAY_PUMP_WORKERS must be >= 1")
if cfg.display_hand_raise_hz <= 0:
raise ValueError("DISPLAY_HAND_RAISE_HZ must be > 0")
if cfg.display_hand_raise_hold_s < 0:
raise ValueError("DISPLAY_HAND_RAISE_HOLD_S must be >= 0")
if cfg.display_hand_raise_model not in ("lightning", "thunder"):
raise ValueError(
f"DISPLAY_HAND_RAISE_MODEL must be lightning|thunder, got {cfg.display_hand_raise_model}")
if cfg.display_speak_margin_db < 0:
raise ValueError("DISPLAY_SPEAK_MARGIN_DB must be >= 0")
if cfg.display_speak_hold_s < 0:
raise ValueError("DISPLAY_SPEAK_HOLD_S must be >= 0")
if cfg.display_speak_rise_db < 0:
raise ValueError("DISPLAY_SPEAK_RISE_DB must be >= 0")
if not 0.0 <= cfg.display_speak_corr <= 1.0:
raise ValueError("DISPLAY_SPEAK_CORR must be in 0..1")
if cfg.display_speak_confirm < 1:
raise ValueError("DISPLAY_SPEAK_CONFIRM must be >= 1")
if cfg.portrait_hz <= 0:
raise ValueError("PORTRAIT_HZ must be > 0")
if cfg.portrait_blur_px < 1:
raise ValueError("PORTRAIT_BLUR_PX must be >= 1")
if cfg.portrait_hold_s < 0:
raise ValueError("PORTRAIT_HOLD_S must be >= 0")
return cfg
+10
View File
@@ -0,0 +1,10 @@
# Logitech Rally / Rally Bar as the speaker camera (not a cam-NN C920 slot).
# Product ids: Rally Camera 0881, Rally Bar Huddle 087c, Rally Bar 089b,
# Rally Bar Mini 08d3. Install when the device is on site:
# cp deploy/99-rally-pin.rules /etc/udev/rules.d/
# udevadm control --reload-rules && udevadm trigger
#
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="0881", ATTR{index}=="0", SYMLINK+="rally"
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="087c", ATTR{index}=="0", SYMLINK+="rally"
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="089b", ATTR{index}=="0", SYMLINK+="rally"
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="08d3", ATTR{index}=="0", SYMLINK+="rally"
+2 -4
View File
@@ -1,11 +1,9 @@
[Unit]
Description=Publish C920 cameras (speechbrain audio) to LiveKit
Documentation=file:///home/fdenkena/livekit-cameras/README.md
After=network-online.target sound.target livekit-server.service
After=network-online.target sound.target
Wants=network-online.target
# Wants, not Requires: local livekit-server is a test stand-in.
# Point LIVEKIT_URL at a remote server later and you can disable this unit.
Wants=livekit-server.service
# Local livekit-server is disabled; publishers use LIVEKIT_URL in .env.
[Service]
Type=simple
+5 -3
View File
@@ -1,8 +1,10 @@
[Unit]
Description=Drive 3 DRM displays with LiveKit camera output (GStreamer kmssink)
Documentation=file:///home/fdenkena/livekit-cameras/README.md
After=network-online.target livekit-server.service livekit-cameras.service
Wants=network-online.target livekit-cameras.service
After=network-online.target
Wants=network-online.target
# Do not After=/Wants= livekit-cameras: the wall must stay up when cameras
# are missing and when the SFU is down (on-screen error, then reconnect).
[Service]
Type=simple
@@ -10,7 +12,7 @@ WorkingDirectory=/home/fdenkena/livekit-cameras
EnvironmentFile=-/home/fdenkena/livekit-cameras/.env
Environment=PYTHONUNBUFFERED=1
# Exclusive KMS access — no X/Wayland on these connectors.
ExecStartPre=/home/fdenkena/livekit-cameras/deploy/wait-for-livekit.sh
# Start kmssink immediately; run_displays.py shows LiveKit errors on the wall.
ExecStart=/home/fdenkena/livekit-cameras/.venv/bin/python /home/fdenkena/livekit-cameras/run_displays.py
ExecStop=/bin/kill -TERM $MAINPID
Restart=always
+4 -1
View File
@@ -2,6 +2,9 @@
# Wait until LIVEKIT_URL (from .env) accepts TCP. Used so publishers do not
# race the local test livekit-server on boot. When LIVEKIT_URL points at a
# remote host this waits on that host instead.
#
# bash /dev/tcp to an unreachable host can hang until the kernel SYN timeout
# (longer than systemd TimeoutStartSec). Cap each probe at 1s.
set -u
ENV_FILE=/home/fdenkena/livekit-cameras/.env
if [ -f "$ENV_FILE" ]; then
@@ -29,7 +32,7 @@ if [ "$host" = "localhost" ]; then
fi
echo "wait-for-livekit: $host:$port"
for _ in $(seq 1 30); do
if (echo >/dev/tcp/"$host"/"$port") 2>/dev/null; then
if timeout 1 bash -c "echo >/dev/tcp/${host}/${port}" 2>/dev/null; then
echo "wait-for-livekit: ready"
exit 0
fi
+20
View File
@@ -18,6 +18,13 @@ import subprocess
from dataclasses import dataclass, field
from pathlib import Path
from usb_ids import (
is_fleet_video,
is_rally_pid,
looks_like_rally_name,
usb_vid_pid_of_video_node,
)
log = logging.getLogger("cameras.discovery")
_SLOTS_FILE = Path(__file__).parent / "slots.json"
@@ -32,6 +39,9 @@ class Camera:
usb_port: str = ""
serial: str = ""
extra_video_devices: list[str] = field(default_factory=list)
# None = use fleet VIDEO_WIDTH / VIDEO_HEIGHT (C920 360p). Rally is 720p.
width: int | None = None
height: int | None = None
def select_cameras(cameras: list[Camera], spec: list[str]) -> list[Camera]:
@@ -192,8 +202,18 @@ def discover_cameras() -> list[Camera]:
for name, nodes in groups.items():
if not nodes:
continue
if looks_like_rally_name(name):
log.info("skipping Rally/speaker device %r (not a cam-NN slot)", name)
continue
nodes.sort(key=lambda n: int(Path(n).name.replace("video", "")))
primary = nodes[0]
ids = usb_vid_pid_of_video_node(primary)
if ids is not None and not is_fleet_video(ids[0], ids[1]):
log.info("skipping non-C920 USB %04x:%04x at %s", ids[0], ids[1], primary)
continue
if ids is not None and is_rally_pid(ids[1]):
log.info("skipping Rally USB %04x:%04x at %s", ids[0], ids[1], primary)
continue
serial = _serial_of_video_node(primary)
port = _usb_port_of_video_node(primary)
+866 -7
View File
@@ -1,11 +1,302 @@
"""20-camera mosaic + framed tiles with name placeholders for the KMS wall."""
from __future__ import annotations
import math
import threading
from typing import Iterable, Optional
import cv2
import numpy as np
# Compact mosaic for n live tiles (cols, rows). Landscape-first, max 5x4.
_GRID_SHAPES = {
0: (1, 1),
1: (1, 1),
2: (2, 1),
3: (2, 2),
4: (2, 2),
5: (3, 2),
6: (3, 2),
7: (4, 2),
8: (4, 2),
9: (3, 3),
10: (4, 3),
11: (4, 3),
12: (4, 3),
13: (5, 3),
14: (5, 3),
15: (5, 3),
16: (4, 4),
17: (5, 4),
18: (5, 4),
19: (5, 4),
20: (5, 4),
}
def grid_shape(n: int, max_cols: int = 5, max_rows: int = 4) -> tuple[int, int]:
"""Cols/rows for `n` live tiles. Empty room still occupies 1x1."""
cap = max(1, int(max_cols) * int(max_rows))
n = max(0, min(int(n), cap))
if max_cols == 5 and max_rows == 4 and n in _GRID_SHAPES:
return _GRID_SHAPES[n]
if n <= 1:
return 1, 1
cols = min(int(max_cols), n)
rows = min(int(max_rows), math.ceil(n / cols))
while cols * rows < n and (cols < max_cols or rows < max_rows):
if cols < max_cols:
cols += 1
else:
rows += 1
return cols, rows
def describe_livekit_error(exc: BaseException) -> str:
"""Short on-screen reason for a LiveKit connect/session failure."""
if isinstance(exc, TimeoutError):
return "connection timed out"
if isinstance(exc, ConnectionRefusedError):
return "connection refused"
raw = str(exc) or type(exc).__name__
low = raw.lower()
if "113" in raw or "no route" in low or "unreachable" in low:
return "host unreachable"
if "111" in raw or "refused" in low:
return "connection refused"
if "timed out" in low or "timeout" in low:
return "connection timed out"
if "name or service not known" in low or "nodename nor servname" in low:
return "DNS lookup failed"
if "connection reset" in low or "broken pipe" in low:
return "connection reset"
if "websocket" in low or "ws://" in low or "wss://" in low:
return raw[:96]
return raw[:96]
def format_livekit_status(
*,
url: str,
state: str,
detail: str = "",
retry_s: float | None = None,
room: str = "",
) -> list[str]:
"""Lines for the KMS status overlay. Empty list means show the mosaic."""
url = (url or "").strip()
st = (state or "").strip().lower()
room_line = f"room {room.strip()}" if (room or "").strip() else ""
if st in ("connected", "ok", ""):
return []
if st == "connecting":
lines = ["CONNECTING TO LIVEKIT"]
if url:
lines.append(url)
if room_line:
lines.append(room_line)
return lines
lines = ["LIVEKIT DISCONNECTED"]
if url:
lines.append(url)
if room_line:
lines.append(room_line)
if detail:
lines.append(str(detail)[:96])
if retry_s is not None:
lines.append(f"retrying in {max(0, int(round(float(retry_s))))}s")
return lines
# Status-screen palettes are BGRA. Amber = offline, cyan = connecting.
_STATUS_AMBER = (36, 158, 255, 255)
_STATUS_CYAN = (255, 186, 64, 255)
_STATUS_STEEL = (196, 148, 92, 255)
_STATUS_INK = (18, 14, 12, 255)
_STATUS_PAPER = (245, 242, 238, 255)
_STATUS_MUTED = (168, 158, 150, 255)
def _status_kind(title: str) -> str:
t = (title or "").upper()
if "CONNECTING" in t:
return "connecting"
if "NO LIVE" in t or "NO CAMERA" in t:
return "empty"
return "disconnected"
def _round_rect(img: np.ndarray, x: int, y: int, w: int, h: int, r: int,
color, thickness: int = -1) -> None:
r = max(0, min(int(r), w // 2, h // 2))
x, y, w, h = int(x), int(y), int(w), int(h)
if r <= 0:
cv2.rectangle(img, (x, y), (x + w, y + h), color, thickness)
return
if thickness < 0:
cv2.rectangle(img, (x + r, y), (x + w - r, y + h), color, -1)
cv2.rectangle(img, (x, y + r), (x + w, y + h - r), color, -1)
cv2.circle(img, (x + r, y + r), r, color, -1)
cv2.circle(img, (x + w - r, y + r), r, color, -1)
cv2.circle(img, (x + r, y + h - r), r, color, -1)
cv2.circle(img, (x + w - r, y + h - r), r, color, -1)
return
cv2.ellipse(img, (x + r, y + r), (r, r), 180, 0, 90, color, thickness, cv2.LINE_AA)
cv2.ellipse(img, (x + w - r, y + r), (r, r), 270, 0, 90, color, thickness, cv2.LINE_AA)
cv2.ellipse(img, (x + r, y + h - r), (r, r), 90, 0, 90, color, thickness, cv2.LINE_AA)
cv2.ellipse(img, (x + w - r, y + h - r), (r, r), 0, 0, 90, color, thickness, cv2.LINE_AA)
cv2.line(img, (x + r, y), (x + w - r, y), color, thickness, cv2.LINE_AA)
cv2.line(img, (x + r, y + h), (x + w - r, y + h), color, thickness, cv2.LINE_AA)
cv2.line(img, (x, y + r), (x, y + h - r), color, thickness, cv2.LINE_AA)
cv2.line(img, (x + w, y + r), (x + w, y + h - r), color, thickness, cv2.LINE_AA)
def _text_size(text: str, scale: float, thick: int) -> tuple[int, int]:
(tw, th), _ = cv2.getTextSize(text, cv2.FONT_HERSHEY_SIMPLEX, scale, thick)
return tw, th
def _draw_text(img: np.ndarray, text: str, x: int, y: int, scale: float,
color, thick: int = 1) -> None:
cv2.putText(img, text, (int(x), int(y)), cv2.FONT_HERSHEY_SIMPLEX,
scale, color, thick, cv2.LINE_AA)
def _draw_signal_mark(img: np.ndarray, cx: int, cy: int, accent, *, ok: bool) -> None:
"""Wifi-style arcs; a slash when the SFU is down."""
for rad in (18, 30, 42):
cv2.ellipse(img, (cx, cy + 10), (rad, rad), 0, 225, 315, accent, 2, cv2.LINE_AA)
cv2.circle(img, (cx, cy + 18), 4, accent, -1, cv2.LINE_AA)
if not ok:
cv2.line(img, (cx - 28, cy - 22), (cx + 28, cy + 28), (40, 40, 220, 255), 3, cv2.LINE_AA)
def status_screen(width: int, height: int, lines: list[str] | None) -> np.ndarray:
"""Full-frame KMS status: gradient, ghost grid, accent rail, info card."""
w, h = max(64, int(width)), max(64, int(height))
raw = [str(x).strip() for x in (lines or []) if str(x).strip()]
title = raw[0] if raw else "NO LIVE CAMERAS"
url = next((ln for ln in raw[1:] if ln.startswith(("ws://", "wss://", "http://"))), "")
retry = next((ln for ln in raw[1:] if "retry" in ln.lower()), "")
room = next((ln for ln in raw[1:] if ln.lower().startswith("room ")), "")
detail = next(
(ln for ln in raw[1:] if ln not in (url, retry, room) and ln != title),
"",
)
kind = _status_kind(title)
accent = {"connecting": _STATUS_CYAN, "empty": _STATUS_STEEL}.get(kind, _STATUS_AMBER)
badge = {"connecting": "CONNECTING", "empty": "STANDBY"}.get(kind, "OFFLINE")
if kind == "connecting" and not retry:
retry = "waiting for signal"
yy = np.linspace(0.0, 1.0, h, dtype=np.float32)[:, None]
if kind == "connecting":
top = np.array([28, 22, 16], dtype=np.float32)
bot = np.array([42, 28, 14], dtype=np.float32)
elif kind == "empty":
top = np.array([24, 20, 18], dtype=np.float32)
bot = np.array([32, 26, 22], dtype=np.float32)
else:
top = np.array([22, 16, 12], dtype=np.float32)
bot = np.array([16, 14, 20], dtype=np.float32)
img = np.empty((h, w, 4), dtype=np.uint8)
rgb = top * (1.0 - yy) + bot * yy
img[:, :, :3] = rgb.astype(np.uint8)[:, None, :]
img[:, :, 3] = 255
# Ghost 5x4 mosaic — the wall that should be here.
cols, rows, gap = 5, 4, max(4, w // 240)
gx0, gy0 = int(w * 0.08), int(h * 0.12)
gx1, gy1 = int(w * 0.92), int(h * 0.88)
tw = max(8, (gx1 - gx0 - gap * (cols + 1)) // cols)
th = max(8, (gy1 - gy0 - gap * (rows + 1)) // rows)
ghost = (58, 48, 42, 255) if kind != "connecting" else (48, 50, 42, 255)
for r in range(rows):
for c in range(cols):
x = gx0 + gap + c * (tw + gap)
y = gy0 + gap + r * (th + gap)
cv2.rectangle(img, (x, y), (x + tw, y + th), ghost, 1, cv2.LINE_AA)
rail = max(6, w // 240)
img[:, :rail] = accent
s = max(0.55, min(w / 1920.0, h / 1080.0))
pad = int(48 * s)
foot_h = int(58 * s) if retry else 0
body = int(188 * s)
if url:
body += int(72 * s)
if room:
body += int(72 * s)
if detail:
body += int(80 * s)
body += foot_h + int(28 * s)
cw = min(int(w * 0.62), 1080)
ch = int(min(max(body, int(h * 0.34)), int(h * 0.56)))
cx0 = (w - cw) // 2
cy0 = (h - ch) // 2
card = (32, 26, 24, 255) if kind != "connecting" else (36, 30, 22, 255)
_round_rect(img, cx0, cy0, cw, ch, 18, card, -1)
_round_rect(img, cx0, cy0, cw, ch, 18, accent, 2)
img[cy0:cy0 + 6, cx0 + 18:cx0 + cw - 18] = accent
x = cx0 + pad
y = cy0 + int(56 * s)
pill_w = int(188 * s)
_round_rect(img, x, y - int(28 * s), pill_w, int(36 * s), 8, accent, -1)
bw, bh = _text_size(badge, 0.52 * s, 1)
_draw_text(img, badge, x + (pill_w - bw) // 2, y - int(28 * s) + bh + int(8 * s),
0.52 * s, _STATUS_INK, 1)
_draw_signal_mark(img, cx0 + cw - pad - 20, y - 4, accent, ok=(kind == "connecting"))
y += int(70 * s)
tscale = 1.35 * s
tw, th = _text_size(title, tscale, 2)
while tw > cw - 2 * pad and tscale > 0.6:
tscale *= 0.92
tw, th = _text_size(title, tscale, 2)
_draw_text(img, title, x, y + th, tscale, _STATUS_PAPER, 2)
y += th + int(22 * s)
cv2.line(img, (x, y), (cx0 + cw - pad, y), (64, 54, 48, 255), 1, cv2.LINE_AA)
y += int(36 * s)
if url:
_draw_text(img, "ENDPOINT", x, y, 0.48 * s, _STATUS_MUTED, 1)
y += int(28 * s)
uscale = 0.78 * s
uw, uh = _text_size(url, uscale, 1)
while uw > cw - 2 * pad and uscale > 0.4:
uscale *= 0.92
uw, uh = _text_size(url, uscale, 1)
_draw_text(img, url, x, y + uh, uscale, _STATUS_PAPER, 1)
y += uh + int(28 * s)
if room:
_draw_text(img, "ROOM", x, y, 0.48 * s, _STATUS_MUTED, 1)
y += int(28 * s)
rscale = 0.9 * s
rw, rh = _text_size(room, rscale, 2)
while rw > cw - 2 * pad and rscale > 0.45:
rscale *= 0.92
rw, rh = _text_size(room, rscale, 2)
_draw_text(img, room, x, y + rh, rscale, _STATUS_PAPER, 2)
y += rh + int(28 * s)
if detail:
_draw_text(img, "REASON", x, y, 0.48 * s, _STATUS_MUTED, 1)
y += int(28 * s)
_draw_text(img, detail[:72], x, y + int(22 * s), 0.78 * s, accent, 1)
if retry:
fy = cy0 + ch - foot_h
band = (26, 20, 16, 255) if kind != "connecting" else (28, 24, 16, 255)
img[fy:cy0 + ch - 8, cx0 + 10:cx0 + cw - 10] = band
_draw_text(img, retry.upper(), x, fy + int(38 * s), 0.62 * s, accent, 1)
corner = room.upper() if room else "CAMERA WALL"
_draw_text(img, corner, int(w * 0.08), int(h * 0.07), 0.48 * s, (140, 132, 126, 255), 1)
return img
ROLES = ("grid", "speaker", "screenshare")
# BGRA
@@ -14,6 +305,8 @@ CELL_BG = (18, 18, 18, 255)
EMPTY_FILL = (36, 36, 36, 255)
FRAME_COLOR = (196, 196, 196, 255)
FRAME_LIVE = (168, 168, 168, 255)
FRAME_RAISE = (0, 220, 255, 255) # BGRA yellow — hand raised
FRAME_SPEAK = (0, 210, 40, 255) # BGRA green — LiveKit active speaker
LABEL_BG = (12, 12, 12, 255)
LABEL_FG = (245, 245, 245, 255)
NAME_FG = (220, 220, 220, 255)
@@ -29,6 +322,193 @@ def is_camera_identity(identity: str, prefix: str = "cam") -> bool:
return ident.startswith(p + "-") or ident.startswith(p)
def active_speaker_ids(speakers, *, want=None) -> list[str]:
"""LiveKit active_speakers_changed payload -> identity list."""
out: list[str] = []
for p in speakers:
ident = (getattr(p, "identity", None) or "").strip()
if not ident:
continue
if want is not None and not want(p):
continue
out.append(ident)
return out
def primary_grid_speaker(ids: Iterable[str], identities: Iterable[str]) -> list[str]:
"""Loudest LiveKit speaker that occupies a grid tile (skip rally)."""
tiles = set(identities)
for ident in ids:
if ident in tiles:
return [ident]
return []
def select_grid_speakers(
ranked: Iterable[tuple[str, float]],
*,
min_db: float = -40.0,
margin_db: float = 6.0,
floors: dict[str, float] | None = None,
rise_db: float = 6.0,
) -> list[str]:
"""Keep real talkers; drop idle floor and quieter bleed of the same source.
Speech is ``rise_db`` above that mic's noise floor (not a global dBFS
cut). Two adjacent people stay within ``margin_db`` of each other.
"""
rise = max(0.0, float(rise_db))
live: list[tuple[str, float]] = []
for ident, raw in ranked:
if not ident:
continue
db = float(raw)
if floors is not None:
fl = floors.get(ident)
if fl is None or db < float(fl) + rise:
continue
elif db < float(min_db):
continue
live.append((ident, db))
if not live:
return []
peak = max(db for _, db in live)
margin = max(0.0, float(margin_db))
return [i for i, db in live if (peak - db) <= margin]
def update_noise_floor(
floor: float,
db: float,
*,
speaking: bool = False,
rise_db: float = 4.0,
alpha: float = 0.2,
) -> float:
"""Follow typical idle; freeze while this hop looks like speech."""
del speaking
fl = float(floor)
x = float(db)
a = min(1.0, max(0.0, float(alpha)))
rise = max(0.0, float(rise_db))
if x >= fl + rise:
return fl
return (1.0 - a) * fl + a * x
def confirm_speakers(
counts: dict[str, int],
picked: Iterable[str],
*,
need: int = 2,
already: Iterable[str] | None = None,
) -> tuple[list[str], dict[str, int]]:
"""Drop one-hop idle spikes; already-on seats stay without re-arming."""
picked_l = [i for i in picked if i]
already_s = set(already or ())
n_need = max(1, int(need))
new_counts: dict[str, int] = {}
out: list[str] = []
for ident in picked_l:
n = int(counts.get(ident, 0)) + 1
new_counts[ident] = n
if n >= n_need or ident in already_s:
out.append(ident)
return out, new_counts
def waveform_similarity(
a: np.ndarray,
b: np.ndarray,
max_lag: int = 320,
) -> float:
"""Peak |normalized xcorr| over ±max_lag samples (bleed vs two talkers)."""
x = np.asarray(a, dtype=np.float64).reshape(-1)
y = np.asarray(b, dtype=np.float64).reshape(-1)
n = min(x.size, y.size)
if n < 32:
return 0.0
x = x[:n] - x[:n].mean()
y = y[:n] - y[:n].mean()
nx = float(np.linalg.norm(x))
ny = float(np.linalg.norm(y))
if nx < 1e-9 or ny < 1e-9:
return 0.0
lag = max(0, min(int(max_lag), n - 1))
corr = np.correlate(x, y, mode="full")
mid = n - 1
sl = corr[mid - lag: mid + lag + 1]
return float(np.max(np.abs(sl)) / (nx * ny))
def collapse_bleed(
ids: Iterable[str],
waves: dict[str, np.ndarray],
levels: dict[str, float],
*,
corr_min: float = 0.4,
max_lag: int = 320,
) -> list[str]:
"""If two hops are the same waveform, keep the louder mic only."""
ordered = [i for i in ids if i]
if len(ordered) < 2:
return ordered
keep = set(ordered)
thresh = float(corr_min)
for i, a in enumerate(ordered):
if a not in keep:
continue
wa = waves.get(a)
if wa is None:
continue
for b in ordered[i + 1:]:
if b not in keep:
continue
wb = waves.get(b)
if wb is None:
continue
if waveform_similarity(wa, wb, max_lag=max_lag) < thresh:
continue
if float(levels.get(a, -120.0)) >= float(levels.get(b, -120.0)):
keep.discard(b)
else:
keep.discard(a)
break
return [i for i in ordered if i in keep]
def hold_speakers(
prev: list[str],
new: list[str],
*,
prev_peak: float,
new_peak: float,
now: float,
hold_until: float,
hold_s: float = 0.6,
switch_db: float = 3.0,
) -> tuple[list[str], float]:
"""Stick chrome across SFU jitter; never block a second real talker."""
prev_l = list(prev)
new_l = list(new)
hold = max(0.0, float(hold_s))
if new_l == prev_l:
if new_l:
return new_l, max(hold_until, now + hold)
return [], hold_until
if not prev_l:
return new_l, now + hold
if set(prev_l) <= set(new_l):
return new_l, now + hold
if not new_l:
if now < hold_until:
return prev_l, hold_until
return [], hold_until
if new_peak >= prev_peak + float(switch_db) or now >= hold_until:
return new_l, now + hold
return prev_l, hold_until
def display_name(identity: str) -> str:
"""Human label for a camera identity (cam-07 -> CAM-07)."""
return (identity or "").strip().upper() or "CAMERA"
@@ -56,6 +536,75 @@ def letterbox_bgra(src: np.ndarray, tw: int, th: int) -> np.ndarray:
return out
def i420_nbytes(width: int, height: int) -> int:
return int(width) * int(height) * 3 // 2
def _even(n: int) -> int:
n = int(n)
if n <= 0:
return 0
return n if n % 2 == 0 else n - 1
def letterbox_i420(src: bytes | memoryview, sw: int, sh: int, tw: int, th: int) -> bytes:
"""Scale packed I420 into tw x th, even UV, studio-black bars (Y=16).
Y=0 is below-black and sparkles on HDMI (kmssink limited range).
"""
tw, th = max(2, _even(tw)), max(2, _even(th))
sw, sh = int(sw), int(sh)
need = i420_nbytes(sw, sh)
raw = bytes(src) if not isinstance(src, (bytes, bytearray)) else src
if len(raw) < need or sw < 2 or sh < 2:
raise ValueError("bad I420 source")
arr = np.frombuffer(raw, dtype=np.uint8, count=need)
ysz = sw * sh
uvw, uvh = sw // 2, sh // 2
uvsz = uvw * uvh
y = arr[:ysz].reshape(sh, sw)
u = arr[ysz:ysz + uvsz].reshape(uvh, uvw)
v = arr[ysz + uvsz:ysz + 2 * uvsz].reshape(uvh, uvw)
scale = min(tw / sw, th / sh)
nw, nh = _even(round(sw * scale)), _even(round(sh * scale))
y2 = cv2.resize(y, (nw, nh), interpolation=cv2.INTER_AREA)
u2 = cv2.resize(u, (nw // 2, nh // 2), interpolation=cv2.INTER_AREA)
v2 = cv2.resize(v, (nw // 2, nh // 2), interpolation=cv2.INTER_AREA)
out = np.empty(i420_nbytes(tw, th), dtype=np.uint8)
oy = tw * th
ouv = (tw // 2) * (th // 2)
out[:oy] = 16
out[oy:] = 128
yo = out[:oy].reshape(th, tw)
uo = out[oy:oy + ouv].reshape(th // 2, tw // 2)
vo = out[oy + ouv:].reshape(th // 2, tw // 2)
x = _even((tw - nw) // 2)
y0 = _even((th - nh) // 2)
yo[y0:y0 + nh, x:x + nw] = y2
uo[y0 // 2:y0 // 2 + nh // 2, x // 2:x // 2 + nw // 2] = u2
vo[y0 // 2:y0 // 2 + nh // 2, x // 2:x // 2 + nw // 2] = v2
return out.tobytes()
def bgra_to_i420(bgra: np.ndarray) -> bytes:
"""Packed I420 bytes (h*w*3//2) from HxWx4 BGRA."""
if bgra.ndim != 3 or bgra.shape[2] < 3:
raise ValueError(f"expected HxWx3/4, got {bgra.shape}")
bgr = bgra[:, :, :3]
yuv = cv2.cvtColor(bgr, cv2.COLOR_BGR2YUV_I420)
return np.ascontiguousarray(yuv).tobytes()
def bgra_to_yuv_pixel(bgra: tuple[int, int, int, int] | tuple[int, int, int]) -> tuple[int, int, int]:
"""Sample Y,U,V of a solid BGRA color via OpenCV I420."""
img = np.empty((2, 2, 4), dtype=np.uint8)
b, g, r = int(bgra[0]), int(bgra[1]), int(bgra[2])
a = int(bgra[3]) if len(bgra) > 3 else 255
img[:] = (b, g, r, a)
arr = np.frombuffer(bgra_to_i420(img), dtype=np.uint8)
return int(arr[0]), int(arr[4]), int(arr[5])
def bgra_from_livekit(data: memoryview | bytes, width: int, height: int) -> np.ndarray:
arr = np.frombuffer(data, dtype=np.uint8)
return arr.reshape((height, width, 4)).copy()
@@ -123,33 +672,138 @@ class CameraGrid:
gap: int = 8,
border: int = 3,
label_h: int = 28,
chrome_out: int = 4,
) -> None:
if cols < 1 or rows < 1:
raise ValueError("cols/rows must be >= 1")
self.width = int(width)
self.height = int(height)
self.max_cols = int(cols)
self.max_rows = int(rows)
self.cols = int(cols)
self.rows = int(rows)
self.gap = max(0, int(gap))
self.border = max(1, int(border))
# Extra speak/raise thickness in the gutter only (even, for I420 UV).
out = min(self.gap, max(0, int(chrome_out)))
self.chrome_out = out if out % 2 == 0 else out - 1
self.label_h = max(16, int(label_h))
self.identities = list(identities) if identities is not None else camera_ids(cols * rows)
self._status_lines: list[str] | None = None
self._live: set[str] = set()
self._raised: set[str] = set()
self._speaking: set[str] = set()
self._tiles: dict[str, np.ndarray] = {}
self._lock = threading.Lock()
self._yuv_raise = bgra_to_yuv_pixel(FRAME_RAISE)
self._yuv_speak = bgra_to_yuv_pixel(FRAME_SPEAK)
self._canvas = np.zeros((self.height, self.width, 4), dtype=np.uint8)
self._apply_layout(self.identities, cols=self.cols, rows=self.rows)
def _layout_metrics(self, cols: int, rows: int) -> None:
cols = max(1, int(cols))
rows = max(1, int(rows))
self.cols = cols
self.rows = rows
self.tile_w = max(32, (self.width - self.gap * (self.cols + 1)) // self.cols)
self.tile_h = max(32, (self.height - self.gap * (self.rows + 1)) // self.rows)
self.inner_w = max(1, self.tile_w - 2 * self.border)
self.inner_h = max(1, self.tile_h - 2 * self.border - self.label_h)
self._live: set[str] = set()
self._tiles: dict[str, np.ndarray] = {}
for ident in self.identities:
self._tiles[ident] = self._compose_cell(ident, None)
self._canvas = np.zeros((self.height, self.width, 4), dtype=np.uint8)
def _rebuild_chrome(self) -> None:
chrome = np.frombuffer(self._render_mosaic(), dtype=np.uint8).reshape(
self.height, self.width, 4)
self._i420 = np.frombuffer(bgra_to_i420(chrome), dtype=np.uint8).copy()
self._i420_chrome = self._i420.copy()
def _apply_layout(
self,
identities: Iterable[str],
*,
cols: int | None = None,
rows: int | None = None,
) -> None:
ids = [i for i in identities if i]
if cols is None or rows is None:
cols, rows = grid_shape(len(ids), self.max_cols, self.max_rows)
self.identities = ids
keep = set(ids)
self._live &= keep
self._raised &= keep
self._speaking &= keep
self._layout_metrics(cols, rows)
self._tiles = {ident: self._compose_cell(ident, None) for ident in ids}
if self._canvas.shape != (self.height, self.width, 4):
self._canvas = np.zeros((self.height, self.width, 4), dtype=np.uint8)
self._rebuild_chrome()
def set_identities(self, identities: Iterable[str]) -> None:
"""Replace the tile set and shrink/grow cols x rows to match."""
ids = [i for i in identities if i]
cols, rows = grid_shape(len(ids), self.max_cols, self.max_rows)
with self._lock:
if ids == self.identities and cols == self.cols and rows == self.rows:
return
self._apply_layout(ids, cols=cols, rows=rows)
def add_identity(self, identity: str) -> bool:
ident = (identity or "").strip()
if not ident:
return False
if ident in self.identities:
return True
cap = self.max_cols * self.max_rows
if len(self.identities) >= cap:
return False
ids = sorted(self.identities + [ident])
self.set_identities(ids)
return ident in self.identities
def remove_identity(self, identity: str) -> None:
ident = (identity or "").strip()
if ident not in self.identities:
return
self.set_identities([i for i in self.identities if i != ident])
def set_status(self, lines: list[str] | None) -> None:
"""Full-frame overlay (LiveKit errors). None restores the mosaic."""
cleaned = [str(x) for x in lines if str(x).strip()] if lines else None
with self._lock:
self._status_lines = cleaned or None
def _status_frame(self) -> np.ndarray:
lines = self._status_lines or ["NO LIVE CAMERAS"]
return status_screen(self.width, self.height, lines)
def tile_origin(self, index: int) -> tuple[int, int]:
r, c = divmod(int(index), self.cols)
x = self.gap + c * (self.tile_w + self.gap)
y = self.gap + r * (self.tile_h + self.gap)
return x, y
def inner_rect(self, index: int) -> tuple[int, int, int, int]:
"""Inner video rect (x, y, w, h) for tile index, inside the frame."""
x, y = self.tile_origin(index)
return x + self.border, y + self.border, self.inner_w, self.inner_h
def ensure_slot(self, identity: str) -> bool:
"""Bind `identity` to an empty tile so remotes can occupy the grid."""
if identity in self._tiles:
return True
for i, ident in enumerate(self.identities):
if ident not in self._live:
del self._tiles[ident]
self.identities[i] = identity
self._tiles[identity] = self._compose_cell(identity, None)
return True
return False
def _compose_cell(self, identity: str, inner: Optional[np.ndarray]) -> np.ndarray:
tw, th = self.tile_w, self.tile_h
b = self.border
cell = np.empty((th, tw, 4), dtype=np.uint8)
cell[:] = CELL_BG
frame = FRAME_LIVE if inner is not None else FRAME_COLOR
frame = self._frame_color(identity, inner is not None)
cv2.rectangle(cell, (0, 0), (tw - 1, th - 1), frame, b)
# video / placeholder region
iy0, ix0 = b, b
@@ -170,19 +824,218 @@ class CameraGrid:
)
return cell
def _frame_color(self, identity: str, live: bool) -> tuple[int, int, int, int]:
if identity in self._speaking:
return FRAME_SPEAK
if identity in self._raised:
return FRAME_RAISE
return FRAME_LIVE if live else FRAME_COLOR
def highlighted(self, identity: str) -> bool:
return identity in self._raised
def speaking(self, identity: str) -> bool:
return identity in self._speaking
def _paint_border(self, identity: str) -> None:
"""Stamp I420 chrome for the current speak/raise/live winner."""
try:
idx = self.identities.index(identity)
except ValueError:
return
ox, oy = self.tile_origin(idx)
color = self._frame_color(identity, identity in self._live)
if color == FRAME_SPEAK:
yv, uv, vv = self._yuv_speak
self._fill_i420_border(self._i420, ox, oy, yv, uv, vv)
elif color == FRAME_RAISE:
yv, uv, vv = self._yuv_raise
self._fill_i420_border(self._i420, ox, oy, yv, uv, vv)
else:
self._copy_i420_border(self._i420_chrome, self._i420, ox, oy)
cell = self._tiles.get(identity)
if cell is not None:
cv2.rectangle(
cell, (0, 0), (self.tile_w - 1, self.tile_h - 1),
color, self.border)
def set_highlight(self, identity: str, raised: bool) -> None:
"""Yellow tile chrome when `raised`. Live I420 path only blits the inner rect."""
if identity not in self._tiles:
return
want = bool(raised)
with self._lock:
had = identity in self._raised
if want:
self._raised.add(identity)
else:
self._raised.discard(identity)
if had == want:
return
self._paint_border(identity)
def set_speakers(self, identities: Iterable[str]) -> None:
"""Green tile chrome for LiveKit active speakers (grid identities only)."""
want = {i for i in identities if i in self._tiles}
with self._lock:
previous = set(self._speaking)
if want == previous:
return
self._speaking = want
for ident in previous | want:
self._paint_border(ident)
def set_frame(self, identity: str, bgra: np.ndarray) -> None:
if identity not in self._tiles:
return
self._live.add(identity)
self._tiles[identity] = self._compose_cell(identity, bgra)
def _chrome_geom(self, x: int, y: int) -> tuple[int, int, int, int, int]:
"""Outer speak/raise ring: original border plus gutter pad."""
out = self.chrome_out
return x - out, y - out, self.tile_w + 2 * out, self.tile_h + 2 * out, self.border + out
def _fill_i420_border(
self, mosa: np.ndarray, x: int, y: int, yv: int, uv: int, vv: int,
) -> None:
W, H = self.width, self.height
x, y, tw, th, b = self._chrome_geom(x, y)
self._stamp_i420_border(mosa, W, H, x, y, tw, th, b, yv, uv, vv)
def _copy_i420_border(self, src: np.ndarray, dst: np.ndarray, x: int, y: int) -> None:
W, H = self.width, self.height
x, y, tw, th, b = self._chrome_geom(x, y)
x0, y0 = max(0, x), max(0, y)
x1, y1 = min(W, x + tw), min(H, y + th)
if x1 <= x0 or y1 <= y0 or b <= 0:
return
oy = W * H
ouv = (W // 2) * (H // 2)
sy = src[:oy].reshape(H, W)
su = src[oy:oy + ouv].reshape(H // 2, W // 2)
sv = src[oy + ouv:].reshape(H // 2, W // 2)
dy = dst[:oy].reshape(H, W)
du = dst[oy:oy + ouv].reshape(H // 2, W // 2)
dv = dst[oy + ouv:].reshape(H // 2, W // 2)
yt1 = min(y0 + b, y1)
yb0 = max(y1 - b, y0)
xl1 = min(x0 + b, x1)
xr0 = max(x1 - b, x0)
dy[y0:yt1, x0:x1] = sy[y0:yt1, x0:x1]
dy[yb0:y1, x0:x1] = sy[yb0:y1, x0:x1]
dy[y0:y1, x0:xl1] = sy[y0:y1, x0:xl1]
dy[y0:y1, xr0:x1] = sy[y0:y1, xr0:x1]
ux0, uy0 = x0 // 2, y0 // 2
ux1, uy1 = (x1 + 1) // 2, (y1 + 1) // 2
ub = max(1, (b + 1) // 2)
du[uy0:uy0 + ub, ux0:ux1] = su[uy0:uy0 + ub, ux0:ux1]
du[uy1 - ub:uy1, ux0:ux1] = su[uy1 - ub:uy1, ux0:ux1]
du[uy0:uy1, ux0:ux0 + ub] = su[uy0:uy1, ux0:ux0 + ub]
du[uy0:uy1, ux1 - ub:ux1] = su[uy0:uy1, ux1 - ub:ux1]
dv[uy0:uy0 + ub, ux0:ux1] = sv[uy0:uy0 + ub, ux0:ux1]
dv[uy1 - ub:uy1, ux0:ux1] = sv[uy1 - ub:uy1, ux0:ux1]
dv[uy0:uy1, ux0:ux0 + ub] = sv[uy0:uy1, ux0:ux0 + ub]
dv[uy0:uy1, ux1 - ub:ux1] = sv[uy0:uy1, ux1 - ub:ux1]
@staticmethod
def _stamp_i420_border(
mosa: np.ndarray, W: int, H: int, x: int, y: int, tw: int, th: int,
b: int, yv: int, uv: int, vv: int,
) -> None:
x0, y0 = max(0, x), max(0, y)
x1, y1 = min(W, x + tw), min(H, y + th)
if x1 <= x0 or y1 <= y0 or b <= 0:
return
oy = W * H
ouv = (W // 2) * (H // 2)
yp = mosa[:oy].reshape(H, W)
up = mosa[oy:oy + ouv].reshape(H // 2, W // 2)
vp = mosa[oy + ouv:].reshape(H // 2, W // 2)
yt1 = min(y0 + b, y1)
yb0 = max(y1 - b, y0)
xl1 = min(x0 + b, x1)
xr0 = max(x1 - b, x0)
yp[y0:yt1, x0:x1] = yv
yp[yb0:y1, x0:x1] = yv
yp[y0:y1, x0:xl1] = yv
yp[y0:y1, xr0:x1] = yv
ux0, uy0 = x0 // 2, y0 // 2
ux1, uy1 = (x1 + 1) // 2, (y1 + 1) // 2
ub = max(1, (b + 1) // 2)
up[uy0:uy0 + ub, ux0:ux1] = uv
up[uy1 - ub:uy1, ux0:ux1] = uv
up[uy0:uy1, ux0:ux0 + ub] = uv
up[uy0:uy1, ux1 - ub:ux1] = uv
vp[uy0:uy0 + ub, ux0:ux1] = vv
vp[uy1 - ub:uy1, ux0:ux1] = vv
vp[uy0:uy1, ux0:ux0 + ub] = vv
vp[uy0:uy1, ux1 - ub:ux1] = vv
def set_frame_i420(self, identity: str, data: bytes | memoryview,
width: int, height: int) -> None:
try:
idx = self.identities.index(identity)
except ValueError:
return
iw, ih = self.inner_w, self.inner_h
fitted = letterbox_i420(data, width, height, iw, ih)
with self._lock:
self._live.add(identity)
self._blit_i420_inner(idx, fitted)
def _blit_i420_inner(self, index: int, tile: bytes) -> None:
x, y, w, h = self.inner_rect(index)
x, y, w, h = _even(x), _even(y), _even(w), _even(h)
W, H = self.width, self.height
need = i420_nbytes(w, h)
if len(tile) < need:
return
src = np.frombuffer(tile, dtype=np.uint8, count=need)
ysz = w * h
uvsz = (w // 2) * (h // 2)
sy = src[:ysz].reshape(h, w)
su = src[ysz:ysz + uvsz].reshape(h // 2, w // 2)
sv = src[ysz + uvsz:ysz + 2 * uvsz].reshape(h // 2, w // 2)
mosa = self._i420
oy = W * H
ouv = (W // 2) * (H // 2)
mosa[:oy].reshape(H, W)[y:y + h, x:x + w] = sy
mosa[oy:oy + ouv].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = su
mosa[oy + ouv:].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = sv
def clear(self, identity: str) -> None:
if identity not in self._tiles:
return
self._live.discard(identity)
self._raised.discard(identity)
self._speaking.discard(identity)
self._tiles[identity] = self._compose_cell(identity, None)
try:
idx = self.identities.index(identity)
except ValueError:
return
x, y = self.tile_origin(idx)
w, h = self.tile_w, self.tile_h
x, y, w, h = _even(x), _even(y), _even(w), _even(h)
W, H = self.width, self.height
with self._lock:
oy = W * H
ouv = (W // 2) * (H // 2)
chrome, mosa = self._i420_chrome, self._i420
mosa[:oy].reshape(H, W)[y:y + h, x:x + w] = chrome[:oy].reshape(H, W)[y:y + h, x:x + w]
mosa[oy:oy + ouv].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = (
chrome[oy:oy + ouv].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2])
mosa[oy + ouv:].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = (
chrome[oy + ouv:].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2])
def render(self) -> bytes:
def render_i420(self) -> bytes:
with self._lock:
if self._status_lines:
return bgra_to_i420(self._status_frame())
return self._i420.tobytes()
def _render_mosaic(self) -> bytes:
canvas = self._canvas
canvas[:] = CANVAS_BG
g = self.gap
@@ -195,3 +1048,9 @@ class CameraGrid:
tile = self._tiles[ident]
canvas[y : y + self.tile_h, x : x + self.tile_w] = tile
return canvas.tobytes()
def render(self) -> bytes:
with self._lock:
if self._status_lines:
return self._status_frame().tobytes()
return self._render_mosaic()
+56 -22
View File
@@ -189,33 +189,67 @@ def list_outputs() -> list[DrmOutput]:
return outputs
def _rank_wall(o: DrmOutput) -> tuple:
"""Arc HDMI first, then any connected HDMI. Names shuffle across boots."""
return (
0 if o.connected else 1,
0 if o.driver == "xe" else 1,
0 if o.name.startswith("HDMI-A") else 1,
o.name,
)
def pick_outputs(
all_outs: list[DrmOutput],
n: int = 3,
names: Optional[list[str]] = None,
connected_only: bool = False,
) -> list[DrmOutput]:
"""Choose DRM heads. Stale DISPLAY_CONNECTORS names must not bind a dead port
when a connected Arc HDMI is available — the wall has to stay lit."""
if names:
by_name: dict[str, list[DrmOutput]] = {}
for o in all_outs:
by_name.setdefault(o.name, []).append(o)
missing = [name for name in names if name not in by_name]
if missing:
have = sorted({o.name for o in all_outs})
raise KeyError(f"DRM connector {missing[0]!r} not found. Have: {have}")
picked: list[DrmOutput | None] = []
used: set[str] = set()
for name in names:
cands = sorted(by_name[name], key=_rank_wall)
chosen = next((o for o in cands if o.connected and o.sysfs not in used), None)
if chosen is None:
picked.append(None)
else:
used.add(chosen.sysfs)
picked.append(chosen)
fallback = sorted(
(o for o in all_outs if o.connected and o.sysfs not in used),
key=_rank_wall,
)
fi = 0
for i, o in enumerate(picked):
if o is None and fi < len(fallback):
picked[i] = fallback[fi]
used.add(fallback[fi].sysfs)
fi += 1
return [o for o in picked if o is not None]
connected = [o for o in all_outs if o.connected]
rest = [o for o in all_outs if not o.connected]
connected.sort(key=_rank_wall)
rest.sort(key=_rank_wall)
ordered = connected + ([] if connected_only else rest)
return ordered[:n]
def select_outputs(
n: int = 3,
names: Optional[list[str]] = None,
connected_only: bool = False,
) -> list[DrmOutput]:
all_outs = list_outputs()
if names:
by_name = {o.name: o for o in all_outs}
picked = []
for name in names:
if name not in by_name:
raise KeyError(f"DRM connector {name!r} not found. Have: {sorted(by_name)}")
picked.append(by_name[name])
return picked
connected = [o for o in all_outs if o.connected]
rest = [o for o in all_outs if not o.connected]
# Prefer the discrete GPU (xe / Arc) HDMI heads for a 3-display wall.
def rank(o: DrmOutput) -> tuple:
return (
0 if o.driver == "xe" else 1,
0 if o.name.startswith("HDMI-A") else 1,
o.name,
)
connected.sort(key=rank)
rest.sort(key=rank)
ordered = connected + ([] if connected_only else rest)
return ordered[:n]
return pick_outputs(list_outputs(), n=n, names=names, connected_only=connected_only)
def format_outputs(outs: list[DrmOutput]) -> str:
+324
View File
@@ -0,0 +1,324 @@
"""I420 SHM -> iGPU H.264 (or Arc AV1) -> latest-AU mmap. Does not open /dev/camNN."""
from __future__ import annotations
import argparse
import json
import logging
import os
import select
import shutil
import signal
import subprocess
import sys
import time
from pathlib import Path
from annexb import AnnexBSplitter, nal_unit_type
from av1_obu import Av1TuSplitter, is_keyframe as av1_is_keyframe
from frame_shm import LatestFrameReader, frame_bytes, wait_for_ready
from i420_nv12 import coded_dim, i420_to_nv12_into, nv12_bytes
from nal_shm import AU_MAX, NalWriter
from portrait import i420_source_path
log = logging.getLogger("cameras.encode-daemon")
GST = shutil.which("gst-launch-1.0") or "gst-launch-1.0"
READY_NAME = "ready"
GROUP = 5
def _pipes(n: int) -> tuple[list[int], list[int]]:
rs, ws = [], []
for _ in range(n):
r, w = os.pipe()
try:
import fcntl
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 1048576)
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 1048576)
except OSError:
pass
rs.append(r)
ws.append(w)
return rs, ws
def _one_enc(width: int, height: int, fps: int, bitrate_kbps: int,
fd_in: int, fd_out: int, codec: str = "h264",
render: str = "/dev/dri/renderD129", hi_kbps: int = 20000) -> list[str]:
need = frame_bytes(width, height)
kbps = max(100, int(bitrate_kbps))
hi = int(height) >= 720
if hi:
kbps = max(kbps, max(100, int(hi_kbps)))
# GOP=1: a 1s GOP left motion trails until the next IDR. Rally is PTZ.
key_int = 1
usage = 2 if hi else 7
cpb = max(80, kbps // 8)
if codec == "av1":
# HW AV1 is NV12-only (no I420 entrypoint). Convert is CPU I420→NV12
# in this process; gst is parse → vaav1enc only (no vapostproc).
enc = [
"!", "vaav1enc", f"bitrate={kbps}", f"key-int-max={int(fps)}",
"target-usage=7", "hierarchical-level=1", "ref-frames=1",
"gf-group-size=1", "rate-control=cbr",
f"cpb-size={cpb}",
]
fmt = "nv12"
else:
# iGPU H.264. NV12-only. GOP=1 so fdsrc preroll does not hang.
# vah264enc.device-path is read-only; use the per-node factory.
enc_name = f"va{Path(render).name}h264enc"
enc = [
"!", enc_name, f"bitrate={kbps}",
f"key-int-max={key_int}", "b-frames=0", "ref-frames=1",
f"target-usage={usage}", "rate-control=cbr", f"cpb-size={cpb}",
"!", "h264parse", "config-interval=-1",
"!", "video/x-h264,stream-format=byte-stream,alignment=au",
]
fmt = "nv12"
need = nv12_bytes(width, height)
head = [
"fdsrc", f"fd={int(fd_in)}", f"blocksize={need}",
"do-timestamp=true",
"!", "rawvideoparse", f"width={int(width)}", f"height={int(height)}",
f"format={fmt}", f"framerate={int(fps)}/1",
"!", "queue", "max-size-buffers=1", "max-size-bytes=0", "max-size-time=0",
"leaky=downstream",
]
tail = [
"!", "queue", "max-size-buffers=1", "max-size-bytes=0", "max-size-time=0",
"leaky=downstream",
"!", "fdsink", f"fd={int(fd_out)}", "sync=false",
]
return head + enc + tail
def grouped(cams: list[dict], size: int) -> list[list[dict]]:
by: dict[tuple[int, int], list[dict]] = {}
for c in cams:
key = (int(c["width"]), int(c["height"]))
by.setdefault(key, []).append(c)
out: list[list[dict]] = []
for group in by.values():
for i in range(0, len(group), size):
out.append(group[i:i + size])
return out
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--cameras-json", required=True)
ap.add_argument("--i420-dir", default="/run/livekit-cameras/raw")
ap.add_argument("--h264-dir", default="/run/livekit-cameras/h264")
ap.add_argument("--fps", type=int, default=30)
ap.add_argument("--bitrate", type=int, default=2_000_000)
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
ap.add_argument("--group-size", type=int, default=GROUP)
ap.add_argument("--codec", default="h264")
ap.add_argument("--hi-bitrate", type=int, default=20_000_000,
help="bps for height>=720 (Rally). Rollback: 12000000")
ap.add_argument("--portrait-dir", default="",
help="C920 I420 from portrait daemon; rally stays in --i420-dir")
args = ap.parse_args(argv)
codec = (args.codec or "h264").lower()
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [encode-daemon] %(levelname)s %(message)s")
cams = json.loads(args.cameras_json)
if not cams:
log.error("no cameras")
return 2
from vaapi_pin import igpu_render
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
os.environ["GST_VA_ALL_DRIVERS"] = "1"
os.environ.pop("LIBVA_DRM_DEVICE", None)
encode_render = igpu_render()
if codec == "av1":
log.info("encoder=vaav1enc card0=/dev/dri/renderD128 codec=%s", codec)
else:
log.info("encoder=%s codec=%s", f"va{Path(encode_render).name}h264enc", codec)
os.makedirs(args.h264_dir, exist_ok=True)
kbps = max(100, int(args.bitrate) // 1000)
stop = False
def _stop(*_a):
nonlocal stop
stop = True
signal.signal(signal.SIGTERM, _stop)
signal.signal(signal.SIGINT, _stop)
gst_procs: list[subprocess.Popen] = []
states = []
extra_fds: list[int] = []
for group in grouped(cams, 1 if codec == "av1" else max(1, args.group_size)):
n = len(group)
in_r, in_w = _pipes(n)
out_r, out_w = _pipes(n)
for fd in in_r + out_w:
os.set_inheritable(fd, True)
for fd in in_w + out_r:
os.set_inheritable(fd, False)
for cam, w_in in zip(group, in_w):
w, h = int(cam["width"]), int(cam["height"])
if codec in ("av1", "h264"):
need = nv12_bytes(coded_dim(w), coded_dim(h))
else:
need = frame_bytes(w, h)
try:
import fcntl
fcntl.fcntl(w_in, fcntl.F_SETPIPE_SZ, max(1048576, need + 4096))
except OSError:
pass
os.write(w_in, bytes(need))
for fd in in_w + out_r:
os.set_blocking(fd, False)
cmd = [GST, "-q"]
for cam, r, wfd in zip(group, in_r, out_w):
w, h = int(cam["width"]), int(cam["height"])
if codec in ("av1", "h264"):
w, h = coded_dim(w), coded_dim(h)
cmd += _one_enc(w, h, args.fps, kbps, r, wfd, codec,
render=encode_render,
hi_kbps=max(100, int(args.hi_bitrate) // 1000))
env = dict(os.environ)
env["GST_REGISTRY_UPDATE"] = "no"
proc = subprocess.Popen(
cmd, pass_fds=tuple(in_r + out_w), env=env,
stderr=subprocess.STDOUT)
gst_procs.append(proc)
for fd in in_r + out_w:
extra_fds.append(fd)
for cam, w_in, r_out in zip(group, in_w, out_r):
ident = cam["identity"]
w, h = int(cam["width"]), int(cam["height"])
i420 = i420_source_path(ident, args.i420_dir, args.portrait_dir)
h264 = os.path.join(args.h264_dir, f"{ident}.h264")
au_max = 512 * 1024 if codec == "av1" else AU_MAX
i420_n = frame_bytes(w, h)
if codec in ("av1", "h264"):
cw, ch = coded_dim(w), coded_dim(h)
pipe_n = nv12_bytes(cw, ch)
nv12 = bytearray(pipe_n)
else:
pipe_n = i420_n
nv12 = None
states.append({
"ident": ident, "w": w, "h": h,
"need": pipe_n,
"reader": LatestFrameReader(i420),
"writer": NalWriter(h264, max_payload=au_max),
"in_w": w_in, "out_r": r_out,
"scratch": bytearray(i420_n),
"nv12": nv12,
"last": -1,
"split": Av1TuSplitter() if codec == "av1" else AnnexBSplitter(),
"sps": b"", "pps": b"", "woff": 0, "wseq": -1,
"codec": codec,
})
extra_fds.extend([w_in, r_out])
log.info("gst pid=%s cams=%s", proc.pid,
",".join(c["identity"] for c in group))
states.sort(key=lambda s: 0 if s["ident"] == "rally" else 1)
ready = os.path.join(args.h264_dir, READY_NAME)
Path(ready).write_text("ok\n")
out_map = {s["out_r"]: s for s in states}
t0 = time.monotonic()
seqs = {s["ident"]: 0 for s in states}
try:
while not stop:
for s in states:
pipe = s["nv12"] if s.get("nv12") is not None else s["scratch"]
if s["woff"]:
view = memoryview(pipe)
try:
n = os.write(s["in_w"], view[s["woff"]:])
s["woff"] += n
if s["woff"] >= s["need"]:
s["woff"] = 0
s["last"] = s["wseq"]
except BlockingIOError:
pass
except OSError:
pass
continue
seq = s["reader"].copy_into(s["scratch"])
if seq is None or seq == s["last"]:
continue
if s.get("nv12") is not None:
i420_to_nv12_into(s["scratch"], s["w"], s["h"], s["nv12"])
pipe = s["nv12"]
view = memoryview(pipe)
try:
n = os.write(s["in_w"], view)
if n < s["need"]:
s["woff"] = n
s["wseq"] = seq
else:
s["last"] = seq
except BlockingIOError:
s["woff"] = 0
except OSError:
pass
rds, _, _ = select.select(list(out_map), [], [], 0.01)
for fd in rds:
s = out_map[fd]
try:
chunk = os.read(fd, 65536)
except OSError:
continue
if not chunk:
continue
for nal in s["split"].push(chunk):
if s.get("codec") == "av1":
if av1_is_keyframe(nal) or nal:
seqs[s["ident"]] = s["writer"].write(nal)
continue
ntype = nal_unit_type(nal)
if ntype == 7:
s["sps"] = nal
continue
if ntype == 8:
s["pps"] = nal
continue
if ntype == 5:
payload = s["sps"] + s["pps"] + nal
elif ntype in (1, 2, 3, 4):
payload = nal
else:
continue
seqs[s["ident"]] = s["writer"].write(payload)
if time.monotonic() - t0 >= 10:
log.info("seqs %s", " ".join(
f"{k}:{v}" for k, v in sorted(seqs.items())))
t0 = time.monotonic()
for p in gst_procs:
if p.poll() is not None:
log.error("gst exited rc=%s", p.returncode)
stop = True
break
finally:
for p in gst_procs:
if p.poll() is None:
p.terminate()
for fd in extra_fds:
try:
os.close(fd)
except OSError:
pass
try:
os.unlink(ready)
except OSError:
pass
return 0
if __name__ == "__main__":
raise SystemExit(main())
+48
View File
@@ -0,0 +1,48 @@
"""ctypes: push pre-encoded AUs into LiveKit VideoSource (pass-through)."""
from __future__ import annotations
import ctypes
_CODEC = {"h264": 0, "h265": 1, "vp8": 2, "vp9": 3, "av1": 4}
def capture_encoded(
video_source,
payload: bytes,
width: int,
height: int,
keyframe: bool,
timestamp_us: int = 0,
codec: str = "h264",
) -> int:
from livekit.rtc._ffi_client import FfiClient
if not payload:
return -1
lib = FfiClient.instance._ffi_lib
fn = lib.livekit_ffi_capture_encoded
fn.argtypes = [
ctypes.c_uint64,
ctypes.c_void_p,
ctypes.c_size_t,
ctypes.c_int32,
ctypes.c_int32,
ctypes.c_int32,
ctypes.c_int64,
ctypes.c_int32,
]
fn.restype = ctypes.c_int32
handle = int(video_source._ffi_handle.handle)
buf = (ctypes.c_uint8 * len(payload)).from_buffer_copy(payload)
return int(
fn(
handle,
ctypes.cast(buf, ctypes.c_void_p),
len(payload),
int(width),
int(height),
1 if keyframe else 0,
int(timestamp_us),
int(_CODEC.get(codec, 0)),
)
)
+166
View File
@@ -0,0 +1,166 @@
"""Shared MetricGAN daemon: one model, many camera clients."""
from __future__ import annotations
import logging
import os
import socket
import threading
import time
from typing import Callable, Optional
import numpy as np
from enhance_ipc import read_frame, write_frame
log = logging.getLogger("audio.enhance_daemon")
Handler = Callable[[str, np.ndarray], np.ndarray]
class SharedEnhanceEngine:
"""One AudioCleaner model; per-identity context and TDOA."""
def __init__(self, make_cleaner: Callable):
self._template = make_cleaner()
self._sessions: dict[str, object] = {}
self._lock = threading.Lock()
def identities(self) -> list[str]:
return list(self._sessions)
def _session(self, ident: str):
sess = self._sessions.get(ident)
if sess is not None:
return sess
sess = self._clone()
self._sessions[ident] = sess
return sess
def _clone(self):
import copy
sess = copy.copy(self._template)
sess._ctx = np.zeros(0, dtype=np.float32)
sess._last_tdoas = None
sess._tdoa_age = 0
sess._client = None
sess.last_dt_ms = -1.0
compiled = getattr(sess, "_ov_compiled", None)
if compiled is not None:
sess._ov_request = compiled.create_infer_request()
return sess
def process(self, ident: str, pcm: np.ndarray) -> np.ndarray:
with self._lock:
sess = self._session(ident)
return sess.process(pcm)
def serve_unix(path: str, handler: Handler, stop: Optional[threading.Event] = None) -> None:
"""Accept unix-socket clients; handler(ident, pcm) -> int16 mono."""
if os.path.exists(path):
os.unlink(path)
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
srv = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
srv.bind(path)
srv.listen(64)
srv.settimeout(0.2)
def _client(conn: socket.socket) -> None:
try:
while stop is None or not stop.is_set():
ident, pcm = read_frame(conn)
out = handler(ident, pcm)
out = np.asarray(out, dtype=np.int16)
write_frame(conn, ident, out)
except (EOFError, ConnectionError, BrokenPipeError, TimeoutError, OSError):
pass
finally:
conn.close()
threads: list[threading.Thread] = []
try:
while stop is None or not stop.is_set():
try:
conn, _ = srv.accept()
except TimeoutError:
continue
conn.settimeout(15.0)
t = threading.Thread(target=_client, args=(conn,), daemon=True)
t.start()
threads.append(t)
finally:
srv.close()
if os.path.exists(path):
os.unlink(path)
def wait_for_socket(path: str, timeout: float = 90.0) -> bool:
"""Return True once a unix socket at path accepts a connection."""
deadline = time.time() + timeout
while time.time() < deadline:
if os.path.exists(path):
try:
s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
s.settimeout(0.3)
s.connect(path)
s.close()
return True
except OSError:
pass
time.sleep(0.05)
return False
def make_production_cleaner(**kw):
from audio_cleanup import AudioCleaner, apply_torch_thread_limits
threads = int(kw.pop("torch_num_threads", 2))
apply_torch_thread_limits(threads)
return AudioCleaner(torch_num_threads=threads, **kw)
def main(argv: Optional[list[str]] = None) -> int:
import argparse
ap = argparse.ArgumentParser(description="Shared MetricGAN enhance daemon")
ap.add_argument("--socket", default="/tmp/livekit-enhance.sock")
ap.add_argument("--enhance-mode", default="force")
ap.add_argument("--enhance-model",
default="speechbrain/metricgan-plus-voicebank")
ap.add_argument("--enhance-chunk-s", type=float, default=1.0)
ap.add_argument("--enhance-hop-s", type=float, default=0.1)
ap.add_argument("--enhance-infer", default="auto")
ap.add_argument("--beamform-mode", default="auto")
ap.add_argument("--torch-num-threads", type=int, default=2)
ap.add_argument("--tdoa-every", type=int, default=5)
ap.add_argument("--audio-rate", type=int, default=16000)
args = ap.parse_args(argv)
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [enhance-daemon] %(levelname)s %(message)s")
log.info("loading shared MetricGAN (one copy for all cameras)")
def make():
return make_production_cleaner(
mode=args.enhance_mode,
model_source=args.enhance_model,
sample_rate=args.audio_rate,
chunk_s=args.enhance_chunk_s,
hop_s=args.enhance_hop_s,
beamform=args.beamform_mode,
torch_num_threads=args.torch_num_threads,
tdoa_every=args.tdoa_every,
enhance_infer=args.enhance_infer,
)
engine = SharedEnhanceEngine(make)
log.info("model ready backend=%s socket=%s",
getattr(engine._template, "backend", "?"), args.socket)
serve_unix(args.socket, engine.process)
return 0
if __name__ == "__main__":
import sys
sys.exit(main())
+87
View File
@@ -0,0 +1,87 @@
"""Length-prefixed int16 audio frames for the shared enhance daemon."""
from __future__ import annotations
import socket
import struct
import numpy as np
MAGIC = b"ENH1"
# magic(4) + ident_len(H) + channels(H) + n_samples(I)
_HEADER = struct.Struct("!4sHHI")
def dump_frame(identity: str, pcm: np.ndarray) -> bytes:
"""Serialize identity + int16 pcm ([N] or [N, C]) to one frame."""
pcm = np.asarray(pcm, dtype=np.int16)
if pcm.ndim == 1:
pcm = pcm.reshape(-1, 1)
elif pcm.ndim != 2:
raise ValueError(f"pcm must be 1-D or 2-D, got {pcm.shape}")
n_samples, channels = pcm.shape
ident = identity.encode("utf-8")
header = _HEADER.pack(MAGIC, len(ident), channels, n_samples)
return header + ident + np.ascontiguousarray(pcm).tobytes()
def load_frame(blob: bytes) -> tuple[str, np.ndarray]:
"""Parse a frame into (identity, int16 array)."""
if len(blob) < _HEADER.size:
raise ValueError("truncated enhance frame")
magic, ident_len, channels, n_samples = _HEADER.unpack_from(blob, 0)
if magic != MAGIC:
raise ValueError(f"bad enhance magic {magic!r}")
ident_end = _HEADER.size + ident_len
payload = blob[ident_end:]
need = n_samples * channels * 2
if len(payload) != need:
raise ValueError(f"pcm size {len(payload)} != {need}")
ident = blob[_HEADER.size:ident_end].decode("utf-8")
pcm = np.frombuffer(payload, dtype=np.int16).reshape(n_samples, channels)
if channels == 1:
pcm = pcm[:, 0]
return ident, pcm
def recvall(sock: socket.socket, n: int) -> bytes:
buf = bytearray()
while len(buf) < n:
chunk = sock.recv(n - len(buf))
if not chunk:
raise EOFError("enhance socket closed")
buf.extend(chunk)
return bytes(buf)
def read_frame(sock: socket.socket) -> tuple[str, np.ndarray]:
header = recvall(sock, _HEADER.size)
magic, ident_len, channels, n_samples = _HEADER.unpack(header)
if magic != MAGIC:
raise ValueError(f"bad enhance magic {magic!r}")
rest = recvall(sock, ident_len + n_samples * channels * 2)
return load_frame(header + rest)
def write_frame(sock: socket.socket, identity: str, pcm: np.ndarray) -> None:
sock.sendall(dump_frame(identity, pcm))
class EnhanceClient:
"""One persistent unix-socket connection to the enhance daemon."""
def __init__(self, path: str):
self.path = path
self._sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
self._sock.settimeout(15.0)
self._sock.connect(path)
def process(self, identity: str, pcm: np.ndarray) -> np.ndarray:
write_frame(self._sock, identity, pcm)
_, out = read_frame(self._sock)
return np.asarray(out, dtype=np.int16)
def close(self) -> None:
try:
self._sock.close()
except OSError:
pass
+136
View File
@@ -0,0 +1,136 @@
"""Latest-frame I420 slots in mmap files (single writer, many readers).
Double-buffered: the writer fills the inactive slot, then publishes seq.
Readers copy the active slot; a torn read (seq changed) is discarded.
"""
from __future__ import annotations
import mmap
import os
import struct
import time
from typing import Optional
MAGIC = b"LKFR"
VER = 2
HEADER = 64
_HDR = struct.Struct("<4sIIIIIQ") # magic, ver, w, h, seq, nbytes, ts_ns
SLOTS = 2
def frame_bytes(width: int, height: int) -> int:
return int(width) * int(height) * 3 // 2
def slot_size(width: int, height: int) -> int:
return HEADER + SLOTS * frame_bytes(width, height)
def _mmap(fd: int, size: int, writable: bool) -> mmap.mmap:
prot = mmap.PROT_READ | (mmap.PROT_WRITE if writable else 0)
return mmap.mmap(fd, size, flags=mmap.MAP_SHARED, prot=prot)
class LatestFrameWriter:
def __init__(self, path: str, width: int, height: int):
self.path = path
self.width = int(width)
self.height = int(height)
self.nbytes = frame_bytes(self.width, self.height)
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
size = slot_size(self.width, self.height)
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
try:
os.ftruncate(fd, size)
self._map = _mmap(fd, size, True)
finally:
os.close(fd)
self._seq = 0
self._pack_header(0)
def _pack_header(self, seq: int) -> None:
_HDR.pack_into(
self._map, 0, MAGIC, VER, self.width, self.height, seq, self.nbytes, 0)
def write(self, payload: bytes, ts_ns: int = 0) -> int:
if len(payload) < self.nbytes:
return self._seq
nxt = self._seq + 1
slot = nxt % SLOTS
off = HEADER + slot * self.nbytes
self._map[off:off + self.nbytes] = payload[:self.nbytes]
self._seq = nxt
_HDR.pack_into(
self._map, 0, MAGIC, VER, self.width, self.height,
self._seq, self.nbytes, int(ts_ns))
return self._seq
def close(self) -> None:
self._map.close()
class LatestFrameReader:
def __init__(self, path: str):
self.path = path
self._map: Optional[mmap.mmap] = None
self.width = 0
self.height = 0
self.nbytes = 0
def _open(self) -> bool:
if self._map is not None:
return True
try:
fd = os.open(self.path, os.O_RDONLY)
except FileNotFoundError:
return False
try:
st = os.fstat(fd)
if st.st_size < HEADER:
return False
buf = _mmap(fd, st.st_size, False)
finally:
os.close(fd)
magic, ver, w, h, _seq, nbytes, _ts = _HDR.unpack_from(buf, 0)
if magic != MAGIC or ver != VER or nbytes <= 0:
buf.close()
return False
if st.st_size < HEADER + SLOTS * nbytes:
buf.close()
return False
self._map = buf
self.width, self.height, self.nbytes = w, h, nbytes
return True
def latest(self) -> Optional[tuple[int, bytes]]:
buf = bytearray(self.nbytes or frame_bytes(16, 16))
seq = self.copy_into(buf)
if seq is None:
return None
return seq, bytes(buf[:self.nbytes])
def copy_into(self, dest: bytearray) -> Optional[int]:
"""Copy the active slot into dest (must be >= nbytes). None if torn/empty."""
if not self._open() or self._map is None:
return None
if len(dest) < self.nbytes:
raise ValueError("dest too small")
s1 = struct.unpack_from("<I", self._map, 16)[0]
if s1 == 0:
return None
slot = s1 % SLOTS
off = HEADER + slot * self.nbytes
dest[:self.nbytes] = self._map[off:off + self.nbytes]
s2 = struct.unpack_from("<I", self._map, 16)[0]
if s1 != s2:
return None
return s1
def wait_for_ready(path: str, timeout: float = 90.0) -> bool:
deadline = time.time() + timeout
while time.time() < deadline:
if os.path.exists(path):
return True
time.sleep(0.05)
return False
+7
View File
@@ -13,6 +13,8 @@ import re
import subprocess
import sys
from usb_ids import is_fleet_video, is_rally_pid, usb_vid_pid_from_sysfs
HEADER = """\
# Stable slots for 20x Logitech HD Pro Webcam C920 (046d:08e5).
# Slot = physical camera, identified by USB serial. Survives reboots,
@@ -67,6 +69,11 @@ def current_state():
if idx_attr != "0":
continue
real = sh(["readlink", "-f", str(node / "device")]).strip()
ids = usb_vid_pid_from_sysfs(real)
if ids is not None and not is_fleet_video(ids[0], ids[1]):
continue
if ids is not None and is_rally_pid(ids[1]):
continue
s = _serial_of(real)
primary_by_serial[s] = f"/dev/{node.name}"
return primary_by_serial, cards
+299 -10
View File
@@ -3,6 +3,7 @@ from __future__ import annotations
import logging
import os
import fcntl
import shutil
import subprocess
from typing import Optional
@@ -39,21 +40,131 @@ def test_pattern_cmd(out: DrmOutput, pattern: str = "smpte", width: int = 1280,
]
def appsrc_cmd(out: DrmOutput, width: int, height: int, fps: int = 30) -> list[str]:
"""Raw BGRA frames on stdin -> kmssink."""
block = width * height * 4
def appsrc_cmd(
out: DrmOutput,
width: int,
height: int,
fps: int = 30,
queue_buffers: int = 2,
pixel_format: str = "bgra",
) -> list[str]:
"""Raw BGRA or I420 frames on stdin -> kmssink."""
fmt = (pixel_format or "bgra").strip().lower()
if fmt in ("i420", "yuv420p"):
fmt = "i420"
block = width * height * 3 // 2
else:
fmt = "bgra"
block = width * height * 4
q = max(1, int(queue_buffers))
return [
GST_LAUNCH, "-e",
"fdsrc", "fd=0", f"blocksize={block}",
"!", "videoparse",
f"width={width}", f"height={height}", "format=bgra", f"framerate={fps}/1",
"!", "queue", "max-size-buffers=2", "leaky=downstream",
f"width={width}", f"height={height}", f"format={fmt}", f"framerate={fps}/1",
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
"!", "videoconvert",
"!", "videoscale", "method=0",
"!", *_kms_props(out),
]
def va_grid_cmd(
out: DrmOutput,
grid,
ingest_w: int,
ingest_h: int,
fps: int,
tile_fds: list[int] | None = None,
chrome_fd: int | None = None,
chrome_path: str | None = None,
queue_buffers: int = 1,
tile_h264_paths: list[str] | None = None,
) -> list[str]:
"""Chrome + N tiles -> Arc vacompositor -> kmssink.
Tiles are I420 fdsrcs (default) or Annex-B H.264 files/fifos decoded
with varenderD129h264dec. Chrome is still a frozen BGRA overlay.
"""
q = max(1, int(queue_buffers))
fps = max(1, int(fps))
if tile_h264_paths:
n = len(tile_h264_paths)
elif tile_fds:
n = len(tile_fds)
else:
raise ValueError("tile_fds or tile_h264_paths required")
cmd: list[str] = [GST_LAUNCH, "-e", "varenderD129compositor", "name=c"]
cmd += [
"sink_0::zorder=0", "sink_0::xpos=0", "sink_0::ypos=0",
f"sink_0::width={grid.width}", f"sink_0::height={grid.height}",
]
for i in range(n):
x, y, w, h = grid.inner_rect(i)
s = i + 1
cmd += [
f"sink_{s}::zorder=1",
f"sink_{s}::xpos={x}",
f"sink_{s}::ypos={y}",
f"sink_{s}::width={w}",
f"sink_{s}::height={h}",
]
cmd += [
"!", f"video/x-raw,width={grid.width},height={grid.height},framerate={fps}/1",
"!", "varenderD129postproc",
"!", f"video/x-raw,format=NV12,width={grid.width},height={grid.height}",
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
"!", *_kms_props(out),
]
chrome_block = grid.width * grid.height * 4
if chrome_path:
cmd += [
"filesrc", f"location={chrome_path}", f"blocksize={chrome_block}",
"!", "videoparse",
f"width={grid.width}", f"height={grid.height}",
"format=bgra", f"framerate={fps}/1",
"!", "imagefreeze", "is-live=true",
"!", "varenderD129postproc",
"!", "c.sink_0",
]
else:
if chrome_fd is None:
raise ValueError("chrome_fd or chrome_path required")
cmd += [
"fdsrc", f"fd={chrome_fd}", f"blocksize={chrome_block}",
"!", "videoparse",
f"width={grid.width}", f"height={grid.height}",
"format=bgra", f"framerate={fps}/1",
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
"!", "varenderD129postproc",
"!", "c.sink_0",
]
if tile_h264_paths:
for i, path in enumerate(tile_h264_paths):
s = i + 1
cmd += [
"filesrc", f"location={path}",
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
"!", "h264parse",
"!", "varenderD129h264dec",
"!", f"c.sink_{s}",
]
return cmd
tile_block = ingest_w * ingest_h * 3 // 2
assert tile_fds is not None
for i, fd in enumerate(tile_fds):
s = i + 1
cmd += [
"fdsrc", f"fd={fd}", f"blocksize={tile_block}",
"!", "videoparse",
f"width={ingest_w}", f"height={ingest_h}",
"format=i420", f"framerate={fps}/1",
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
"!", "varenderD129postproc", "add-borders=true",
"!", f"c.sink_{s}",
]
return cmd
class KmsPipeline:
"""One gst-launch kmssink process, fed raw BGRA on stdin (or self-running testsrc)."""
@@ -64,6 +175,8 @@ class KmsPipeline:
self.width = 0
self.height = 0
self.mode = "idle"
self.queue_buffers = 2
self.pixel_format = "bgra"
@property
def alive(self) -> bool:
@@ -103,17 +216,27 @@ class KmsPipeline:
)
self.mode = "pattern"
def start_appsrc(self, width: int, height: int, fps: int = 30) -> None:
if self.alive and self.mode == "appsrc" and self.width == width and self.height == height:
def start_appsrc(self, width: int, height: int, fps: int = 30,
queue_buffers: int = 2, pixel_format: str = "bgra") -> None:
queue_buffers = max(1, int(queue_buffers))
fmt = (pixel_format or "bgra").strip().lower()
if fmt in ("i420", "yuv420p"):
fmt = "i420"
else:
fmt = "bgra"
if (self.alive and self.mode == "appsrc" and self.width == width
and self.height == height and self.queue_buffers == queue_buffers
and self.pixel_format == fmt):
return
self.stop()
if not self.out.connected:
log.warning("%s: %s not connected, cannot start kmssink", self.identity, self.out.name)
return
cmd = appsrc_cmd(self.out, width, height, fps)
cmd = appsrc_cmd(self.out, width, height, fps, queue_buffers=queue_buffers,
pixel_format=fmt)
log.info(
"%s: gst appsrc %dx%d -> %s (connector %s %s)",
self.identity, width, height, self.out.name, self.out.connector_id, self.out.driver,
"%s: gst appsrc %dx%d %s -> %s (connector %s %s)",
self.identity, width, height, fmt, self.out.name, self.out.connector_id, self.out.driver,
)
self.proc = subprocess.Popen(
cmd,
@@ -126,6 +249,8 @@ class KmsPipeline:
self.width = width
self.height = height
self.mode = "appsrc"
self.queue_buffers = queue_buffers
self.pixel_format = fmt
def push(self, frame: bytes) -> bool:
if not self.alive or self.proc is None or self.proc.stdin is None:
@@ -152,10 +277,174 @@ class KmsPipeline:
return chunk.decode("utf-8", "replace")
class VaGridPipeline:
"""Arc vacompositor mosaic: chrome + one BGRA pipe per tile."""
def __init__(self, out: DrmOutput):
self.out = out
self.proc: Optional[subprocess.Popen] = None
self._writes: list[int] = []
self._chrome_path: Optional[str] = None
self._chrome_w: Optional[int] = None
self.ingest_w = 0
self.ingest_h = 0
self.n_tiles = 0
self.codec = "i420"
self._fifo_holds: list[int] = []
@property
def alive(self) -> bool:
return self.proc is not None and self.proc.poll() is None
def start(self, grid, ingest_w: int, ingest_h: int, fps: int = 15,
queue_buffers: int = 1, h264_paths: list[str] | None = None) -> None:
self.stop()
if not self.out.connected:
log.warning("va-grid: %s not connected", self.out.name)
return
n = len(grid.identities)
import tempfile
from display_grid import placeholder, display_name, bgra_to_i420
chrome_path = tempfile.NamedTemporaryFile(
prefix="livekit-grid-chrome-", suffix=".bgra", delete=False).name
with open(chrome_path, "wb") as f:
f.write(grid.render())
self._chrome_path = chrome_path
self.codec = "h264" if h264_paths else "i420"
reads: list[int] = []
writes: list[int] = []
if h264_paths:
from h264_fifo import hold_fifo
self._fifo_holds = [hold_fifo(p) for p in h264_paths]
cmd = va_grid_cmd(
self.out, grid, ingest_w, ingest_h, fps,
tile_h264_paths=h264_paths, chrome_path=chrome_path,
queue_buffers=queue_buffers)
pass_fds: tuple[int, ...] = ()
else:
pipe_sz = 1048576
for _ in range(n):
r, w = os.pipe()
os.set_inheritable(r, True)
try:
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, pipe_sz)
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, pipe_sz)
except OSError:
pass
reads.append(r)
writes.append(w)
cmd = va_grid_cmd(
self.out, grid, ingest_w, ingest_h, fps,
tile_fds=reads, chrome_path=chrome_path,
queue_buffers=queue_buffers)
pass_fds = tuple(reads)
log.info(
"va-grid: Arc compositor %dx%d ingest %dx%d %d tiles codec=%s -> %s id=%s",
grid.width, grid.height, ingest_w, ingest_h, n, self.codec,
self.out.name, self.out.connector_id)
self.proc = subprocess.Popen(
cmd,
stdin=subprocess.DEVNULL,
stdout=subprocess.DEVNULL,
stderr=None,
pass_fds=pass_fds,
env=_gst_env(),
close_fds=True,
)
for r in reads:
try:
os.close(r)
except OSError:
pass
self._writes = writes
self.ingest_w = ingest_w
self.ingest_h = ingest_h
self.n_tiles = n
if self.codec == "i420":
for i, ident in enumerate(grid.identities):
ph = placeholder(ingest_w, ingest_h, [display_name(ident)])
os.set_blocking(writes[i], True)
self._write_fd(writes[i], bgra_to_i420(ph))
os.set_blocking(writes[i], False)
def push_tile(self, index: int, frame: bytes | memoryview) -> bool:
if not self.alive or index < 0 or index >= self.n_tiles:
return False
return self._write_fd(self._writes[index], frame)
def _write_fd(self, fd: Optional[int], data: bytes | memoryview) -> bool:
if fd is None:
return False
view = memoryview(data)
off = 0
try:
while off < len(view):
try:
n = os.write(fd, view[off:])
if n == 0:
return False
off += n
except BlockingIOError:
if off == 0:
return True
continue
return True
except BrokenPipeError:
log.warning("va-grid: pipe closed (rc=%s)",
None if self.proc is None else self.proc.poll())
return False
def gst_errors(self) -> str:
if self.proc is None or self.proc.stderr is None:
return ""
try:
os.set_blocking(self.proc.stderr.fileno(), False)
except OSError:
pass
try:
chunk = self.proc.stderr.read() or b""
except OSError:
return ""
return chunk.decode("utf-8", "replace")
def stop(self) -> None:
for w in self._writes:
try:
os.close(w)
except OSError:
pass
self._writes = []
for fd in getattr(self, "_fifo_holds", []):
try:
os.close(fd)
except OSError:
pass
self._fifo_holds = []
self._chrome_w = None
if self._chrome_path:
try:
os.unlink(self._chrome_path)
except OSError:
pass
self._chrome_path = None
if self.proc is None:
return
if self.proc.poll() is None:
self.proc.terminate()
try:
self.proc.wait(timeout=3)
except subprocess.TimeoutExpired:
self.proc.kill()
self.proc.wait(timeout=2)
self.proc = None
def _gst_env() -> dict:
env = os.environ.copy()
# Never try X11/Wayland; kmssink is the point.
env.pop("DISPLAY", None)
env.pop("WAYLAND_DISPLAY", None)
env.setdefault("GST_GL_PLATFORM", "egl")
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
env.setdefault("LIBVA_DRM_DEVICE", "/dev/dri/renderD129")
return env
+29
View File
@@ -0,0 +1,29 @@
"""Named-pipe helpers for the capture <-> display H.264 sidecar.
Linux fifos deadlock if one side opens O_RDONLY and the other O_WRONLY
before either has a partner. Opening O_RDWR in the parent first makes
filesink/filesrc open() return immediately.
"""
from __future__ import annotations
import os
import stat
def hold_fifo(path: str, mode: int = 0o644) -> int:
"""Create `path` as a fifo if needed and open it O_RDWR | O_NONBLOCK.
Do not unlink an existing fifo: that splits the inode from any process
already blocked in open() and recreates the deadlock.
"""
directory = os.path.dirname(path)
if directory:
os.makedirs(directory, exist_ok=True)
try:
st = os.stat(path)
if not stat.S_ISFIFO(st.st_mode):
os.unlink(path)
os.mkfifo(path, mode)
except FileNotFoundError:
os.mkfifo(path, mode)
return os.open(path, os.O_RDWR | os.O_NONBLOCK)
+287
View File
@@ -0,0 +1,287 @@
"""Cheap MoveNet Lightning hand-raise detector for the KMS wall.
OpenVINO CPU, FP32 single-pose, ~4 Hz per tile. Does not touch capture/encode.
Arc/iGPU stay unused (kmssink + H.264).
"""
from __future__ import annotations
import logging
import threading
import time
from collections import deque
from pathlib import Path
from typing import Callable, Optional
import numpy as np
log = logging.getLogger("cameras.display")
MODEL_PATH = Path(__file__).resolve().parent / "models" / "movenet_singlepose_lightning.onnx"
MODELS_DIR = Path(__file__).resolve().parent / "models"
INPUT_SIZE = 192
MOVENET = {
"lightning": ("movenet_singlepose_lightning.onnx", 192),
"thunder": ("movenet_singlepose_thunder.onnx", 256),
}
def resolve_movenet(
name: str = "thunder",
models_dir: Path | None = None,
) -> tuple[Path, int, str]:
"""lightning=192, thunder=256. Same Xenova FP32 ONNX family as production Lightning."""
key = (name or "thunder").strip().lower()
if key not in MOVENET:
raise ValueError(f"DISPLAY_HAND_RAISE_MODEL must be lightning|thunder, got {name!r}")
fname, size = MOVENET[key]
root = Path(models_dir) if models_dir is not None else MODELS_DIR
return root / fname, int(size), key
# MoveNet COCO-17, output is (y, x, score) in [0, 1], y grows downward.
NOSE = 0
L_SHOULDER, R_SHOULDER = 5, 6
L_ELBOW, R_ELBOW = 7, 8
L_WRIST, R_WRIST = 9, 10
MIN_SCORE = 0.20
# Wrist must sit this far above the shoulder (normalized image y).
LIFT = 0.05
FACE_RADIUS2 = 0.12 * 0.12
def letterbox_rgb(rgb: np.ndarray, size: int = INPUT_SIZE) -> np.ndarray:
"""Pad to square so MoveNet is not stretched (16:9 C920 tiles)."""
import cv2
size = int(size)
h, w = rgb.shape[:2]
if h < 1 or w < 1:
return np.zeros((size, size, 3), dtype=np.uint8)
scale = size / float(max(h, w))
nw = max(1, int(round(w * scale)))
nh = max(1, int(round(h * scale)))
resized = cv2.resize(rgb, (nw, nh), interpolation=cv2.INTER_AREA)
out = np.zeros((size, size, 3), dtype=np.uint8)
y0 = (size - nh) // 2
x0 = (size - nw) // 2
out[y0:y0 + nh, x0:x0 + nw] = resized
return out
def wrist_is_raised(kpts: np.ndarray) -> bool:
"""True if either wrist is clearly up — above the shoulder, or above the head."""
if kpts is None or kpts.shape != (17, 3):
return False
k = kpts
if k[NOSE, 2] < MIN_SCORE and k[L_SHOULDER, 2] < MIN_SCORE and k[R_SHOULDER, 2] < MIN_SCORE:
return False
return (
_arm_up(k, L_WRIST, L_ELBOW, L_SHOULDER)
or _arm_up(k, R_WRIST, R_ELBOW, R_SHOULDER)
or _wrist_above_head(k)
)
def _near_face(k: np.ndarray, wr: np.ndarray) -> bool:
if k[NOSE, 2] < MIN_SCORE:
return False
dy = wr[0] - k[NOSE, 0]
dx = wr[1] - k[NOSE, 1]
return dx * dx + dy * dy < FACE_RADIUS2
def _wrist_above_head(k: np.ndarray) -> bool:
if k[NOSE, 2] < MIN_SCORE:
return False
for wi in (L_WRIST, R_WRIST):
wr = k[wi]
if wr[2] < MIN_SCORE:
continue
if wr[0] >= k[NOSE, 0] - LIFT:
continue
if _near_face(k, wr):
continue
return True
return False
def _arm_up(k: np.ndarray, wi: int, ei: int, si: int) -> bool:
wr, el, sh = k[wi], k[ei], k[si]
if wr[2] < MIN_SCORE or sh[2] < MIN_SCORE:
return False
if wr[0] >= sh[0] - LIFT:
return False
if el[2] >= MIN_SCORE and wr[0] >= el[0]:
return False
if _near_face(k, wr):
return False
return True
class RaiseLatch:
"""Vote `hits` in a sliding `window`; stay lit until `hold_s` without a hit."""
def __init__(self, hits: int = 3, hold_s: float = 1.0, window: int = 5) -> None:
self.hits = max(1, int(hits))
self.hold_s = max(0.0, float(hold_s))
self.window = max(self.hits, int(window))
self._buf: deque[bool] = deque(maxlen=self.window)
self.raised = False
self.last_true = 0.0
def update(self, pred: bool, now: float) -> bool:
self._buf.append(bool(pred))
if pred:
self.last_true = now
if sum(self._buf) >= self.hits:
self.raised = True
elif self.raised and (now - self.last_true) >= self.hold_s:
self.raised = False
return self.raised
class HandRaiseMonitor:
"""Background 2 Hz pose on the latest I420 tile per identity."""
def __init__(
self,
enabled: bool = True,
hz: float = 4.0,
hold_s: float = 1.0,
on_change: Optional[Callable[[str, bool], None]] = None,
model_path: Optional[Path] = None,
model: str = "thunder",
input_size: Optional[int] = None,
) -> None:
self.on_change = on_change
self.hz = min(10.0, max(0.5, float(hz)))
self.hold_s = float(hold_s)
if model_path is not None:
self._path = Path(model_path)
self._variant = (model or "custom").strip().lower()
self._input_size = int(input_size or (256 if "thunder" in self._path.name else 192))
else:
self._path, self._input_size, self._variant = resolve_movenet(model)
if input_size is not None:
self._input_size = int(input_size)
self._lock = threading.Lock()
self._latest: dict[str, tuple[bytes, int, int]] = {}
self._latch: dict[str, RaiseLatch] = {}
self._raised: set[str] = set()
self._stop = threading.Event()
self._thread: Optional[threading.Thread] = None
self._request = None
self._ov_input = None
self.enabled = bool(enabled) and self._load()
def _load(self) -> bool:
if not self._path.is_file():
log.warning("hand-raise model missing: %s", self._path)
return False
try:
import openvino as ov
core = ov.Core()
model = core.read_model(str(self._path))
compiled = core.compile_model(
model,
"CPU",
{"INFERENCE_NUM_THREADS": 1, "NUM_STREAMS": 1},
)
self._request = compiled.create_infer_request()
self._ov_input = compiled.input(0)
log.info(
"hand-raise MoveNet %s FP32 OpenVINO CPU hz=%.1f hold=%.1fs input=%d model=%s",
self._variant, self.hz, self.hold_s, self._input_size, self._path.name,
)
return True
except Exception as exc: # noqa: BLE001
log.warning("hand-raise model load failed: %s", exc)
self._request = None
return False
def start(self) -> None:
if not self.enabled or self._thread is not None:
return
self._thread = threading.Thread(
target=self._run, name="hand-raise", daemon=True)
self._thread.start()
def stop(self) -> None:
self._stop.set()
t = self._thread
if t is not None and t.is_alive():
t.join(timeout=2.0)
self._thread = None
def offer(self, identity: str, i420: bytes | memoryview, width: int, height: int) -> None:
if not self.enabled or width < 16 or height < 16:
return
blob = i420 if isinstance(i420, (bytes, bytearray)) else bytes(i420)
with self._lock:
self._latest[identity] = (blob, int(width), int(height))
def forget(self, identity: str) -> None:
with self._lock:
self._latest.pop(identity, None)
self._latch.pop(identity, None)
was = identity in self._raised
self._raised.discard(identity)
if was and self.on_change is not None:
try:
self.on_change(identity, False)
except Exception: # noqa: BLE001
log.exception("hand-raise on_change(%s, False)", identity)
def raised(self) -> set[str]:
with self._lock:
return set(self._raised)
def _run(self) -> None:
import cv2
period = 1.0 / self.hz
while not self._stop.wait(period):
with self._lock:
snapshot = list(self._latest.items())
now = time.monotonic()
for ident, (blob, w, h) in snapshot:
try:
pred = self._infer(cv2, blob, w, h)
except Exception as exc: # noqa: BLE001
log.warning("hand-raise infer %s: %s", ident, exc)
pred = False
with self._lock:
latch = self._latch.get(ident)
if latch is None:
latch = RaiseLatch(hits=3, hold_s=self.hold_s, window=5)
self._latch[ident] = latch
was = latch.raised
now_raised = latch.update(pred, now)
if now_raised:
self._raised.add(ident)
else:
self._raised.discard(ident)
if now_raised != was:
log.info("hand-raise %s %s", ident, "on" if now_raised else "off")
if self.on_change is not None:
try:
self.on_change(ident, now_raised)
except Exception: # noqa: BLE001
log.exception("hand-raise on_change(%s, %s)", ident, now_raised)
def infer_rgb(self, rgb: np.ndarray) -> np.ndarray:
"""MoveNet (17, 3) y,x,score from an HxWx3 RGB uint8 image."""
if self._request is None:
return np.zeros((17, 3), dtype=np.float32)
small = letterbox_rgb(rgb, self._input_size)
inp = np.expand_dims(np.ascontiguousarray(small, dtype=np.int32), 0)
result = self._request.infer({self._ov_input: inp})
return np.asarray(next(iter(result.values())), dtype=np.float32).reshape(17, 3)
def _infer(self, cv2, blob: bytes, w: int, h: int) -> bool:
need = w * h * 3 // 2
if self._request is None or len(blob) < need:
return False
yuv = np.frombuffer(blob, dtype=np.uint8, count=need).reshape((h * 3 // 2, w))
rgb = cv2.cvtColor(yuv, cv2.COLOR_YUV2RGB_I420)
return wrist_is_raised(self.infer_rgb(rgb))
+121
View File
@@ -0,0 +1,121 @@
"""Int16 PCM hop rings in mmap (single writer, one reader).
Unlike video latest-frame slots, audio must not skip hops. The ring keeps
a few hops; a slow reader drops the oldest.
"""
from __future__ import annotations
import mmap
import os
import struct
from typing import Optional
MAGIC = b"LKHP"
VER = 1
HEADER = 64
_HDR = struct.Struct("<4sIIIIQ") # magic, ver, hop_bytes, nslots, seq, _pad
DEFAULT_SLOTS = 8
def hop_bytes(rate: int, channels: int, hop_s: float) -> int:
n = int(round(float(hop_s) * int(rate)))
return n * int(channels) * 2
def _mmap(fd: int, size: int, writable: bool) -> mmap.mmap:
prot = mmap.PROT_READ | (mmap.PROT_WRITE if writable else 0)
return mmap.mmap(fd, size, flags=mmap.MAP_SHARED, prot=prot)
class HopWriter:
def __init__(self, path: str, hop_nbytes: int, nslots: int = DEFAULT_SLOTS):
self.path = path
self.nbytes = int(hop_nbytes)
self.nslots = int(nslots)
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
size = HEADER + self.nslots * self.nbytes
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
try:
os.ftruncate(fd, size)
self._map = _mmap(fd, size, True)
finally:
os.close(fd)
self._seq = 0
_HDR.pack_into(self._map, 0, MAGIC, VER, self.nbytes, self.nslots, 0, 0)
def write(self, payload: bytes) -> int:
if len(payload) < self.nbytes:
return self._seq
nxt = self._seq + 1
slot = nxt % self.nslots
off = HEADER + slot * self.nbytes
self._map[off:off + self.nbytes] = payload[:self.nbytes]
self._seq = nxt
_HDR.pack_into(
self._map, 0, MAGIC, VER, self.nbytes, self.nslots, self._seq, 0)
return self._seq
def close(self) -> None:
self._map.close()
class HopReader:
def __init__(self, path: str):
self.path = path
self._map: Optional[mmap.mmap] = None
self.nbytes = 0
self.nslots = 0
self._last = 0
def _open(self) -> bool:
if self._map is not None:
return True
try:
fd = os.open(self.path, os.O_RDONLY)
except FileNotFoundError:
return False
try:
st = os.fstat(fd)
if st.st_size < HEADER:
return False
buf = _mmap(fd, st.st_size, False)
finally:
os.close(fd)
magic, ver, nbytes, nslots, seq, _p = _HDR.unpack_from(buf, 0)
if magic != MAGIC or ver != VER or nbytes <= 0 or nslots < 2:
buf.close()
return False
self._map = buf
self.nbytes, self.nslots = nbytes, nslots
self._last = 0
return True
def _close(self) -> None:
if self._map is None:
return
try:
self._map.close()
finally:
self._map = None
self.nbytes = 0
self.nslots = 0
def read(self) -> Optional[bytes]:
if not self._open() or self._map is None:
return None
seq = struct.unpack_from("<I", self._map, 16)[0]
if seq < self._last:
# Writer restarted (cameras-only): seq went back to 0.
self._close()
self._last = 0
if not self._open() or self._map is None:
return None
seq = struct.unpack_from("<I", self._map, 16)[0]
if seq <= self._last:
return None
if seq - self._last > self.nslots:
self._last = seq - 1
self._last += 1
slot = self._last % self.nslots
off = HEADER + slot * self.nbytes
return bytes(self._map[off:off + self.nbytes])
+76
View File
@@ -0,0 +1,76 @@
"""16-align I420→NV12 for Intel VA (NV12 only). Pad up, never crop."""
from __future__ import annotations
import numpy as np
def coded_dim(n: int) -> int:
n = int(n)
if n <= 0:
return 0
return (n + 15) // 16 * 16
def nv12_bytes(width: int, height: int) -> int:
return int(width) * int(height) * 3 // 2
def va_visible_height(width: int, height: int) -> int:
"""Drop the 8-row VA pad when the remainder is 16:9 (360/1080)."""
w, h = int(width), int(height)
vis = h - 8
if w > 0 and vis > 0 and vis * 16 == w * 9:
return vis
return h
def drop_va_pad_i420(src, width: int, height: int):
"""Crop coded pad off I420. Returns (payload, w, visible_h)."""
w, h = int(width), int(height)
vis = va_visible_height(w, h)
if vis == h:
return src, w, h
ysz = w * h
uvw, uvh = w // 2, h // 2
uvsz = uvw * uvh
raw = np.frombuffer(src, dtype=np.uint8, count=ysz + 2 * uvsz)
y = raw[:ysz].reshape(h, w)[:vis, :]
u = raw[ysz:ysz + uvsz].reshape(uvh, uvw)[: vis // 2, :]
v = raw[ysz + uvsz:ysz + 2 * uvsz].reshape(uvh, uvw)[: vis // 2, :]
out = np.empty(w * vis * 3 // 2, dtype=np.uint8)
out[: w * vis] = y.reshape(-1)
out[w * vis:w * vis + (w // 2) * (vis // 2)] = u.reshape(-1)
out[w * vis + (w // 2) * (vis // 2):] = v.reshape(-1)
return out.tobytes(), w, vis
def i420_to_nv12_into(src, width: int, height: int, out: bytearray) -> int:
"""Pad to 16-aligned WxH and write NV12 into out. Returns nbytes."""
w, h = int(width), int(height)
cw, ch = coded_dim(w), coded_dim(h)
need = nv12_bytes(cw, ch)
if len(out) < need:
raise ValueError("NV12 dest too small")
ysz = w * h
uvw, uvh = w // 2, h // 2
uvsz = uvw * uvh
raw = np.frombuffer(src, dtype=np.uint8, count=ysz + 2 * uvsz)
y = raw[:ysz].reshape(h, w)
u = raw[ysz:ysz + uvsz].reshape(uvh, uvw)
v = raw[ysz + uvsz:ysz + 2 * uvsz].reshape(uvh, uvw)
nv = np.frombuffer(out, dtype=np.uint8, count=need).reshape(ch + ch // 2, cw)
nv[:h, :w] = y
if ch > h:
nv[h:ch, :w] = y[-1]
if cw > w:
nv[:ch, w:cw] = nv[:ch, w - 1:w]
uv = nv[ch:, :].reshape(ch // 2, cw // 2, 2)
uh, uw = h // 2, w // 2
uv[:uh, :uw, 0] = u
uv[:uh, :uw, 1] = v
if ch // 2 > uh:
uv[uh:, :uw, 0] = u[-1]
uv[uh:, :uw, 1] = v[-1]
if cw // 2 > uw:
uv[:, uw:, :] = uv[:, uw - 1:uw, :]
return need
+136
View File
@@ -0,0 +1,136 @@
"""Pure helpers for env-flagged capture/display latency knobs."""
from __future__ import annotations
def audio_queue_size_ms(hop_s: float, audio_queue_ms: int | None = None) -> int:
"""LiveKit AudioSource queue. Unset AUDIO_QUEUE_MS keeps the old 150 ms floor."""
hop_ms = int(round(float(hop_s) * 1000.0))
if audio_queue_ms is not None:
return max(1, int(audio_queue_ms))
return max(150, hop_ms + 50)
def video_hold_seconds(
hop_s: float,
has_audio: bool,
video_hold_s: float | None = None,
) -> float:
"""Delay published video to match audio hops. VIDEO_HOLD_S=0 disables."""
if not has_audio:
return 0.0
if video_hold_s is not None:
return max(0.0, float(video_hold_s))
return float(hop_s)
def effective_enhance_mode(
mode: str,
identity: str,
speaker_only: bool,
speaker_camera: str,
) -> str:
"""ENHANCE_SPEAKER_ONLY=1 keeps enhance on DISPLAY_SPEAKER_CAMERA only.
``speaker_camera=active`` has no single identity, so every camera keeps
the configured mode.
"""
if not speaker_only:
return mode
target = (speaker_camera or "").strip()
if not target or target.lower() == "active":
return mode
if identity != target:
return "off"
return mode
def next_video_capture(now: float, due: float, period: float) -> tuple[bool, float]:
"""Pace VideoSource.capture_frame.
When the loop is late, capture once and jump the next due time to
``now + period`` so we never catch up by pushing extra frames into
libwebrtc's EncoderQueue (that path leaked ~2.6 MiB/s native RSS).
"""
period = float(period)
if period <= 0:
period = 1.0 / 30.0
if now < due:
return False, due
return True, now + period
def should_drop_stale_vaapi_frame(
now_us: int,
capture_us: int,
fps: int,
max_frames_behind: int = 3,
is_keyframe: bool = False,
) -> bool:
"""True when VAAPI Encode should drop before ToI420 (encoder behind).
LiveKit's VAAPIH264EncoderWrapper::Encode never dropped; EncoderQueue
kept I420 copies (~0.37 extra frames/s/cam → OOM). Keyframes always
pass. Missing timestamps are not treated as stale.
"""
if is_keyframe:
return False
if int(capture_us) <= 0:
return False
fps = max(1, int(fps))
budget = max(1, int(max_frames_behind))
max_delay_us = budget * 1_000_000 // fps
return int(now_us) - int(capture_us) > max_delay_us
def should_drop_encoder_queue_depth(queued: int, max_queued: int = 1) -> bool:
"""True when VideoTrackSource must not push another frame (latest-wins)."""
return int(queued) >= max(1, int(max_queued))
def video_capture_gate_acquire(gate: dict) -> bool:
"""Admit a capture_frame if in-flight slots remain (AudioSource-like).
``gate`` is ``{in_flight, dropped, max_in_flight}``. Caller must
``video_capture_gate_release`` when the encoder slot is free (or after
one frame period as a stand-in when FFI has no completion callback).
"""
max_in_flight = max(1, int(gate.get("max_in_flight") or 3))
if int(gate.get("in_flight") or 0) >= max_in_flight:
gate["dropped"] = int(gate.get("dropped") or 0) + 1
return False
gate["in_flight"] = int(gate.get("in_flight") or 0) + 1
return True
def video_capture_gate_release(gate: dict) -> None:
"""Free one in-flight capture slot."""
n = int(gate.get("in_flight") or 0)
gate["in_flight"] = n - 1 if n > 0 else 0
def should_recycle_encoders(
rss_anon_kb: int,
limit_kb: int,
now: float,
last_recycle_at: float,
min_interval_s: float = 60.0,
) -> bool:
"""True when native RSS is over the cap and a recycle is allowed."""
if int(limit_kb) <= 0:
return False
if int(rss_anon_kb) < int(limit_kb):
return False
if last_recycle_at > 0 and (now - last_recycle_at) < float(min_interval_s):
return False
return True
def video_stream_kwargs(capacity: int = 0, format_name: str = "") -> dict:
"""LiveKit VideoStream options. capacity=0 is unbounded (SDK default)."""
out: dict = {}
if int(capacity) > 0:
out["capacity"] = int(capacity)
fmt = (format_name or "").strip().lower()
if fmt:
out["format"] = fmt
return out
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+109
View File
@@ -0,0 +1,109 @@
"""Latest-access-unit H.264 slots in mmap (single writer, one reader).
Double-buffered like frame_shm: fill the inactive slot, then publish seq.
Unread AUs are overwritten. Oversize payloads are dropped (seq unchanged).
"""
from __future__ import annotations
import mmap
import os
import struct
from typing import Optional
MAGIC = b"LKNA"
VER = 1
HEADER = 64
_HDR = struct.Struct("<4sIIIIQ") # magic, ver, max_payload, seq, nbytes, ts_ns
SLOTS = 2
AU_MAX = 256 * 1024
def _mmap(fd: int, size: int, writable: bool) -> mmap.mmap:
prot = mmap.PROT_READ | (mmap.PROT_WRITE if writable else 0)
return mmap.mmap(fd, size, flags=mmap.MAP_SHARED, prot=prot)
class NalWriter:
def __init__(self, path: str, max_payload: int = AU_MAX):
self.path = path
self.max_payload = int(max_payload)
if self.max_payload < 1 or self.max_payload > 2 * 1024 * 1024:
raise ValueError("max_payload out of range")
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
size = HEADER + SLOTS * self.max_payload
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
try:
os.ftruncate(fd, size)
self._map = _mmap(fd, size, True)
finally:
os.close(fd)
self._seq = 0
_HDR.pack_into(self._map, 0, MAGIC, VER, self.max_payload, 0, 0, 0)
def write(self, payload: bytes, ts_ns: int = 0) -> int:
n = len(payload)
if n < 1 or n > self.max_payload:
return self._seq
nxt = self._seq + 1
slot = nxt % SLOTS
off = HEADER + slot * self.max_payload
self._map[off:off + n] = payload
self._seq = nxt
_HDR.pack_into(
self._map, 0, MAGIC, VER, self.max_payload, self._seq, n, int(ts_ns))
return self._seq
def close(self) -> None:
self._map.close()
class NalReader:
def __init__(self, path: str):
self.path = path
self._map: Optional[mmap.mmap] = None
self.max_payload = 0
self._last = 0
def _open(self) -> bool:
if self._map is not None:
return True
try:
fd = os.open(self.path, os.O_RDONLY)
except FileNotFoundError:
return False
try:
st = os.fstat(fd)
if st.st_size < HEADER:
return False
self._map = _mmap(fd, st.st_size, False)
finally:
os.close(fd)
magic, ver, max_payload, seq, nbytes, _ts = _HDR.unpack_from(self._map, 0)
if magic != MAGIC or ver != VER or max_payload < 1:
self._map.close()
self._map = None
return False
self.max_payload = int(max_payload)
return True
def read(self) -> tuple[int, bytes] | None:
if not self._open() or self._map is None:
return None
magic, ver, max_payload, seq, nbytes, _ts = _HDR.unpack_from(self._map, 0)
if magic != MAGIC or seq <= self._last or nbytes < 1:
return None
if nbytes > max_payload:
return None
slot = seq % SLOTS
off = HEADER + slot * max_payload
payload = bytes(self._map[off:off + nbytes])
magic2, _v, _m, seq2, nbytes2, _t = _HDR.unpack_from(self._map, 0)
if seq2 != seq or nbytes2 != nbytes:
return None
self._last = seq
return seq, payload
def close(self) -> None:
if self._map is not None:
self._map.close()
self._map = None
BIN
View File
Binary file not shown.
+2 -1
View File
@@ -7,6 +7,7 @@ from __future__ import annotations
import fcntl
import logging
import os
from pathlib import Path
import numpy as np
@@ -85,7 +86,7 @@ def compile_cpu(onnx_path: Path, num_threads: int = 1):
"CPU",
{
"INFERENCE_NUM_THREADS": num_threads,
"NUM_STREAMS": 1,
"NUM_STREAMS": max(1, int(os.environ.get("OV_NUM_STREAMS", "8"))),
},
)
request = compiled.create_infer_request()
+87
View File
@@ -0,0 +1,87 @@
"""LiveKit participant site tags (attributes/metadata) and display hiding."""
from __future__ import annotations
import json
from typing import Iterable, Mapping, Optional
from display_grid import is_camera_identity
def parse_tags(spec: str | None) -> list[str]:
out: list[str] = []
seen: set[str] = set()
for part in (spec or "").split(","):
tag = part.strip().lower()
if tag and tag not in seen:
seen.add(tag)
out.append(tag)
return out
def token_attributes(tags: Iterable[str]) -> dict[str, str]:
cleaned = parse_tags(",".join(t for t in tags if t))
if not cleaned:
return {}
return {"tag": cleaned[0], "tags": ",".join(cleaned)}
def tags_from_fields(
attributes: Optional[Mapping[str, str]],
metadata: Optional[str],
) -> set[str]:
tags: set[str] = set()
attrs = attributes or {}
if attrs.get("tag"):
tags.update(parse_tags(str(attrs["tag"])))
if attrs.get("tags"):
tags.update(parse_tags(str(attrs["tags"])))
meta = (metadata or "").strip()
if not meta:
return tags
try:
data = json.loads(meta)
except json.JSONDecodeError:
tags.update(parse_tags(meta))
return tags
if isinstance(data, dict):
if data.get("tag"):
tags.update(parse_tags(str(data["tag"])))
if data.get("tags"):
tags.update(parse_tags(str(data["tags"])))
elif isinstance(data, str):
tags.update(parse_tags(data))
return tags
def should_hide_participant(
attributes: Optional[Mapping[str, str]],
metadata: Optional[str],
hide_tags: Iterable[str],
) -> bool:
hide = {t.strip().lower() for t in hide_tags if t and str(t).strip()}
if not hide:
return False
return bool(tags_from_fields(attributes, metadata) & hide)
def want_display_video(
*,
identity: str,
attributes: Optional[Mapping[str, str]],
metadata: Optional[str],
hide_tags: Iterable[str],
participant_prefix: str,
display_identity: str,
speaker_identity: str = "",
) -> bool:
ident = (identity or "").strip()
if not ident or ident == display_identity:
return False
if should_hide_participant(attributes, metadata, hide_tags):
return False
hide = [t for t in hide_tags if t]
if hide:
return True
if speaker_identity and ident == speaker_identity:
return True
return is_camera_identity(ident, participant_prefix)
+15
View File
@@ -0,0 +1,15 @@
# livekit-ffi-backpressure.patch
Apply on livekit/rust-sdks (webrtc-sys VAAPI + VideoTrackSource).
1. VideoSource: latest-wins pending slot — at most one OnFrame in flight;
a new capture overwrites pending instead of pushing another I420 onto
EncoderQueue.
2. VAAPI H264 Encode: drop delta frames older than 1 frame period, or
Encode calls closer than 8 ms (queue drain), *before* ToI420.
Logs `VAAPI encode sample delay_ms=` every 90 frames so the clock
domain is visible in the cameras journal.
Rebuild: `CC=clang CXX=clang++ cargo build --release -p livekit-ffi`
Install: copy `target/release/liblivekit_ffi.so` to `native/` and set
`LIVEKIT_LIB_PATH` (see systemd drop-in `ffi-backpressure.conf`).
+210
View File
@@ -0,0 +1,210 @@
diff --git a/livekit-ffi/src/server/video_source.rs b/livekit-ffi/src/server/video_source.rs
index ee1410a..b2ff9ca 100644
--- a/livekit-ffi/src/server/video_source.rs
+++ b/livekit-ffi/src/server/video_source.rs
@@ -88,7 +88,9 @@ impl FfiVideoSource {
buffer,
};
- source.capture_frame(&frame);
+ // false = adapter / in-flight cap dropped the frame (do not
+ // treat as an FFI error; the source is still valid).
+ let _forwarded = source.capture_frame(&frame);
}
_ => {}
}
diff --git a/webrtc-sys/include/livekit/video_track.h b/webrtc-sys/include/livekit/video_track.h
index 4146c45..733d110 100644
--- a/webrtc-sys/include/livekit/video_track.h
+++ b/webrtc-sys/include/livekit/video_track.h
@@ -18,6 +18,7 @@
#include <atomic>
#include <memory>
+#include <optional>
#include "api/media_stream_interface.h"
#include "api/video/video_frame.h"
@@ -127,6 +128,10 @@ class VideoTrackSource {
std::shared_ptr<livekit::EncodedRateControlState> rate_control_state_ =
std::make_shared<livekit::EncodedRateControlState>();
bool is_screencast_;
+ int64_t last_forwarded_us_ = 0;
+ // Latest-wins: at most one OnFrame in flight; new captures overwrite.
+ bool forwarding_ = false;
+ std::optional<webrtc::VideoFrame> pending_frame_;
};
public:
diff --git a/webrtc-sys/src/vaapi/h264_encoder_impl.cpp b/webrtc-sys/src/vaapi/h264_encoder_impl.cpp
index 5a3b76e..8c7ce11 100644
--- a/webrtc-sys/src/vaapi/h264_encoder_impl.cpp
+++ b/webrtc-sys/src/vaapi/h264_encoder_impl.cpp
@@ -189,18 +189,6 @@ int32_t VAAPIH264EncoderWrapper::Encode(
return WEBRTC_VIDEO_CODEC_UNINITIALIZED;
}
- webrtc::scoped_refptr<I420BufferInterface> frame_buffer =
- input_frame.video_frame_buffer()->ToI420();
- if (!frame_buffer) {
- RTC_LOG(LS_ERROR) << "Failed to convert "
- << VideoFrameBufferTypeToString(
- input_frame.video_frame_buffer()->type())
- << " image to I420. Can't encode frame.";
- return WEBRTC_VIDEO_CODEC_ENCODER_FAILURE;
- }
- RTC_CHECK(frame_buffer->type() == VideoFrameBuffer::Type::kI420 ||
- frame_buffer->type() == VideoFrameBuffer::Type::kI420A);
-
bool is_keyframe_needed = false;
if (configuration_.key_frame_request && configuration_.sending) {
is_keyframe_needed = true;
@@ -214,20 +202,67 @@ int32_t VAAPIH264EncoderWrapper::Encode(
configuration_.key_frame_request = false;
}
- RTC_DCHECK_EQ(configuration_.width, frame_buffer->width());
- RTC_DCHECK_EQ(configuration_.height, frame_buffer->height());
-
if (!configuration_.sending) {
return WEBRTC_VIDEO_CODEC_NO_OUTPUT;
}
if (frame_types != nullptr) {
- // Skip frame?
if ((*frame_types)[0] == VideoFrameType::kEmptyFrame) {
return WEBRTC_VIDEO_CODEC_NO_OUTPUT;
}
}
+ // Drop stale/burst delta frames BEFORE ToI420.
+ // 1) Age: capture older than 1 frame period (was 3; that never fired).
+ // 2) Burst: Encode called closer than 8 ms = EncoderQueue draining.
+ // Mirror: livekit-cameras latency.should_drop_stale_vaapi_frame /
+ // should_drop_encoder_queue_depth.
+ const int fps = std::max(
+ 1, static_cast<int>(configuration_.max_frame_rate > 0.0f
+ ? configuration_.max_frame_rate
+ : codec_.maxFramerate));
+ const int64_t now_us = webrtc::TimeMicros();
+ const int64_t capture_us = input_frame.timestamp_us();
+ const int64_t delay_us = (capture_us > 0) ? (now_us - capture_us) : 0;
+ encode_count_++;
+ if (encode_count_ % 90 == 1) {
+ RTC_LOG(LS_WARNING) << "VAAPI encode sample delay_ms=" << delay_us / 1000
+ << " now_us=" << now_us << " capture_us=" << capture_us
+ << " fps=" << fps << " n=" << encode_count_;
+ }
+ if (!send_key_frame) {
+ const int64_t max_delay_us = 1000000 / fps; // 1 frame
+ const bool stale =
+ capture_us > 0 && now_us - capture_us > max_delay_us;
+ const bool burst =
+ last_encode_us_ != 0 && now_us - last_encode_us_ < 8000;
+ if (stale || burst) {
+ RTC_LOG(LS_WARNING) << "VAAPI H264 drop "
+ << (stale ? "stale" : "burst")
+ << " frame delay_ms=" << delay_us / 1000
+ << " dt_ms=" << (now_us - last_encode_us_) / 1000
+ << " fps=" << fps;
+ last_encode_us_ = now_us;
+ return WEBRTC_VIDEO_CODEC_OK;
+ }
+ }
+ last_encode_us_ = now_us;
+
+ webrtc::scoped_refptr<I420BufferInterface> frame_buffer =
+ input_frame.video_frame_buffer()->ToI420();
+ if (!frame_buffer) {
+ RTC_LOG(LS_ERROR) << "Failed to convert "
+ << VideoFrameBufferTypeToString(
+ input_frame.video_frame_buffer()->type())
+ << " image to I420. Can't encode frame.";
+ return WEBRTC_VIDEO_CODEC_ENCODER_FAILURE;
+ }
+ RTC_CHECK(frame_buffer->type() == VideoFrameBuffer::Type::kI420 ||
+ frame_buffer->type() == VideoFrameBuffer::Type::kI420A);
+
+ RTC_DCHECK_EQ(configuration_.width, frame_buffer->width());
+ RTC_DCHECK_EQ(configuration_.height, frame_buffer->height());
+
std::vector<uint8_t> output;
encoder_->Encode(VA_FOURCC_I420, frame_buffer->DataY(), frame_buffer->DataU(),
frame_buffer->DataV(), send_key_frame, output);
@@ -267,6 +302,7 @@ VideoEncoder::EncoderInfo VAAPIH264EncoderWrapper::GetEncoderInfo() const {
info.implementation_name = "VAAPI H264 Encoder";
info.scaling_settings = VideoEncoder::ScalingSettings::kOff;
info.is_hardware_accelerated = true;
+ info.has_trusted_rate_controller = false;
info.supports_simulcast = false;
info.preferred_pixel_formats = {VideoFrameBuffer::Type::kI420};
return info;
diff --git a/webrtc-sys/src/vaapi/h264_encoder_impl.h b/webrtc-sys/src/vaapi/h264_encoder_impl.h
index dc26355..bce1681 100644
--- a/webrtc-sys/src/vaapi/h264_encoder_impl.h
+++ b/webrtc-sys/src/vaapi/h264_encoder_impl.h
@@ -76,6 +76,8 @@ class VAAPIH264EncoderWrapper : public VideoEncoder {
const SdpVideoFormat format_;
H264Profile profile_ = H264Profile::kProfileConstrainedBaseline;
H264Level level_ = H264Level::kLevel1_b;
+ int64_t last_encode_us_ = 0;
+ int encode_count_ = 0;
};
} // namespace webrtc
diff --git a/webrtc-sys/src/video_track.cpp b/webrtc-sys/src/video_track.cpp
index 92410b9..89e6791 100644
--- a/webrtc-sys/src/video_track.cpp
+++ b/webrtc-sys/src/video_track.cpp
@@ -165,6 +165,7 @@ VideoResolution VideoTrackSource::InternalSource::video_resolution() const {
bool VideoTrackSource::InternalSource::on_captured_frame(
const webrtc::VideoFrame& frame,
const FrameMetadata& frame_metadata) {
+ {
webrtc::MutexLock lock(&mutex_);
int64_t aligned_timestamp_us = timestamp_aligner_.TranslateTimestamp(
@@ -235,13 +236,37 @@ bool VideoTrackSource::InternalSource::on_captured_frame(
frame_metadata.has_packet_trailer ? frame_metadata.frame_id : 0);
}
- OnFrame(webrtc::VideoFrame::Builder()
- .set_video_frame_buffer(buffer)
- .set_rotation(rotation)
- .set_timestamp_us(aligned_timestamp_us)
- .build());
+ webrtc::VideoFrame built = webrtc::VideoFrame::Builder()
+ .set_video_frame_buffer(buffer)
+ .set_rotation(rotation)
+ .set_timestamp_us(aligned_timestamp_us)
+ .build();
+
+ // Latest-wins depth cap (max_queued=1). A capture that arrives while
+ // OnFrame is in flight overwrites pending_frame_ instead of pushing
+ // another I420 onto EncoderQueue.
+ // Mirror: livekit-cameras latency.should_drop_encoder_queue_depth.
+ pending_frame_ = std::move(built);
+ if (forwarding_) {
+ return true;
+ }
+ forwarding_ = true;
+ last_forwarded_us_ = aligned_timestamp_us;
+ }
- return true;
+ for (;;) {
+ std::optional<webrtc::VideoFrame> to_send;
+ {
+ webrtc::MutexLock lock(&mutex_);
+ if (!pending_frame_) {
+ forwarding_ = false;
+ return true;
+ }
+ to_send = std::move(pending_frame_);
+ pending_frame_.reset();
+ }
+ OnFrame(*to_send);
+ }
}
void VideoTrackSource::InternalSource::set_packet_trailer_handler(
+403
View File
@@ -0,0 +1,403 @@
"""Person-aware background blur for C920 I420 SHM.
OpenVINO CPU only. iGPU stays H.264; Arc stays kmssink.
"""
from __future__ import annotations
import os
from pathlib import Path
from typing import Optional
import numpy as np
MODELS_DIR = Path(__file__).resolve().parent / "models"
SELFIE_ONNX = MODELS_DIR / "selfie_segmentation.onnx"
MOVENET_LIGHTNING = MODELS_DIR / "movenet_singlepose_lightning.onnx"
POSE_SIZE = 192
YUNET_CANDIDATES = (
"face_detection_yunet_2026may.onnx",
"face_detection_yunet_2023mar.onnx",
)
RALLY_ID = "rally"
def i420_source_path(identity: str, raw_dir: str, portrait_dir: str = "") -> str:
"""C920s read portrait SHM when enabled; rally always stays on capture SHM."""
ident = (identity or "").strip()
raw = (raw_dir or "").rstrip("/")
por = (portrait_dir or "").rstrip("/")
if por and ident and ident != RALLY_ID:
return f"{por}/{ident}.i420"
return f"{raw}/{ident}.i420"
def i420_nbytes(width: int, height: int) -> int:
return int(width) * int(height) * 3 // 2
def i420_to_bgr(buf, width: int, height: int) -> np.ndarray:
import cv2
w, h = int(width), int(height)
need = i420_nbytes(w, h)
raw = np.frombuffer(buf, dtype=np.uint8, count=need)
yuv = raw.reshape((h * 3 // 2, w))
return cv2.cvtColor(yuv, cv2.COLOR_YUV2BGR_I420)
def bgr_to_i420(bgr: np.ndarray) -> bytes:
import cv2
yuv = cv2.cvtColor(bgr, cv2.COLOR_BGR2YUV_I420)
return np.ascontiguousarray(yuv).tobytes()
def soft_blur_bgr(img: np.ndarray, radius: int = 7) -> np.ndarray:
"""Cheap bokeh: half-res box blur, then upsample."""
import cv2
k = max(3, int(radius) | 1)
h, w = img.shape[:2]
sw, sh = max(2, w // 2), max(2, h // 2)
small = cv2.resize(img, (sw, sh), interpolation=cv2.INTER_AREA)
small = cv2.blur(small, (k, k))
return cv2.resize(small, (w, h), interpolation=cv2.INTER_LINEAR)
def feather_mask(mask: np.ndarray, ksize: int = 7) -> np.ndarray:
import cv2
k = max(3, int(ksize) | 1)
m = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
return cv2.GaussianBlur(m, (k, k), 0)
def composite_bgr(sharp: np.ndarray, blurred: np.ndarray, mask: np.ndarray) -> np.ndarray:
a = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
if a.ndim == 2:
a = a[:, :, None]
out = sharp.astype(np.float32) * a + blurred.astype(np.float32) * (1.0 - a)
return np.clip(out, 0, 255).astype(np.uint8)
def person_present(
mask: np.ndarray,
min_peak: float = 0.35,
min_area: float = 0.04,
) -> bool:
m = np.asarray(mask, dtype=np.float32)
if m.size == 0:
return False
if float(m.max()) < float(min_peak):
return False
return float((m >= min_peak).mean()) >= float(min_area)
def union_masks(*masks: Optional[np.ndarray]) -> np.ndarray:
acc: Optional[np.ndarray] = None
for m in masks:
if m is None:
continue
a = np.clip(np.asarray(m, dtype=np.float32), 0.0, 1.0)
acc = a if acc is None else np.maximum(acc, a)
if acc is None:
return np.zeros((1, 1), dtype=np.float32)
return acc
# MoveNet COCO-17: nose, eyes, ears, shoulders, elbows, wrists.
_POSE_R = {
0: 0.14, # nose / face
1: 0.08, 2: 0.08, 3: 0.08, 4: 0.08, # eyes, ears
5: 0.10, 6: 0.10, # shoulders
7: 0.09, 8: 0.09, # elbows
9: 0.11, 10: 0.11, # wrists / hands
}
_POSE_MIN_SCORE = 0.25
def pose_protect_mask(
height: int,
width: int,
kpts: np.ndarray,
min_score: float = _POSE_MIN_SCORE,
) -> np.ndarray:
"""Soft disks on face + arms so raised hands stay sharp."""
h, w = int(height), int(width)
out = np.zeros((h, w), dtype=np.float32)
if kpts is None or kpts.shape != (17, 3) or h < 2 or w < 2:
return out
scale = float(min(h, w))
yy, xx = np.ogrid[:h, :w]
for idx, frac in _POSE_R.items():
y, x, s = float(kpts[idx, 0]), float(kpts[idx, 1]), float(kpts[idx, 2])
if s < min_score:
continue
cy, cx = y * h, x * w
r = max(6.0, frac * scale)
dist = ((xx - cx) / r) ** 2 + ((yy - cy) / r) ** 2
out = np.maximum(out, np.clip(1.0 - dist, 0.0, 1.0).astype(np.float32))
return out
def apply_portrait(
bgr: np.ndarray,
mask: np.ndarray,
radius: int = 7,
feather: int = 7,
) -> np.ndarray:
"""Passthrough never: empty seats are full-frame blur."""
if bgr is None:
return bgr
if mask is None or not person_present(mask):
z = np.zeros(bgr.shape[:2], dtype=np.float32)
return composite_bgr(bgr, soft_blur_bgr(bgr, radius), z)
soft = feather_mask(mask, feather)
return composite_bgr(bgr, soft_blur_bgr(bgr, radius), soft)
class MaskHold:
"""EMA + hold so a missed infer does not shimmer or drop the person."""
def __init__(self, hold_s: float = 0.8, ema: float = 0.4) -> None:
self.hold_s = max(0.0, float(hold_s))
self.ema = min(1.0, max(0.05, float(ema)))
self.mask: Optional[np.ndarray] = None
self.last_true = 0.0
def update(
self,
mask: Optional[np.ndarray],
present: bool,
now: float,
) -> Optional[np.ndarray]:
if present and mask is not None:
m = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
if self.mask is None or self.ema >= 1.0:
self.mask = m
else:
a = self.ema
self.mask = a * m + (1.0 - a) * self.mask
self.last_true = float(now)
return self.mask
if self.mask is not None and (float(now) - self.last_true) < self.hold_s:
return self.mask
self.mask = None
return None
def apply_i420(
payload,
width: int,
height: int,
mask: Optional[np.ndarray],
radius: int = 7,
feather: int = 7,
) -> bytes:
"""I420 in/out. Same WxH so encode MCU crop is unchanged."""
n = i420_nbytes(width, height)
raw = payload[:n]
if mask is None or not person_present(mask):
z = np.zeros((int(height), int(width)), dtype=np.float32)
return _composite_i420(raw, width, height, z, radius, feather=0)
return _composite_i420(raw, width, height, mask, radius, feather)
def _composite_i420(
payload,
width: int,
height: int,
mask: np.ndarray,
radius: int,
feather: int,
) -> bytes:
"""Blur background luma in I420. UV stays; avoids BGR round-trip."""
import cv2
w, h = int(width), int(height)
n = i420_nbytes(w, h)
arr = np.frombuffer(memoryview(payload)[:n], dtype=np.uint8).copy()
y = arr[: w * h].reshape(h, w)
a = np.asarray(mask, dtype=np.float32)
if a.shape != (h, w):
a = cv2.resize(a, (w, h), interpolation=cv2.INTER_LINEAR)
if feather and feather > 1:
a = feather_mask(a, feather)
k = max(3, int(radius) | 1)
sw, sh = max(2, w // 2), max(2, h // 2)
y_s = cv2.resize(y, (sw, sh), interpolation=cv2.INTER_AREA)
y_s = cv2.blur(y_s, (k, k))
yb = cv2.resize(y_s, (w, h), interpolation=cv2.INTER_LINEAR)
af = a
y[:] = (y.astype(np.float32) * af + yb.astype(np.float32) * (1.0 - af)).clip(0, 255).astype(np.uint8)
return arr.tobytes()
def letterbox_bgr(bgr: np.ndarray, size: int = 256) -> tuple[np.ndarray, tuple[int, int, int, int]]:
"""Pad 16:9 into a square. Returns canvas and (x0, y0, nw, nh) content box."""
import cv2
size = int(size)
h, w = bgr.shape[:2]
if h < 1 or w < 1:
return np.zeros((size, size, 3), dtype=np.uint8), (0, 0, size, size)
scale = size / float(max(h, w))
nw = max(1, int(round(w * scale)))
nh = max(1, int(round(h * scale)))
resized = cv2.resize(bgr, (nw, nh), interpolation=cv2.INTER_AREA)
canvas = np.zeros((size, size, 3), dtype=np.uint8)
y0 = (size - nh) // 2
x0 = (size - nw) // 2
canvas[y0:y0 + nh, x0:x0 + nw] = resized
return canvas, (x0, y0, nw, nh)
def unletterbox_mask(
mask: np.ndarray,
box: tuple[int, int, int, int],
height: int,
width: int,
) -> np.ndarray:
import cv2
x0, y0, nw, nh = (int(v) for v in box)
crop = np.asarray(mask, dtype=np.float32)[y0:y0 + nh, x0:x0 + nw]
if crop.size == 0:
return np.zeros((int(height), int(width)), dtype=np.float32)
return cv2.resize(crop, (int(width), int(height)), interpolation=cv2.INTER_LINEAR)
def resolve_yunet(models_dir: Path | None = None) -> Path:
root = Path(models_dir) if models_dir is not None else MODELS_DIR
for name in YUNET_CANDIDATES:
path = root / name
if path.is_file():
return path
return root / YUNET_CANDIDATES[0]
def _face_soft_mask(bgr: np.ndarray, faces) -> Optional[np.ndarray]:
"""Soft ellipse per YuNet face so close-up heads stay unblurred."""
if faces is None or len(faces) == 0:
return None
h, w = bgr.shape[:2]
acc = None
yy, xx = np.ogrid[:h, :w]
for best in faces:
x, y, fw, fh = (float(best[0]), float(best[1]), float(best[2]), float(best[3]))
if fw < 8 or fh < 8:
continue
cx = x + fw * 0.5
cy = y + fh * 0.42
rx = max(fw * 1.15, w * 0.12)
ry = max(fh * 1.35, h * 0.16)
dist = ((xx - cx) / rx) ** 2 + ((yy - cy) / ry) ** 2
m = np.clip(1.0 - dist, 0.0, 1.0).astype(np.float32)
acc = m if acc is None else np.maximum(acc, m)
return acc
class PersonSeg:
"""MediaPipe selfie ONNX on OpenVINO CPU. Never GPU.0/GPU.1."""
def __init__(
self,
model_path: Path | None = None,
yunet_path: Path | None = None,
num_threads: int = 2,
) -> None:
self.device = "CPU"
self._request = None
self._ov_input = None
self._size = 256
self._yunet = None
self._cv2 = None
self._pose_request = None
self._pose_input = None
self._pose_size = POSE_SIZE
path = Path(model_path) if model_path is not None else SELFIE_ONNX
import cv2
import openvino as ov
self._cv2 = cv2
try:
cv2.setNumThreads(1)
except Exception:
pass
if not path.is_file():
raise FileNotFoundError(path)
core = ov.Core()
model = core.read_model(str(path))
compiled = core.compile_model(
model,
"CPU",
{"INFERENCE_NUM_THREADS": max(1, int(num_threads)), "NUM_STREAMS": 1},
)
self._request = compiled.create_infer_request()
self._ov_input = compiled.input(0)
ypath = Path(yunet_path) if yunet_path is not None else resolve_yunet()
if ypath.is_file():
self._yunet = cv2.FaceDetectorYN_create(
str(ypath), "", (320, 180), 0.45, 0.3, 5000)
if MOVENET_LIGHTNING.is_file():
pose = core.read_model(str(MOVENET_LIGHTNING))
pose_c = core.compile_model(
pose,
"CPU",
{"INFERENCE_NUM_THREADS": 1, "NUM_STREAMS": 1},
)
self._pose_request = pose_c.create_infer_request()
self._pose_input = pose_c.input(0)
def infer_bgr(self, bgr: np.ndarray) -> np.ndarray:
h, w = bgr.shape[:2]
canvas, box = letterbox_bgr(bgr, self._size)
rgb = canvas[:, :, ::-1].astype(np.float32) * (1.0 / 255.0)
inp = np.transpose(rgb, (2, 0, 1))[None]
self._request.infer({self._ov_input: inp})
raw = np.asarray(self._request.get_output_tensor(0).data, dtype=np.float32)
raw = np.squeeze(raw)
mask = unletterbox_mask(raw, box, h, w)
face = self._yunet_mask(bgr)
if face is not None:
mask = union_masks(mask, face)
if person_present(mask) or face is not None:
pose = self._pose_mask(bgr)
if pose is not None and float(pose.max()) > 0:
mask = union_masks(mask, pose)
return mask
def _pose_mask(self, bgr: np.ndarray) -> Optional[np.ndarray]:
if self._pose_request is None or self._pose_input is None:
return None
h, w = bgr.shape[:2]
canvas, box = letterbox_bgr(bgr, self._pose_size)
rgb = np.ascontiguousarray(canvas[:, :, ::-1], dtype=np.int32)
inp = rgb[None]
self._pose_request.infer({self._pose_input: inp})
kpts = np.asarray(
self._pose_request.get_output_tensor(0).data, dtype=np.float32
).reshape(17, 3)
x0, y0, nw, nh = box
size = float(self._pose_size)
mapped = kpts.copy()
mapped[:, 0] = (kpts[:, 0] * size - y0) / max(1.0, float(nh))
mapped[:, 1] = (kpts[:, 1] * size - x0) / max(1.0, float(nw))
return pose_protect_mask(h, w, mapped)
def _yunet_mask(self, bgr: np.ndarray) -> Optional[np.ndarray]:
if self._yunet is None or self._cv2 is None:
return None
h, w = bgr.shape[:2]
dw, dh = 320, 180
small = self._cv2.resize(bgr, (dw, dh), interpolation=self._cv2.INTER_AREA)
self._yunet.setInputSize((dw, dh))
_ok, faces = self._yunet.detect(small)
if faces is None or len(faces) == 0:
return None
sx = w / float(dw)
sy = h / float(dh)
scaled = []
for f in faces:
scaled.append([f[0] * sx, f[1] * sy, f[2] * sx, f[3] * sy])
return _face_soft_mask(bgr, scaled)
+233
View File
@@ -0,0 +1,233 @@
"""C920 I420 SHM -> person mask -> blurred background SHM.
OpenVINO CPU selfie seg. Does not open DRM. Rally is not processed.
"""
from __future__ import annotations
import os
os.environ.setdefault("OMP_NUM_THREADS", "1")
os.environ.setdefault("OPENBLAS_NUM_THREADS", "1")
os.environ.setdefault("MKL_NUM_THREADS", "1")
import argparse
import json
import logging
import signal
import threading
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Callable
import numpy as np
from frame_shm import LatestFrameReader, LatestFrameWriter, wait_for_ready
from portrait import (
MaskHold,
PersonSeg,
apply_i420,
i420_nbytes,
i420_to_bgr,
person_present,
)
log = logging.getLogger("cameras.portrait-daemon")
READY_NAME = "ready"
InferFn = Callable[[np.ndarray], np.ndarray]
@dataclass
class CamSlot:
identity: str
width: int
height: int
reader: LatestFrameReader
writer: LatestFrameWriter
infer_hz: float = 8.0
hold_s: float = 0.8
blur_px: int = 7
last_seq: int = -1
last_infer: float = 0.0
hold: MaskHold = field(default_factory=MaskHold)
scratch: bytearray = field(default_factory=bytearray)
def __post_init__(self) -> None:
self.hold = MaskHold(hold_s=self.hold_s, ema=0.4)
n = i420_nbytes(self.width, self.height)
self.scratch = bytearray(n)
self.infer_scratch = bytearray(n)
self.infer_reader = LatestFrameReader(self.reader.path)
def tick_once(
slots: list[CamSlot],
infer: InferFn,
now: float,
infer_budget_s: float = 0.008,
) -> int:
"""Infer a few due cameras, composite every new SHM frame. Returns writes."""
wrote = 0
deadline = time.monotonic() + max(0.001, float(infer_budget_s))
pending: list[tuple[CamSlot, int]] = []
for s in sorted(slots, key=lambda c: c.last_infer):
seq = s.reader.copy_into(s.scratch)
if seq is None or seq == s.last_seq:
continue
due = (now - s.last_infer) >= (1.0 / max(0.5, s.infer_hz))
if due and time.monotonic() < deadline:
bgr = i420_to_bgr(s.scratch, s.width, s.height)
raw = infer(bgr)
present = person_present(raw)
s.hold.update(raw if present else None, present, now)
s.last_infer = now
pending.append((s, seq))
for s, seq in pending:
out = apply_i420(
s.scratch, s.width, s.height, s.hold.mask, radius=s.blur_px)
s.writer.write(out)
s.last_seq = seq
wrote += 1
return wrote
def composite_once(slots: list[CamSlot]) -> int:
"""30 fps path: no OpenVINO."""
wrote = 0
for s in slots:
seq = s.reader.copy_into(s.scratch)
if seq is None or seq == s.last_seq:
continue
out = apply_i420(
s.scratch, s.width, s.height, s.hold.mask, radius=s.blur_px)
s.writer.write(out)
s.last_seq = seq
wrote += 1
return wrote
def infer_once(slots: list[CamSlot], infer: InferFn, now: float, budget_s: float = 0.012) -> int:
n = 0
deadline = time.monotonic() + max(0.001, float(budget_s))
for s in sorted(slots, key=lambda c: c.last_infer):
if time.monotonic() >= deadline:
break
if (now - s.last_infer) < (1.0 / max(0.5, s.infer_hz)):
continue
seq = s.infer_reader.copy_into(s.infer_scratch)
if seq is None:
continue
bgr = i420_to_bgr(s.infer_scratch, s.width, s.height)
raw = infer(bgr)
present = person_present(raw)
s.hold.update(raw if present else None, present, now)
s.last_infer = now
n += 1
return n
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--cameras-json", required=True)
ap.add_argument("--in-dir", default="/run/livekit-cameras/raw")
ap.add_argument("--out-dir", default="/run/livekit-cameras/portrait")
ap.add_argument("--hz", type=float, default=8.0)
ap.add_argument("--blur-px", type=int, default=7)
ap.add_argument("--hold-s", type=float, default=0.8)
ap.add_argument("--model", default="")
args = ap.parse_args(argv)
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [portrait-daemon] %(levelname)s %(message)s")
import cv2
cv2.setNumThreads(1)
os.environ.setdefault("OMP_NUM_THREADS", "1")
os.environ.setdefault("OPENBLAS_NUM_THREADS", "1")
cams = json.loads(args.cameras_json)
cams = [c for c in cams if c.get("identity") != "rally"]
if not cams:
log.error("no C920 cameras")
return 2
model_path = Path(args.model) if args.model else None
seg = PersonSeg(model_path=model_path, num_threads=2)
log.info(
"portrait OpenVINO %s model=%s cams=%d hz=%.1f blur=%d hold=%.2fs",
seg.device,
(model_path or Path("models/selfie_segmentation.onnx")).name,
len(cams), args.hz, args.blur_px, args.hold_s,
)
os.makedirs(args.out_dir, exist_ok=True)
slots: list[CamSlot] = []
for c in cams:
ident = c["identity"]
w, h = int(c["width"]), int(c["height"])
src = os.path.join(args.in_dir, f"{ident}.i420")
dst = os.path.join(args.out_dir, f"{ident}.i420")
if not wait_for_ready(src, timeout=20):
log.warning("missing capture SHM %s", src)
continue
slots.append(CamSlot(
identity=ident, width=w, height=h,
reader=LatestFrameReader(src),
writer=LatestFrameWriter(dst, w, h),
infer_hz=float(args.hz),
hold_s=float(args.hold_s),
blur_px=int(args.blur_px),
))
if not slots:
log.error("no portrait slots")
return 2
ready = os.path.join(args.out_dir, READY_NAME)
Path(ready).write_text("ok\n")
stop = False
t0 = time.monotonic()
writes = 0
infers = 0
stop_ev = threading.Event()
def _stop(*_a):
nonlocal stop
stop = True
stop_ev.set()
signal.signal(signal.SIGTERM, _stop)
signal.signal(signal.SIGINT, _stop)
def _infer_loop() -> None:
nonlocal infers
while not stop_ev.is_set():
infers += infer_once(slots, seg.infer_bgr, time.monotonic(), 0.012)
time.sleep(0.004)
th = threading.Thread(target=_infer_loop, name="portrait-infer", daemon=True)
th.start()
try:
while not stop:
n = composite_once(slots)
writes += n
if time.monotonic() - t0 >= 10:
log.info(
"writes=%d infers=%d last_seq %s",
writes, infers,
" ".join(f"{s.identity}:{s.last_seq}" for s in slots),
)
writes = 0
infers = 0
t0 = time.monotonic()
if n == 0:
time.sleep(0.003)
finally:
stop_ev.set()
try:
os.unlink(ready)
except OSError:
pass
return 0
if __name__ == "__main__":
raise SystemExit(main())
+99
View File
@@ -0,0 +1,99 @@
"""Logitech Rally Camera V-R0010 + Dante room mic (speaker identity)."""
from __future__ import annotations
import logging
import re
from pathlib import Path
from discovery import Camera
from usb_ids import (
RALLY_IDENTITY,
is_rally_pid,
looks_like_rally_name,
usb_vid_pid_of_video_node,
)
log = logging.getLogger("cameras.rally")
_CARD_RE = re.compile(r"\s*(\d+)\s+\[([A-Za-z0-9_]+)\s*\]:\s*(.*)")
def dante_capture_card() -> int | None:
"""ALSA card index for the Audinate Dante USB I/O module, if present."""
path = Path("/proc/asound/cards")
if not path.is_file():
return None
for line in path.read_text().splitlines():
m = _CARD_RE.match(line)
if not m:
continue
rest = (m.group(3) or "").lower()
name = (m.group(2) or "").lower()
if "dante" in rest or "dante" in name:
return int(m.group(1))
return None
def rally_video_node() -> str | None:
pinned = Path("/dev/rally")
if pinned.exists():
return str(pinned)
sysfs = Path("/sys/class/video4linux")
if not sysfs.is_dir():
return None
for p in sorted(sysfs.glob("video*"), key=lambda x: int(x.name[5:] or 0)):
name = (p / "name").read_text().strip() if (p / "name").is_file() else ""
idx = (p / "index").read_text().strip() if (p / "index").is_file() else ""
node = f"/dev/{p.name}"
if idx != "0":
continue
if looks_like_rally_name(name):
return node
ids = usb_vid_pid_of_video_node(node)
if ids is not None and is_rally_pid(ids[1]):
return node
return None
def _serial_of_video(node: str) -> str:
link = Path(f"/sys/class/video4linux/{Path(node).name}/device")
try:
p = link.resolve()
except OSError:
return ""
for _ in range(12):
cand = p / "serial"
if cand.is_file():
return cand.read_text().strip()
if str(p).startswith("/sys/devices"):
p = p.parent
else:
break
return ""
def discover_rally(*, width: int = 1920, height: int = 1080) -> Camera | None:
"""Speaker camera: Rally UVC + Dante capture. None if the head is missing."""
video = rally_video_node()
if not video:
return None
card = dante_capture_card()
serial = _serial_of_video(video)
cam = Camera(
index=20,
identity=RALLY_IDENTITY,
video_device=video,
audio_card=card,
serial=serial,
width=width,
height=height,
)
log.info(
"rally video=%s audio=%s serial=%s %dx%d",
video,
f"hw:{card},0" if card is not None else "none",
serial or "-",
width,
height,
)
return cam
+580
View File
@@ -0,0 +1,580 @@
"""Rally speaker follow: capture /dev/rally to I420 SHM + YuNet PTZ.
No X11. Local HDMI preview is a LiveKit subscriber (run_displays speaker
role on i915 HDMI-A-5). This process does not kmssink unless --connector
is set. Person tracking is host software. V-R0010 has no native VISCA.
"""
from __future__ import annotations
import argparse
import logging
import os
import select
import signal
import subprocess
import sys
import time
from pathlib import Path
import numpy as np
from drm_outputs import DrmOutput, list_outputs
from frame_shm import LatestFrameWriter, frame_bytes
from gst_sink import GST_LAUNCH, _kms_props
from hop_shm import HopWriter, hop_bytes
from visca_xu_bridge import RallyV4L2
log = logging.getLogger("cameras.rally_follow")
def pick_preview(name: str) -> DrmOutput:
for o in list_outputs():
if o.name == name:
return o
have = [f"{o.name}/{o.driver}" for o in list_outputs() if o.connected]
raise SystemExit(f"DRM connector {name!r} not found. connected={have}")
def video_cmd(device: str, width: int, height: int, fps: int,
fd: int, preview: DrmOutput | None = None) -> list[str]:
caps = f"image/jpeg,width={int(width)},height={int(height)},framerate={int(fps)}/1"
head = [
GST_LAUNCH, "-q",
"v4l2src", f"device={device}", "do-timestamp=false",
"!", caps,
"!", "jpegparse",
"!", "jpegdec", "qos=false",
]
i420 = [
"!", "videoconvert", "qos=false", "n-threads=2",
"!", f"video/x-raw,format=I420,width={int(width)},height={int(height)}",
"!", "fdsink", f"fd={int(fd)}", "sync=false",
]
if preview is None:
return head + i420
return head + [
"!", "tee", "name=t",
"t.", "!", "queue", "max-size-buffers=1", "max-size-bytes=0",
"max-size-time=0", "leaky=downstream",
"!", "videoconvert", "qos=false",
"!", "videoscale", "method=0",
"!", *_kms_props(preview),
"t.", "!", "queue", "max-size-buffers=1", "max-size-bytes=0",
"max-size-time=0", "leaky=downstream",
] + i420
def audio_cmd(card: int, rate: int, fd: int) -> list[str]:
return [
GST_LAUNCH, "-q",
"alsasrc", f"device=hw:{int(card)},0", "do-timestamp=false",
"!", "audioconvert",
"!", "audioresample",
"!", f"audio/x-raw,format=S16LE,rate={int(rate)},channels=2",
"!", "queue", "max-size-buffers=2", "max-size-bytes=0",
"max-size-time=0", "leaky=downstream",
"!", "fdsink", f"fd={int(fd)}", "sync=false",
]
_YUNET_2026 = Path(__file__).parent / "models" / "face_detection_yunet_2026may.onnx"
_YUNET_2023 = Path(__file__).parent / "models" / "face_detection_yunet_2023mar.onnx"
_YUNET = _YUNET_2026 if _YUNET_2026.is_file() else _YUNET_2023
# YuNet row: x,y,w,h, re, le, nose, rmouth, lmouth, score (15).
_SCORE_MIN = 0.80
_MIN_FRAC = 0.06
_MAX_FRAC = 0.50
_TOP_BAND = 0.10
_BOT_BAND = 0.90
_ASPECT_LO = 0.45
_ASPECT_HI = 1.40
_CONFIRM = 3
_HOLD_S = 2.0
_MOTION_MAD = 8.0
def resolve_yunet(models_dir: Path | None = None) -> Path:
root = Path(models_dir) if models_dir is not None else Path(__file__).parent / "models"
newer = root / "face_detection_yunet_2026may.onnx"
old = root / "face_detection_yunet_2023mar.onnx"
return newer if newer.is_file() else old
def _box_ok(x: float, y: float, w: float, h: float, score: float,
frame_w: int, frame_h: int) -> bool:
if score < _SCORE_MIN or w <= 1 or h <= 1:
return False
fw = float(frame_w)
fh = float(frame_h)
short = min(fw, fh)
if min(w, h) < _MIN_FRAC * short:
return False
if max(w, h) > _MAX_FRAC * max(fw, fh):
return False
ar = w / h
if ar < _ASPECT_LO or ar > _ASPECT_HI:
return False
cx = (x + w / 2.0) / fw
cy = (y + h / 2.0) / fh
if cy < _TOP_BAND or cy > _BOT_BAND:
return False
if cx < 0.02 or cx > 0.98:
return False
return True
def _frontal_landmarks(row, x: float, y: float, w: float, h: float) -> bool:
re_x, re_y = float(row[4]), float(row[5])
le_x, le_y = float(row[6]), float(row[7])
ns_x, ns_y = float(row[8]), float(row[9])
rm_x, rm_y = float(row[10]), float(row[11])
lm_x, lm_y = float(row[12]), float(row[13])
eye_y = 0.5 * (re_y + le_y)
mouth_y = 0.5 * (rm_y + lm_y)
if not (eye_y + 0.04 * h < ns_y < mouth_y - 0.04 * h):
return False
eye_lo = min(re_x, le_x)
eye_hi = max(re_x, le_x)
if not (eye_lo + 0.08 * w <= ns_x <= eye_hi - 0.08 * w):
return False
eye_dist = ((le_x - re_x) ** 2 + (le_y - re_y) ** 2) ** 0.5
if eye_dist < 0.22 * w or eye_dist > 0.85 * w:
return False
if abs(le_y - re_y) > 0.35 * h:
return False
return True
def _profile_landmarks(row, x: float, y: float, w: float, h: float) -> bool:
"""Side / turned head: landmarks clustered, nose not between the eyes."""
pts = [(float(row[i]), float(row[i + 1])) for i in range(4, 14, 2)]
pad_x, pad_y = 0.2 * w, 0.2 * h
inside = 0
for px, py in pts:
if (x - pad_x) <= px <= (x + w + pad_x) and (y - pad_y) <= py <= (y + h + pad_y):
inside += 1
if inside < 3:
return False
re_x, re_y = float(row[4]), float(row[5])
le_x, le_y = float(row[6]), float(row[7])
eye_dist = ((le_x - re_x) ** 2 + (le_y - re_y) ** 2) ** 0.5
if eye_dist >= 0.22 * w:
return False
ns_y = float(row[9])
rm_y, lm_y = float(row[11]), float(row[13])
eye_y = 0.5 * (re_y + le_y)
mouth_y = 0.5 * (rm_y + lm_y)
if mouth_y + 0.02 * h < eye_y:
return False
if ns_y + 0.12 * h < eye_y:
return False
return True
def accept_yunet_face(row, frame_w: int, frame_h: int) -> bool:
"""True if this YuNet hit looks like a real face (frontal or turned), not ceiling/stand."""
if row is None or len(row) < 15:
return False
x, y, w, h = (float(row[0]), float(row[1]), float(row[2]), float(row[3]))
score = float(row[14])
if not _box_ok(x, y, w, h, score, frame_w, frame_h):
return False
return _frontal_landmarks(row, x, y, w, h) or _profile_landmarks(row, x, y, w, h)
def select_yunet_face(faces, frame_w: int, frame_h: int):
"""Highest-score accepted face, or None."""
if faces is None or len(faces) == 0:
return None
best = None
best_score = -1.0
for row in faces:
if not accept_yunet_face(row, frame_w, frame_h):
continue
sc = float(row[14])
if sc > best_score:
best_score = sc
best = row
return best
class FaceGate:
"""Need `need` similar hits before the box is trusted for PTZ."""
def __init__(self, need: int = _CONFIRM, max_jump: float = 0.18) -> None:
self.need = max(1, int(need))
self.max_jump = float(max_jump)
self.hits = 0
self.cx = 0.0
self.cy = 0.0
def update(self, box: tuple[int, int, int, int] | None,
frame_w: int, frame_h: int) -> tuple[int, int, int, int] | None:
if box is None:
self.hits = 0
return None
x, y, w, h = box
cx = (x + w / 2.0) / float(frame_w)
cy = (y + h / 2.0) / float(frame_h)
if self.hits > 0:
jump = ((cx - self.cx) ** 2 + (cy - self.cy) ** 2) ** 0.5
if jump > self.max_jump:
self.hits = 1
else:
self.hits += 1
else:
self.hits = 1
self.cx, self.cy = cx, cy
if self.hits >= self.need:
return box
return None
def roi_motion(
prev: np.ndarray | None,
gray: np.ndarray | None,
box: tuple[int, int, int, int] | None,
*,
expand: float = 1.4,
min_mad: float = _MOTION_MAD,
) -> bool:
"""True if the last head ROI moved; motion elsewhere is ignored."""
if prev is None or gray is None or box is None:
return False
if prev.shape != gray.shape:
return False
h_img, w_img = gray.shape[:2]
x, y, w, h = (int(box[0]), int(box[1]), int(box[2]), int(box[3]))
extra_w = int((expand - 1.0) * w / 2.0)
extra_h = int((expand - 1.0) * h / 2.0)
x0 = max(0, x - extra_w)
y0 = max(0, y - extra_h)
x1 = min(w_img, x + w + extra_w)
y1 = min(h_img, y + h + extra_h)
if x1 - x0 < 4 or y1 - y0 < 4:
return False
a = prev[y0:y1, x0:x1].astype(np.float32)
b = gray[y0:y1, x0:x1].astype(np.float32)
return float(np.mean(np.abs(a - b))) >= float(min_mad)
class HeadHold:
"""Keep PTZ on the last confirmed face while the head ROI is still moving."""
def __init__(self, hold_s: float = _HOLD_S) -> None:
self.hold_s = max(0.0, float(hold_s))
self.box: tuple[int, int, int, int] | None = None
self.until = 0.0
def update(
self,
box: tuple[int, int, int, int] | None,
*,
moving: bool,
now: float,
) -> tuple[int, int, int, int] | None:
if box is not None:
self.box = box
self.until = now + self.hold_s
return box
if self.box is None:
return None
if moving:
self.until = now + self.hold_s
return self.box
if now < self.until:
return self.box
self.box = None
return None
def steer_speeds(
box: tuple[int, int, int, int],
frame_w: int,
frame_h: int,
deadband: float = 0.14,
) -> tuple[int, int]:
x, y, w, h = box
err_x = ((x + w / 2.0) - frame_w / 2.0) / (frame_w / 2.0)
err_y = ((y + h / 2.0) - frame_h / 2.0) / (frame_h / 2.0)
pan = 1 if err_x > deadband else (-1 if err_x < -deadband else 0)
tilt = -1 if err_y > deadband else (1 if err_y < -deadband else 0)
return pan, tilt
def lock_action(
confirmed: tuple[int, int, int, int] | None,
held: tuple[int, int, int, int] | None,
) -> str:
"""Steer only on a live face. A stale hold freezes PTZ (no ghost pan)."""
if confirmed is not None:
return "steer"
if held is not None:
return "freeze"
return "lost"
class Follower:
"""Keep a confirmed YuNet face near frame center. Hold on loss.
OpenCV 5 dropped Haar/HOG; YuNet FaceDetectorYN is the replacement.
Tiny / ceiling / stand / landmark-invalid hits are dropped before PTZ.
"""
def __init__(self, ptz: RallyV4L2, width: int, height: int,
deadband: float = 0.14, detect_every: int = 5):
self.ptz = ptz
self.w = int(width)
self.h = int(height)
self.deadband = float(deadband)
self.detect_every = max(1, int(detect_every))
self._n = 0
self._lost = 0
self._det = None
self._cv2 = None
self._gate = FaceGate()
self._hold = HeadHold()
self._frozen = False
self._gray: np.ndarray | None = None
self._prev: np.ndarray | None = None
self._dw = max(160, self.w // 2)
self._dh = max(90, self.h // 2)
model = resolve_yunet()
try:
import cv2
self._cv2 = cv2
if not model.is_file():
raise FileNotFoundError(model)
self._det = cv2.FaceDetectorYN_create(
str(model), "", (self._dw, self._dh), _SCORE_MIN, 0.3, 5000)
log.info("yunet face detector %s input=%dx%d score>=%.2f",
model.name, self._dw, self._dh, _SCORE_MIN)
except Exception as exc: # noqa: BLE001
log.warning("opencv tracker unavailable: %s", exc)
def _boxes(self, payload: bytes) -> list[tuple[int, int, int, int]]:
if self._cv2 is None or self._det is None:
return []
yuv = np.frombuffer(payload, dtype=np.uint8).reshape((self.h * 3 // 2, self.w))
bgr = self._cv2.cvtColor(yuv, self._cv2.COLOR_YUV2BGR_I420)
small = self._cv2.resize(bgr, (self._dw, self._dh))
self._gray = self._cv2.cvtColor(small, self._cv2.COLOR_BGR2GRAY)
self._det.setInputSize((self._dw, self._dh))
_ok, faces = self._det.detect(small)
picked = select_yunet_face(faces, self._dw, self._dh)
if picked is None:
return []
sx = self.w / float(self._dw)
sy = self.h / float(self._dh)
x, yy, ww, hh = (float(picked[0]), float(picked[1]),
float(picked[2]), float(picked[3]))
return [(int(x * sx), int(yy * sy), int(ww * sx), int(hh * sy))]
def on_i420(self, payload: bytes) -> None:
self._n += 1
if self._n % self.detect_every != 0:
return
boxes = self._boxes(payload)
now = time.monotonic()
hold_box = self._hold.box
if hold_box is not None and self._gray is not None:
sx = self._dw / float(self.w)
sy = self._dh / float(self.h)
small_box = (
int(hold_box[0] * sx), int(hold_box[1] * sy),
int(hold_box[2] * sx), int(hold_box[3] * sy),
)
else:
small_box = None
moving = roi_motion(self._prev, self._gray, small_box)
self._prev = None if self._gray is None else self._gray.copy()
face = boxes[0] if boxes else None
confirmed = self._gate.update(face, self.w, self.h)
held = self._hold.update(confirmed, moving=moving, now=now)
action = lock_action(confirmed, held)
if action == "steer" and confirmed is not None:
self._lost = 0
self._frozen = False
pan, tilt = steer_speeds(confirmed, self.w, self.h, self.deadband)
self.ptz.set_speed(pan, tilt)
return
if action == "freeze":
self._lost = 0
if not self._frozen:
self.ptz.stop()
self._frozen = True
return
self._lost += 1
if self._lost == 1 or self._lost == 6:
self.ptz.stop()
self._frozen = True
if self._lost == 6:
log.info("target lost; hold")
def _pipe() -> tuple[int, int]:
r, w = os.pipe()
try:
import fcntl
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 1048576)
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 1048576)
except OSError:
pass
os.set_inheritable(w, True)
os.set_inheritable(r, False)
return r, w
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--device", default="/dev/rally")
ap.add_argument("--connector", default="",
help="optional local kmssink; empty = LiveKit hairpin only")
ap.add_argument("--width", type=int, default=1920)
ap.add_argument("--height", type=int, default=1080)
ap.add_argument("--fps", type=int, default=30)
ap.add_argument("--shm-dir", default="/run/livekit-cameras/raw")
ap.add_argument("--audio-shm-dir", default="/run/livekit-cameras/pcm")
ap.add_argument("--audio-card", type=int, default=-1)
ap.add_argument("--audio-rate", type=int, default=16000)
ap.add_argument("--hop-s", type=float, default=0.1)
ap.add_argument("--identity", default="rally")
ap.add_argument("--no-follow", action="store_true")
args = ap.parse_args(argv)
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [rally-follow] %(levelname)s %(message)s")
preview = pick_preview(args.connector) if args.connector else None
if preview is not None and preview.driver == "xe":
log.error("refusing to kmssink on Arc (%s); wall stays there", preview.name)
return 2
if preview is None:
log.info("preview via LiveKit hairpin (no local kmssink)")
else:
log.info("preview %s id=%s %s %s", preview.name, preview.connector_id,
preview.driver, preview.node)
need = frame_bytes(args.width, args.height)
os.makedirs(args.shm_dir, exist_ok=True)
video_w = LatestFrameWriter(
os.path.join(args.shm_dir, f"{args.identity}.i420"),
args.width, args.height)
hop_n = hop_bytes(args.audio_rate, 2, args.hop_s)
audio_w = None
audio_r = audio_wr = None
audio_proc = None
audio_buf = bytearray()
if args.audio_card >= 0:
os.makedirs(args.audio_shm_dir, exist_ok=True)
audio_w = HopWriter(
os.path.join(args.audio_shm_dir, f"{args.identity}.pcm"), hop_n)
audio_r, audio_wr = _pipe()
vr, vw = _pipe()
vcmd = video_cmd(args.device, args.width, args.height, args.fps, vw, preview)
log.info("gst video %s", " ".join(vcmd))
vproc = subprocess.Popen(
vcmd, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL,
stderr=None, pass_fds=(vw,))
os.close(vw)
if audio_r is not None and audio_wr is not None:
acmd = audio_cmd(args.audio_card, args.audio_rate, audio_wr)
log.info("gst audio %s", " ".join(acmd))
audio_proc = subprocess.Popen(
acmd, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL,
stderr=None, pass_fds=(audio_wr,))
os.close(audio_wr)
os.set_blocking(audio_r, False)
os.set_blocking(vr, False)
follower = None if args.no_follow else Follower(
RallyV4L2(args.device), args.width, args.height)
stop = False
def _stop(_signum=None, _frame=None):
nonlocal stop
stop = True
signal.signal(signal.SIGTERM, _stop)
signal.signal(signal.SIGINT, _stop)
vbuf = bytearray()
seq = 0
last = time.monotonic()
try:
while not stop:
if vproc.poll() is not None:
log.error("video gst exited rc=%s", vproc.returncode)
break
if audio_proc is not None and audio_proc.poll() is not None:
log.error("audio gst exited rc=%s", audio_proc.returncode)
break
fds = [vr]
if audio_r is not None:
fds.append(audio_r)
ready, _, _ = select.select(fds, [], [], 0.2)
if vr in ready:
while True:
try:
chunk = os.read(vr, max(need, need - len(vbuf)))
except BlockingIOError:
break
if not chunk:
break
vbuf.extend(chunk)
while len(vbuf) >= need:
frame = bytes(vbuf[:need])
del vbuf[:need]
seq = video_w.write(frame)
if follower is not None:
follower.on_i420(frame)
if audio_r is not None and audio_r in ready and audio_w is not None:
while True:
try:
chunk = os.read(audio_r, max(hop_n, hop_n - len(audio_buf)))
except BlockingIOError:
break
if not chunk:
break
audio_buf.extend(chunk)
while len(audio_buf) >= hop_n:
hop = bytes(audio_buf[:hop_n])
del audio_buf[:hop_n]
audio_w.write(hop)
now = time.monotonic()
if now - last >= 5.0:
last = now
log.info("video seq=%d preview=%s follow=%s",
seq, preview.name if preview else "livekit",
follower is not None)
finally:
if follower is not None:
follower.ptz.stop()
for proc in (vproc, audio_proc):
if proc is None or proc.poll() is not None:
continue
proc.terminate()
try:
proc.wait(timeout=5)
except subprocess.TimeoutExpired:
proc.kill()
video_w.close()
if audio_w is not None:
audio_w.close()
for fd in (vr, audio_r):
if fd is None:
continue
try:
os.close(fd)
except OSError:
pass
return 0
if __name__ == "__main__":
sys.exit(main())
+435 -35
View File
@@ -21,19 +21,183 @@ import time
from pathlib import Path
from config import load_config, _load_dotenv
from discovery import discover_cameras, select_cameras
from discovery import Camera, discover_cameras, select_cameras
from latency import effective_enhance_mode
from participant_tags import token_attributes
from rally import discover_rally
from tokens import make_token
log = logging.getLogger("cameras.main")
def worker_args_for(cfg, cam) -> list[str]:
return [
"--camera-json", json.dumps(cam.__dict__),
"--room", cfg.room,
def spawn_enhance_daemon(cfg) -> subprocess.Popen:
from enhance_daemon import wait_for_socket
cmd = [
sys.executable, str(Path(__file__).parent / "enhance_daemon.py"),
"--socket", cfg.enhance_socket,
"--enhance-mode", "off" if cfg.enhance_mode == "off" else "force",
"--enhance-model", cfg.enhance_model,
"--enhance-chunk-s", str(cfg.enhance_chunk_s),
"--enhance-hop-s", str(cfg.enhance_hop_s),
"--enhance-infer", cfg.enhance_infer,
"--beamform-mode", cfg.beamform_mode,
"--torch-num-threads", str(max(2, cfg.torch_num_threads)),
"--tdoa-every", str(cfg.beamform_tdoa_every),
"--audio-rate", str(cfg.audio_rate),
]
log.info("starting shared enhance daemon at %s", cfg.enhance_socket)
proc = subprocess.Popen(cmd)
if not wait_for_socket(cfg.enhance_socket, timeout=90):
proc.kill()
raise RuntimeError(
f"enhance daemon did not listen on {cfg.enhance_socket}")
log.info("enhance daemon ready pid=%s", proc.pid)
return proc
def spawn_capture_daemon(cfg, cameras) -> subprocess.Popen:
from frame_shm import wait_for_ready
payload = [{"identity": c.identity, "video_device": c.video_device}
for c in cameras]
cmd = [
sys.executable, str(Path(__file__).parent / "capture_daemon.py"),
"--cameras-json", json.dumps(payload),
"--width", str(cfg.width),
"--height", str(cfg.height),
"--fps", str(cfg.fps),
"--shm-dir", cfg.capture_shm_dir,
"--vaapi-device", cfg.vaapi_device,
]
log.info("starting shared capture daemon shm=%s cams=%d",
cfg.capture_shm_dir, len(payload))
env = dict(os.environ)
env["GST_REGISTRY_UPDATE"] = "no"
env.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
proc = subprocess.Popen(cmd, env=env)
ready = str(Path(cfg.capture_shm_dir) / "ready")
if not wait_for_ready(ready, timeout=30):
proc.kill()
raise RuntimeError(f"capture daemon did not become ready at {ready}")
log.info("capture daemon ready pid=%s", proc.pid)
return proc
def spawn_audio_daemon(cfg, cameras) -> subprocess.Popen:
from frame_shm import wait_for_ready
payload = [{"identity": c.identity, "audio_card": c.audio_card}
for c in cameras if c.audio_card is not None]
if not payload:
raise RuntimeError("no cameras with audio cards")
cmd = [
sys.executable, str(Path(__file__).parent / "audio_daemon.py"),
"--cameras-json", json.dumps(payload),
"--rate", str(cfg.audio_rate),
"--hop-s", str(cfg.enhance_hop_s),
"--shm-dir", cfg.audio_shm_dir,
]
log.info("starting shared audio daemon shm=%s mics=%d",
cfg.audio_shm_dir, len(payload))
proc = subprocess.Popen(cmd)
ready = str(Path(cfg.audio_shm_dir) / "ready")
if not wait_for_ready(ready, timeout=30):
proc.kill()
raise RuntimeError(f"audio daemon did not become ready at {ready}")
log.info("audio daemon ready pid=%s", proc.pid)
return proc
def spawn_encode_daemon(cfg, cameras, portrait_dir: str = "") -> subprocess.Popen:
from frame_shm import wait_for_ready
payload = []
for c in cameras:
payload.append({
"identity": c.identity,
"width": int(c.width or cfg.width),
"height": int(c.height or cfg.height),
})
cmd = [
sys.executable, str(Path(__file__).parent / "encode_daemon.py"),
"--cameras-json", json.dumps(payload),
"--i420-dir", cfg.capture_shm_dir,
"--h264-dir", cfg.encode_shm_dir,
"--fps", str(cfg.fps),
"--bitrate", str(cfg.video_bitrate),
"--hi-bitrate", str(cfg.rally_video_bitrate),
"--vaapi-device", cfg.vaapi_device,
"--codec", cfg.video_codec,
]
if portrait_dir:
cmd.extend(["--portrait-dir", portrait_dir])
log.info("starting encode daemon h264=%s cams=%d codec=%s portrait=%s",
cfg.encode_shm_dir, len(payload), cfg.video_codec,
portrait_dir or "off")
env = dict(os.environ)
env["GST_REGISTRY_UPDATE"] = "no"
env.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
proc = subprocess.Popen(cmd, env=env)
ready = str(Path(cfg.encode_shm_dir) / "ready")
if not wait_for_ready(ready, timeout=60):
proc.kill()
raise RuntimeError(f"encode daemon did not become ready at {ready}")
log.info("encode daemon ready pid=%s", proc.pid)
return proc
def spawn_portrait_daemon(cfg, cameras) -> subprocess.Popen:
from frame_shm import wait_for_ready
payload = []
for c in cameras:
if c.identity == "rally":
continue
payload.append({
"identity": c.identity,
"width": int(c.width or cfg.width),
"height": int(c.height or cfg.height),
})
cmd = [
sys.executable, str(Path(__file__).parent / "portrait_daemon.py"),
"--cameras-json", json.dumps(payload),
"--in-dir", cfg.capture_shm_dir,
"--out-dir", cfg.portrait_shm_dir,
"--hz", str(cfg.portrait_hz),
"--blur-px", str(cfg.portrait_blur_px),
"--hold-s", str(cfg.portrait_hold_s),
]
log.info("starting portrait daemon shm=%s cams=%d hz=%.1f",
cfg.portrait_shm_dir, len(payload), cfg.portrait_hz)
env = dict(os.environ)
env["OMP_NUM_THREADS"] = "1"
env["OPENBLAS_NUM_THREADS"] = "1"
env["MKL_NUM_THREADS"] = "1"
proc = subprocess.Popen(cmd, env=env)
ready = str(Path(cfg.portrait_shm_dir) / "ready")
if not wait_for_ready(ready, timeout=45):
proc.kill()
raise RuntimeError(f"portrait daemon did not become ready at {ready}")
log.info("portrait daemon ready pid=%s", proc.pid)
return proc
def publisher_args_for(cfg, cameras) -> list[str]:
payload = []
for cam in cameras:
d = dict(cam.__dict__)
d["token"] = make_token(
cfg.url, cfg.api_key, cfg.api_secret,
cfg.room, cam.identity,
attributes=token_attributes(cfg.participant_tags))
payload.append(d)
args = [
"--cameras-json", json.dumps(payload),
"--url", cfg.url,
"--token", make_token(cfg.url, cfg.api_key, cfg.api_secret,
cfg.room, cam.identity),
"--room", cfg.room,
"--width", str(cfg.width),
"--height", str(cfg.height),
"--fps", str(cfg.fps),
@@ -50,7 +214,128 @@ def worker_args_for(cfg, cam) -> list[str]:
"--beamform-mode", cfg.beamform_mode,
"--torch-num-threads", str(cfg.torch_num_threads),
"--tdoa-every", str(cfg.beamform_tdoa_every),
"--video-shm-dir", cfg.capture_shm_dir,
"--audio-shm-dir", cfg.audio_shm_dir,
"--h264-shm-dir", cfg.encode_shm_dir,
"--speaker-camera", cfg.display_speaker_camera,
"--audio-gate-open-db", str(cfg.audio_gate_open_db),
"--audio-gate-close-db", str(cfg.audio_gate_close_db),
"--audio-gate-hold-s", str(cfg.audio_gate_hold_s),
]
if cfg.audio_queue_ms is not None:
args.extend(["--audio-queue-ms", str(cfg.audio_queue_ms)])
if cfg.enhance_daemon and cfg.enhance_socket:
args.extend(["--enhance-socket", cfg.enhance_socket])
if cfg.enhance_speaker_only:
args.append("--enhance-speaker-only")
if cfg.audio_gate:
args.append("--audio-gate")
return args
def fleet_cameras(cameras: list[Camera]) -> list[Camera]:
return [c for c in cameras if c.identity != "rally"]
def spawn_rally_follow(cfg, cam: Camera) -> subprocess.Popen:
cmd = [
sys.executable, str(Path(__file__).parent / "rally_follow.py"),
"--device", cam.video_device,
"--width", str(cam.width or 1920),
"--height", str(cam.height or 1080),
"--fps", "30",
"--shm-dir", cfg.capture_shm_dir,
"--audio-shm-dir", cfg.audio_shm_dir,
"--identity", cam.identity,
"--audio-rate", str(cfg.audio_rate),
"--hop-s", str(cfg.enhance_hop_s),
]
if cam.audio_card is not None:
cmd.extend(["--audio-card", str(cam.audio_card)])
log.info("starting rally follow video=%s audio=%s (preview=LiveKit hairpin)",
cam.video_device,
f"hw:{cam.audio_card},0" if cam.audio_card is not None else "none")
return subprocess.Popen(cmd)
def spawn_publisher(cfg, cameras) -> subprocess.Popen:
args = publisher_args_for(cfg, cameras)
env = dict(os.environ)
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
env.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
log.info("starting single publisher cams=%d", len(cameras))
return subprocess.Popen(
[sys.executable, str(Path(__file__).parent / "run_publisher.py"), *args],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, bufsize=1, env=env,
)
def stop_proc(proc: subprocess.Popen | None) -> None:
if proc is None or proc.poll() is not None:
return
proc.send_signal(signal.SIGTERM)
with contextlib.suppress(Exception):
proc.wait(timeout=10)
if proc.poll() is None:
proc.kill()
def worker_args_for(cfg, cam) -> list[str]:
enhance_mode = effective_enhance_mode(
cfg.enhance_mode,
cam.identity,
cfg.enhance_speaker_only,
cfg.display_speaker_camera,
)
beamform_mode = cfg.beamform_mode
if cfg.enhance_speaker_only and enhance_mode == "off":
beamform_mode = "off"
args = [
"--camera-json", json.dumps(cam.__dict__),
"--room", cfg.room,
"--url", cfg.url,
"--token", make_token(
cfg.url, cfg.api_key, cfg.api_secret,
cfg.room, cam.identity,
attributes=token_attributes(cfg.participant_tags)),
"--width", str(cfg.width),
"--height", str(cfg.height),
"--fps", str(cfg.fps),
"--bitrate", str(cfg.video_bitrate),
"--video-codec", cfg.video_codec,
"--video-encoder", cfg.video_encoder,
"--vaapi-device", cfg.vaapi_device,
"--audio-rate", str(cfg.audio_rate),
"--enhance-mode", enhance_mode,
"--enhance-model", cfg.enhance_model,
"--enhance-chunk-s", str(cfg.enhance_chunk_s),
"--enhance-hop-s", str(cfg.enhance_hop_s),
"--enhance-infer", cfg.enhance_infer,
"--beamform-mode", beamform_mode,
"--torch-num-threads", str(cfg.torch_num_threads),
"--tdoa-every", str(cfg.beamform_tdoa_every),
"--speaker-camera", cfg.display_speaker_camera,
"--audio-gate-open-db", str(cfg.audio_gate_open_db),
"--audio-gate-close-db", str(cfg.audio_gate_close_db),
"--audio-gate-hold-s", str(cfg.audio_gate_hold_s),
]
if cfg.audio_queue_ms is not None:
args.extend(["--audio-queue-ms", str(cfg.audio_queue_ms)])
if cfg.video_hold_s is not None:
args.extend(["--video-hold-s", str(cfg.video_hold_s)])
if cfg.worker_split_executor:
args.append("--split-executor")
if cfg.enhance_daemon and cfg.enhance_socket:
args.extend(["--enhance-socket", cfg.enhance_socket])
if cfg.audio_gate:
args.append("--audio-gate")
if cfg.capture_daemon:
args.extend([
"--video-shm",
f"{cfg.capture_shm_dir.rstrip('/')}/{cam.identity}.i420",
])
return args
class Supervisor:
@@ -128,6 +413,10 @@ def main() -> int:
log.error("LIVEKIT_URL / LIVEKIT_API_KEY / LIVEKIT_API_SECRET "
"not set (create .env from .env.example)")
return 2
if not cfg.url.startswith("wss://"):
log.error("LIVEKIT_URL must be wss:// (refusing %s)", cfg.url)
return 2
log.info("livekit destination url=%s room=%s", cfg.url, cfg.room)
cameras = discover_cameras()
log.info("discovered %d cameras:", len(cameras))
@@ -138,41 +427,101 @@ def main() -> int:
c.usb_port or "-")
cameras = select_cameras(cameras, cfg.cameras)
if cfg.cameras == ["all"]:
rally = discover_rally()
if rally is not None:
cameras = list(cameras) + [rally]
log.info(" %s video=%s audio=hw:%s usb=%s",
rally.identity, rally.video_device,
rally.audio_card if rally.audio_card is not None else "-",
rally.usb_port or "-")
if not cameras:
log.error("no cameras match --cameras %s", args.cameras)
return 2
fleet = fleet_cameras(cameras)
daemon = None
capture = None
audio = None
encode = None
rally_proc = None
portrait = None
if cfg.enhance_daemon and cfg.enhance_mode != "off":
try:
daemon = spawn_enhance_daemon(cfg)
except Exception as exc:
log.error("enhance daemon failed: %s", exc)
return 2
if cfg.capture_daemon:
try:
capture = spawn_capture_daemon(cfg, fleet)
except Exception as exc:
log.error("capture daemon failed: %s", exc)
stop_proc(daemon)
return 2
if cfg.audio_daemon:
try:
audio = spawn_audio_daemon(cfg, fleet)
except Exception as exc:
log.error("audio daemon failed: %s", exc)
stop_proc(capture)
stop_proc(daemon)
return 2
rally_cam = next((c for c in cameras if c.identity == "rally"), None)
if rally_cam is not None:
try:
rally_proc = spawn_rally_follow(cfg, rally_cam)
shm = Path(cfg.capture_shm_dir) / "rally.i420"
from frame_shm import wait_for_ready
if not wait_for_ready(str(shm), timeout=20):
raise RuntimeError(f"rally follow did not create {shm}")
except Exception as exc:
log.error("rally follow failed: %s", exc)
stop_proc(rally_proc)
stop_proc(audio)
stop_proc(capture)
stop_proc(daemon)
return 2
portrait_dir = ""
if cfg.portrait_blur and cfg.capture_daemon:
try:
portrait = spawn_portrait_daemon(cfg, fleet)
portrait_dir = cfg.portrait_shm_dir
except Exception as exc:
log.error("portrait daemon failed, encode uses raw: %s", exc)
stop_proc(portrait)
portrait = None
portrait_dir = ""
if cfg.encode_daemon and cfg.capture_daemon:
try:
encode = spawn_encode_daemon(cfg, cameras, portrait_dir=portrait_dir)
except Exception as exc:
log.error("encode daemon failed: %s", exc)
stop_proc(portrait)
stop_proc(rally_proc)
stop_proc(audio)
stop_proc(capture)
stop_proc(daemon)
return 2
use_pub = cfg.single_publisher and cfg.capture_daemon
sup = Supervisor(cfg, cameras)
for cam in cameras:
sup.spawn(cam)
log.info("spawned %d workers", len(cameras))
pub = None
if use_pub:
pub = spawn_publisher(cfg, cameras)
log.info("spawned single publisher for %d cameras", len(cameras))
else:
for cam in cameras:
sup.spawn(cam)
log.info("spawned %d workers", len(cameras))
if args.once:
deadline = time.monotonic() + cfg.publish_timeout_s
while time.monotonic() < deadline:
sup.pump()
sup.reap()
time.sleep(0.5)
# final report
up = [i for i, p in sup.procs.items() if p.poll() is None]
log.info("STATUS: %d/%d workers alive", len(up), len(sup.procs))
for ident in up:
pass
# stop everything
for p in sup.procs.values():
p.send_signal(signal.SIGTERM)
for p in sup.procs.values():
def _shutdown():
if pub is not None and pub.poll() is None:
pub.send_signal(signal.SIGTERM)
with contextlib.suppress(Exception):
p.wait(timeout=10)
return 0 if len(up) == len(sup.procs) else 1
try:
while True:
sup.pump()
sup.reap()
time.sleep(0.2)
except KeyboardInterrupt:
log.info("shutting down ...")
pub.wait(timeout=10)
if pub.poll() is None:
pub.kill()
for p in sup.procs.values():
if p.poll() is None:
p.send_signal(signal.SIGTERM)
@@ -182,6 +531,57 @@ def main() -> int:
for p in sup.procs.values():
if p.poll() is None:
p.kill()
stop_proc(daemon)
stop_proc(capture)
stop_proc(audio)
stop_proc(encode)
stop_proc(portrait)
stop_proc(rally_proc)
def _pump_pub():
nonlocal pub
if pub is None:
return
line = pub.stdout.readline() if pub.stdout else ""
if line:
sys.stdout.write(line)
sys.stdout.flush()
rc = pub.poll()
if rc is not None:
log.warning("publisher exited rc=%s; restarting", rc)
time.sleep(1.0)
pub = spawn_publisher(cfg, cameras)
if args.once:
deadline = time.monotonic() + cfg.publish_timeout_s
while time.monotonic() < deadline:
if use_pub:
_pump_pub()
else:
sup.pump()
sup.reap()
time.sleep(0.5)
if use_pub:
alive = pub is not None and pub.poll() is None
log.info("STATUS: publisher %s", "UP" if alive else "DOWN")
_shutdown()
return 0 if alive else 1
up = [i for i, p in sup.procs.items() if p.poll() is None]
log.info("STATUS: %d/%d workers alive", len(up), len(sup.procs))
_shutdown()
return 0 if len(up) == len(sup.procs) else 1
try:
while True:
if use_pub:
_pump_pub()
else:
sup.pump()
sup.reap()
time.sleep(0.2)
except KeyboardInterrupt:
log.info("shutting down ...")
_shutdown()
return 0
+454 -54
View File
@@ -14,9 +14,11 @@ from __future__ import annotations
import argparse
import asyncio
import logging
import os
import signal
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from typing import Optional
from urllib.parse import urlparse
@@ -26,19 +28,37 @@ from config import load_config
from display_grid import (
CameraGrid,
ROLES,
active_speaker_ids,
bgra_from_livekit,
camera_ids,
is_camera_identity,
collapse_bleed,
confirm_speakers,
describe_livekit_error,
format_livekit_status,
hold_speakers,
placeholder,
select_grid_speakers,
update_noise_floor,
)
from drm_outputs import DrmOutput, format_outputs, list_outputs, select_outputs
from gst_sink import KmsPipeline
from gst_sink import KmsPipeline, VaGridPipeline
from hand_raise import HandRaiseMonitor
from latency import video_stream_kwargs
from participant_tags import want_display_video
from tokens import make_viewer_token
log = logging.getLogger("cameras.display")
H264_DIR = "/run/livekit-cameras/h264"
def _h264_fifo_paths(identities: list[str]) -> list[str] | None:
# Sidecar H.264 encode removed: grid tiles come from LiveKit I420.
del identities
return None
_PATTERNS = ("smpte", "ball", "snow")
_PLACE_W, _PLACE_H = 1280, 720
_PLACE_W, _PLACE_H = 1920, 1080
def _public_url(cfg) -> str:
@@ -71,6 +91,69 @@ def _bind_roles(roles: list[str], connectors: list[str]) -> list[tuple[str, DrmO
return slots
class HopLevels:
"""Latest hop RMS + idle floor per grid identity (does not mute publish)."""
def __init__(self, shm_dir: str, identities: list[str], rise_db: float = 6.0) -> None:
from hop_shm import HopReader
from audio_gate import hop_mono_int16, rms_dbfs
self._mono = hop_mono_int16
self._rms = rms_dbfs
self.rise_db = float(rise_db)
self._db = {i: -120.0 for i in identities}
self._wave: dict[str, np.ndarray | None] = {i: None for i in identities}
self._floor: dict[str, float | None] = {i: None for i in identities}
self._n = {i: 0 for i in identities}
self._readers = {
i: HopReader(os.path.join(shm_dir, f"{i}.pcm")) for i in identities
}
def poll(self, speaking: set[str] | None = None) -> None:
hot = speaking or set()
for ident, r in self._readers.items():
blob = None
while True:
nxt = r.read()
if nxt is None:
break
blob = nxt
if blob is None:
continue
pcm = np.frombuffer(blob, dtype=np.int16)
if pcm.size % 2 == 0:
pcm = pcm.reshape(-1, 2)
db = self._rms(self._mono(pcm))
self._db[ident] = db
self._wave[ident] = np.asarray(self._mono(pcm), dtype=np.float32)
fl = self._floor[ident]
self._n[ident] += 1
if fl is None:
self._floor[ident] = db
else:
self._floor[ident] = update_noise_floor(
fl, db, speaking=ident in hot, rise_db=self.rise_db)
def levels(self, identities: list[str]) -> list[tuple[str, float]]:
return [
(i, self._db.get(i, -120.0))
for i in identities
if i in self._readers and self._n.get(i, 0) >= 3
]
def waves(self, identities: list[str]) -> dict[str, np.ndarray]:
out: dict[str, np.ndarray] = {}
for i in identities:
w = self._wave.get(i)
if w is not None:
out[i] = w
return out
@property
def floors(self) -> dict[str, float]:
return {i: fl for i, fl in self._floor.items() if fl is not None}
class Wall:
def __init__(self, cfg, slots: list[tuple[str, DrmOutput]]):
self.cfg = cfg
@@ -82,13 +165,23 @@ class Wall:
height=cfg.display_grid_height,
cols=cfg.display_grid_cols,
rows=cfg.display_grid_rows,
identities=camera_ids(n, prefix),
identities=[],
)
self.speaker_camera = (cfg.display_speaker_camera or "cam-01").strip()
self.grid.set_status(format_livekit_status(url=cfg.url, state="connecting"))
self._retry_s = 5.0
# LiveKit FFI retries ~3 times (~12s) on ENETUNREACH; stay above that
# so the overlay can show "host unreachable" instead of a bare timeout.
self._connect_timeout_s = 20.0
self.speaker_camera = (cfg.display_speaker_camera or "rally").strip()
self.active_speaker: Optional[str] = None
self._last_speakers: tuple[str, ...] = ()
self._speaker_frame: Optional[np.ndarray] = None
self._tasks: dict[str, asyncio.Task] = {}
self._stop = asyncio.Event()
self._pump_pool = ThreadPoolExecutor(
max_workers=max(1, int(cfg.display_pump_workers)),
thread_name_prefix="disp-pump",
)
# Screenshare is a stub until production LiveKit is online.
self._share_placeholder = placeholder(_PLACE_W, _PLACE_H, [
"Screen share",
@@ -98,22 +191,162 @@ class Wall:
cam = self.speaker_camera
self._speaker_placeholder = placeholder(_PLACE_W, _PLACE_H, [
"Speaker camera",
"Logitech Rally",
cam if cam.lower() != "active" else "(following active speaker)",
"not connected",
])
self.va_grid: Optional[VaGridPipeline] = None
self.hand_raise = HandRaiseMonitor(
enabled=cfg.display_hand_raise,
hz=cfg.display_hand_raise_hz,
hold_s=cfg.display_hand_raise_hold_s,
on_change=self.grid.set_highlight,
model=cfg.display_hand_raise_model,
)
self._hop_levels = HopLevels(
cfg.audio_shm_dir, self.grid.identities,
rise_db=cfg.display_speak_rise_db)
self._speak_cur: list[str] = []
self._speak_peak = -120.0
self._speak_hold_until = 0.0
self._speak_counts: dict[str, int] = {}
def _refresh_speakers(self) -> None:
"""Hop RMS is the detector; LiveKit is a hint, not a gate."""
tiles = list(self.grid.identities)
ranked = self._hop_levels.levels(tiles)
picked = select_grid_speakers(
ranked,
min_db=self.cfg.display_speak_min_db,
margin_db=self.cfg.display_speak_margin_db,
floors=self._hop_levels.floors,
rise_db=self.cfg.display_speak_rise_db,
)
picked = collapse_bleed(
picked,
self._hop_levels.waves(picked),
dict(ranked),
corr_min=self.cfg.display_speak_corr,
)
picked, self._speak_counts = confirm_speakers(
self._speak_counts, picked,
need=self.cfg.display_speak_confirm,
already=self._speak_cur,
)
new_peak = max((db for i, db in ranked if i in picked), default=-120.0)
now = time.monotonic()
chrome, self._speak_hold_until = hold_speakers(
self._speak_cur, picked,
prev_peak=self._speak_peak, new_peak=new_peak,
now=now, hold_until=self._speak_hold_until,
hold_s=self.cfg.display_speak_hold_s,
)
if chrome == picked:
self._speak_peak = new_peak
self._speak_cur = chrome
self.grid.set_speakers(chrome)
key = tuple(chrome)
if key != self._last_speakers:
self._last_speakers = key
log.info("active-speakers %s", ",".join(chrome) if chrome else "-")
def speaker_identity(self) -> Optional[str]:
if self.speaker_camera.lower() == "active":
return self.active_speaker
return self.speaker_camera
def _push_sink(self, role: str, data: bytes, width: int, height: int, fps: int) -> None:
def _set_lk_status(self, state: str, detail: str = "", retry_s: float | None = None) -> None:
lines = format_livekit_status(
url=self.cfg.url, state=state, detail=detail, retry_s=retry_s,
room=self.cfg.room)
if state == "connected":
if self.grid.identities:
self.grid.set_status(None)
else:
self.grid.set_status(["NO LIVE CAMERAS", f"room {self.cfg.room}"])
return
self.grid.set_status(lines)
def _seat_participant(self, participant) -> None:
ident = getattr(participant, "identity", "") or ""
if not ident or not self._want_video(participant):
return
if ident == self.speaker_identity():
return
if not self.grid.add_identity(ident):
log.warning("grid full, skip %s", ident)
return
if self.grid.identities:
self.grid.set_status(None)
def _unseat_participant(self, identity: str) -> None:
ident = (identity or "").strip()
if not ident:
return
self.grid.remove_identity(ident)
self.grid.clear(ident)
if ident == self.speaker_identity():
self._speaker_frame = None
# Keep mosaic if other tiles remain; otherwise show empty-room text
# only while the SFU session is still up (status overlay owns errors).
if not self.grid.identities and not self.grid._status_lines:
self.grid.set_status(["NO LIVE CAMERAS", f"room {self.cfg.room}"])
def _tile_index(self, identity: str) -> Optional[int]:
try:
return self.grid.identities.index(identity)
except ValueError:
return None
def _start_va_grid(self) -> None:
# vacompositor prerolls one HDMI frame then never flips (crtc CRC unique=1).
# CPU mosaic through appsrc+kmssink is the path that actually presents.
if os.environ.get("DISPLAY_VA_GRID", "0").strip() != "1":
log.info("va-grid skipped; CPU mosaic -> kmssink")
return
sink = self.slots.get("grid")
if sink is None or not sink.out.connected:
return
self.va_grid = VaGridPipeline(sink.out)
self.va_grid.start(
self.grid,
ingest_w=max(16, int(self.cfg.width)),
ingest_h=max(16, int(self.cfg.height)),
fps=max(1, int(self.cfg.fps) or 15),
queue_buffers=self.cfg.display_gst_queue_buffers,
h264_paths=_h264_fifo_paths(self.grid.identities),
)
if not self.va_grid.alive:
log.warning("va-grid failed to start; falling back to CPU mosaic")
self.va_grid = None
def _want_video(self, participant) -> bool:
return want_display_video(
identity=getattr(participant, "identity", "") or "",
attributes=getattr(participant, "attributes", None) or {},
metadata=getattr(participant, "metadata", "") or "",
hide_tags=self.cfg.display_hide_tags,
participant_prefix=self.cfg.participant_prefix or "cam",
display_identity=self.cfg.display_identity,
speaker_identity=self.speaker_identity() or "",
)
def _push_sink(self, role: str, data: bytes, width: int, height: int, fps: int,
pixel_format: str = "bgra") -> None:
sink = self.slots.get(role)
if sink is None or not sink.out.connected:
return
if not sink.alive or sink.width != width or sink.height != height:
sink.start_appsrc(width, height, fps)
if (not sink.alive or sink.width != width or sink.height != height
or getattr(sink, "pixel_format", "bgra") != pixel_format):
sink.start_appsrc(
width, height, fps,
queue_buffers=self.cfg.display_gst_queue_buffers,
pixel_format=pixel_format)
if not sink.push(data):
sink.start_appsrc(width, height, fps)
sink.start_appsrc(
width, height, fps,
queue_buffers=self.cfg.display_gst_queue_buffers,
pixel_format=pixel_format)
sink.push(data)
err = sink.gst_errors()
if err:
@@ -121,26 +354,56 @@ class Wall:
log.warning("%s gst: %s", role, line)
async def _output_loops(self) -> None:
self._start_va_grid()
grid_fps = max(1, int(self.cfg.display_grid_fps))
grid_period = 1.0 / grid_fps
spk_period = 1.0 / max(1, int(self.cfg.fps))
next_grid = time.monotonic()
next_spk = time.monotonic()
next_hop = time.monotonic()
while not self._stop.is_set():
now = time.monotonic()
if now >= next_hop:
self._hop_levels.poll(speaking=set(self._speak_cur))
self._refresh_speakers()
next_hop = now + 0.2
if now >= next_grid:
self._push_sink(
"grid", self.grid.render(),
self.grid.width, self.grid.height, grid_fps,
)
if self.va_grid is not None:
if not self.va_grid.alive:
log.warning("va-grid died; falling back to CPU mosaic")
self.va_grid.stop()
self.va_grid = None
else:
err = self.va_grid.gst_errors()
if err:
for line in err.strip().splitlines()[-6:]:
log.warning("va-grid gst: %s", line)
if self.va_grid is None:
i420 = self.cfg.display_video_format in ("i420", "yuv420p")
if i420:
self._push_sink(
"grid", self.grid.render_i420(),
self.grid.width, self.grid.height, grid_fps,
pixel_format="i420",
)
else:
self._push_sink(
"grid", self.grid.render(),
self.grid.width, self.grid.height, grid_fps,
)
next_grid = now + grid_period
if now >= next_spk:
spk = self._speaker_frame
arrival = self.cfg.display_speaker_push == "arrival"
if spk is None:
img = self._speaker_placeholder
else:
img = spk
self._push_sink("speaker", img.tobytes(), img.shape[1], img.shape[0], self.cfg.fps)
self._push_sink(
"speaker", img.tobytes(), img.shape[1], img.shape[0],
self.cfg.fps)
elif not arrival:
self._push_sink(
"speaker", spk.tobytes(), spk.shape[1], spk.shape[0],
self.cfg.fps)
share = self._share_placeholder
self._push_sink("screenshare", share.tobytes(), share.shape[1], share.shape[0], self.cfg.fps)
next_spk = now + spk_period
@@ -149,23 +412,60 @@ class Wall:
async def _pump_camera(self, track, identity: str) -> None:
import livekit.rtc as rtc
stream = rtc.VideoStream(track)
stream_kw = video_stream_kwargs(
self.cfg.display_video_capacity, self.cfg.display_video_format)
if stream_kw.get("format") == "bgra":
stream_kw["format"] = rtc.VideoBufferType.BGRA
elif stream_kw.get("format") in ("i420", "yuv420p"):
stream_kw["format"] = rtc.VideoBufferType.I420
stream = rtc.VideoStream(track, **stream_kw)
want_bgra = self.cfg.display_video_format == "bgra"
use_i420 = self.cfg.display_video_format in ("i420", "yuv420p")
loop = asyncio.get_running_loop()
try:
async for event in stream:
frame = event.frame
if frame.width <= 0 or frame.height <= 0:
continue
try:
converted = frame.convert(rtc.VideoBufferType.BGRA)
except Exception as exc:
log.warning("%s: convert failed: %s", identity, exc)
if (self.va_grid is not None
and getattr(self.va_grid, "codec", "") == "i420"):
idx = self._tile_index(identity)
if (idx is not None
and frame.width == self.va_grid.ingest_w
and frame.height == self.va_grid.ingest_h):
self.va_grid.push_tile(idx, frame.data)
is_spk = identity == self.speaker_identity()
cpu_grid = self.va_grid is None
if cpu_grid and use_i420 and not is_spk:
blob = bytes(frame.data)
await loop.run_in_executor(
self._pump_pool,
self.grid.set_frame_i420,
identity, blob, frame.width, frame.height)
self.hand_raise.offer(identity, blob, frame.width, frame.height)
continue
bgra = bgra_from_livekit(converted.data, converted.width, converted.height)
self.grid.set_frame(identity, bgra)
if identity == self.speaker_identity():
self._speaker_frame = bgra
if cpu_grid or is_spk:
try:
if frame.type == rtc.VideoBufferType.BGRA:
converted = frame
else:
converted = frame.convert(rtc.VideoBufferType.BGRA)
except Exception as exc:
log.warning("%s: convert failed: %s", identity, exc)
continue
bgra = bgra_from_livekit(
converted.data, converted.width, converted.height)
if cpu_grid:
self.grid.set_frame(identity, bgra)
if is_spk:
self._speaker_frame = bgra
if self.cfg.display_speaker_push == "arrival":
self._push_sink(
"speaker", bgra.tobytes(),
bgra.shape[1], bgra.shape[0], self.cfg.fps)
finally:
await stream.aclose()
self.hand_raise.forget(identity)
self.grid.clear(identity)
if identity == self.speaker_identity():
self._speaker_frame = None
@@ -175,27 +475,29 @@ class Wall:
def attach(self, track, publication, participant, rtc) -> None:
ident = participant.identity
if ident == self.cfg.display_identity:
if not self._want_video(participant):
return
is_spk = ident == self.speaker_identity()
if not is_spk:
self._seat_participant(participant)
if ident not in self.grid.identities:
return
key = self._track_key(ident, track)
old = self._tasks.pop(key, None)
if old:
old.cancel()
# Screenshare subscribe/render is stubbed until production LiveKit.
if is_camera_identity(ident, self.cfg.participant_prefix):
self._tasks[key] = asyncio.create_task(
self._pump_camera(track, ident), name=f"cam-{ident}")
self._tasks[key] = asyncio.create_task(
self._pump_camera(track, ident), name=f"cam-{ident}")
def subscribe_if_wanted(self, participant, rtc) -> None:
ident = participant.identity
if ident == self.cfg.display_identity:
if not self._want_video(participant):
return
cam = is_camera_identity(ident, self.cfg.participant_prefix)
for pub in participant.track_publications.values():
if pub.kind != rtc.TrackKind.KIND_VIDEO:
continue
want = cam
if want and not pub.subscribed:
if not pub.subscribed:
log.info("subscribe %s from %s (source=%s)", pub.sid, ident, pub.source)
pub.set_subscribed(True)
@@ -206,9 +508,69 @@ class Wall:
task.cancel()
async def run(self) -> int:
loops = asyncio.create_task(self._output_loops(), name="output-loops")
loop = asyncio.get_running_loop()
for sig in (signal.SIGINT, signal.SIGTERM):
try:
loop.add_signal_handler(sig, self._stop.set)
except NotImplementedError:
pass
self.hand_raise.start()
self._set_lk_status("connecting")
log.info(
"wall up (livekit optional): grid=%dx%d speaker_camera=%s "
"video_capacity=%d format=%s gst_queue=%d speaker_push=%s "
"pump_workers=%d hide_tags=%s hand_raise=%s url=%s",
self.cfg.display_grid_cols, self.cfg.display_grid_rows, self.speaker_camera,
self.cfg.display_video_capacity,
self.cfg.display_video_format or "native",
self.cfg.display_gst_queue_buffers,
self.cfg.display_speaker_push,
self.cfg.display_pump_workers,
",".join(self.cfg.display_hide_tags) or "off",
"on" if self.hand_raise.enabled else "off",
self.cfg.url,
)
try:
while not self._stop.is_set():
try:
await self._session()
except asyncio.CancelledError:
raise
except Exception as exc:
detail = describe_livekit_error(exc)
log.error("livekit %s: %s", detail, exc or type(exc).__name__)
self.grid.set_identities([])
self._speaker_frame = None
self._set_lk_status(
"disconnected", detail=detail, retry_s=self._retry_s)
if self._stop.is_set():
break
try:
await asyncio.wait_for(self._stop.wait(), timeout=self._retry_s)
except asyncio.TimeoutError:
pass
if not self._stop.is_set():
self._set_lk_status("connecting")
finally:
loops.cancel()
for t in self._tasks.values():
t.cancel()
self._tasks.clear()
for sink in self.slots.values():
sink.stop()
if self.va_grid is not None:
self.va_grid.stop()
self.va_grid = None
self._pump_pool.shutdown(wait=False, cancel_futures=True)
self.hand_raise.stop()
return 0
async def _session(self) -> None:
import livekit.rtc as rtc
room = rtc.Room()
session_done = asyncio.Event()
@room.on("track_subscribed")
def _on_sub(track, publication, participant):
@@ -227,48 +589,80 @@ class Wall:
@room.on("participant_connected")
def _on_pc(participant):
log.info("participant joined: %s", participant.identity)
self._seat_participant(participant)
self.subscribe_if_wanted(participant, rtc)
@room.on("participant_disconnected")
def _on_pd(participant):
ident = getattr(participant, "identity", "") or ""
log.info("participant left: %s", ident)
self._unseat_participant(ident)
@room.on("disconnected")
def _on_disc(*args):
reason = ""
if args:
reason = str(args[0] or "")
detail = describe_livekit_error(RuntimeError(reason)) if reason else "server closed the session"
if reason:
low = reason.lower()
if "disconnect" in low or "close" in low:
detail = reason[:96]
log.warning("livekit disconnected: %s", reason or "no reason")
self.grid.set_identities([])
self._speaker_frame = None
self._set_lk_status(
"disconnected", detail=detail, retry_s=self._retry_s)
session_done.set()
@room.on("active_speakers_changed")
def _on_spk(speakers):
cams = [
p.identity for p in speakers
if is_camera_identity(p.identity, self.cfg.participant_prefix)
]
cams = active_speaker_ids(speakers, want=self._want_video)
self.active_speaker = cams[0] if cams else None
self._hop_levels.poll(speaking=set(self._speak_cur))
self._refresh_speakers()
token = make_viewer_token(
self.cfg.api_key, self.cfg.api_secret, self.cfg.room, self.cfg.display_identity)
log.info("connecting to %s room=%s as %s", self.cfg.url, self.cfg.room, self.cfg.display_identity)
await room.connect(self.cfg.url, token, rtc.RoomOptions(auto_subscribe=False))
self._set_lk_status("connecting")
try:
await asyncio.wait_for(
room.connect(self.cfg.url, token, rtc.RoomOptions(auto_subscribe=False)),
timeout=self._connect_timeout_s,
)
except Exception:
try:
await room.disconnect()
except Exception:
pass
raise
self._set_lk_status("connected")
for rp in room.remote_participants.values():
self._seat_participant(rp)
self.subscribe_if_wanted(rp, rtc)
for pub in rp.track_publications.values():
if pub.track and pub.track.kind == rtc.TrackKind.KIND_VIDEO:
self.attach(pub.track, pub, rp, rtc)
log.info(
"wall up: grid=%dx%d speaker_camera=%s screenshare=STUB",
self.grid.cols, self.grid.rows, self.speaker_camera,
"livekit connected: tiles=%d speaker_camera=%s",
len(self.grid.identities), self.speaker_camera,
)
loops = asyncio.create_task(self._output_loops(), name="output-loops")
loop = asyncio.get_running_loop()
for sig in (signal.SIGINT, signal.SIGTERM):
try:
loop.add_signal_handler(sig, self._stop.set)
except NotImplementedError:
pass
stop_task = asyncio.create_task(self._stop.wait())
disc_task = asyncio.create_task(session_done.wait())
try:
await self._stop.wait()
await asyncio.wait({stop_task, disc_task}, return_when=asyncio.FIRST_COMPLETED)
finally:
loops.cancel()
stop_task.cancel()
disc_task.cancel()
for t in self._tasks.values():
t.cancel()
for sink in self.slots.values():
sink.stop()
await room.disconnect()
return 0
self._tasks.clear()
try:
await room.disconnect()
except Exception:
pass
def run_test_pattern(slots: list[tuple[str, DrmOutput]]) -> int:
@@ -326,6 +720,8 @@ def main(argv: Optional[list[str]] = None) -> int:
cfg = load_config()
logging.getLogger().setLevel(getattr(logging, cfg.log_level, logging.INFO))
if cfg.url or cfg.room:
log.info("livekit destination url=%s room=%s", cfg.url, cfg.room)
if args.speaker_camera:
cfg.display_speaker_camera = args.speaker_camera
roles = [p.strip() for p in args.roles.split(",") if p.strip()] or cfg.display_roles
@@ -341,6 +737,10 @@ def main(argv: Optional[list[str]] = None) -> int:
if args.test_pattern:
return run_test_pattern(slots)
if not (cfg.url or "").startswith("wss://"):
log.error("LIVEKIT_URL must be wss:// (refusing %s)", cfg.url)
return 2
log.info("livekit destination url=%s room=%s", cfg.url, cfg.room)
return asyncio.run(Wall(cfg, slots).run())
+356
View File
@@ -0,0 +1,356 @@
"""One process: N LiveKit identities, shared video SHM + audio hop rings."""
from __future__ import annotations
import argparse
import asyncio
import json
import logging
import os
import signal
import sys
import time
import numpy as np
log = logging.getLogger("cameras.publisher")
def _rss_anon_kb() -> int:
try:
with open("/proc/self/status", encoding="utf-8") as f:
for line in f:
if line.startswith("RssAnon:"):
return int(line.split()[1])
except OSError:
return 0
return 0
def _pub_opts(rtc, args, height: int = 0):
from livekit.rtc import TrackPublishOptions, VideoEncoding, VideoCodec, TrackSource
from livekit.rtc._proto.room_pb2 import (
ENCODER_BACKEND_AUTO,
ENCODER_BACKEND_SOFTWARE,
ENCODER_BACKEND_HARDWARE,
ENCODER_BACKEND_NVENC,
ENCODER_BACKEND_VAAPI,
ENCODER_BACKEND_VIDEOTOOLBOX,
DEGRADATION_PREFERENCE_MAINTAIN_RESOLUTION,
)
encoder_map = {
"auto": ENCODER_BACKEND_AUTO,
"software": ENCODER_BACKEND_SOFTWARE,
"hardware": ENCODER_BACKEND_HARDWARE,
"nvenc": ENCODER_BACKEND_NVENC,
"vaapi": ENCODER_BACKEND_VAAPI,
"videotoolbox": ENCODER_BACKEND_VIDEOTOOLBOX,
}
codec_map = {
"vp8": VideoCodec.VP8,
"h264": VideoCodec.H264,
"av1": VideoCodec.AV1,
"vp9": VideoCodec.VP9,
"h265": VideoCodec.H265,
}
encoder = encoder_map.get(args.video_encoder, ENCODER_BACKEND_AUTO)
codec = codec_map.get(args.video_codec, VideoCodec.H264)
br = int(args.bitrate)
if int(height) >= 720:
br = max(br, int(os.environ.get("VIDEO_RALLY_BITRATE", "20000000")))
video = TrackPublishOptions(
source=TrackSource.SOURCE_CAMERA,
simulcast=False,
video_codec=codec,
video_encoding=VideoEncoding(
max_bitrate=br,
max_framerate=int(args.fps),
),
video_encoder=encoder,
degradation_preference=DEGRADATION_PREFERENCE_MAINTAIN_RESOLUTION,
)
audio = TrackPublishOptions(source=TrackSource.SOURCE_MICROPHONE, dtx=False)
return video, audio
async def publish_one(rtc, cam: dict, args, stop: asyncio.Event,
connect_gate: asyncio.Semaphore, enc_state: dict) -> None:
from audio_cleanup import AudioCleaner
from frame_shm import LatestFrameReader, frame_bytes
from hop_shm import HopReader
from audio_gate import make_gate
from latency import (
audio_queue_size_ms,
effective_enhance_mode,
next_video_capture,
should_recycle_encoders,
video_capture_gate_acquire,
video_capture_gate_release,
)
identity = cam["identity"]
clog = logging.getLogger(f"cameras.publisher.{identity}")
w = int(cam.get("width") or args.width)
h = int(cam.get("height") or args.height)
if args.video_codec in ("av1", "h264"):
h = h - (h % 16)
w = w - (w % 16)
enhance_mode = effective_enhance_mode(
args.enhance_mode, identity, args.enhance_speaker_only,
args.speaker_camera)
beamform = args.beamform_mode
if args.enhance_speaker_only and enhance_mode == "off":
beamform = "off"
cleaner = AudioCleaner(
mode=enhance_mode,
model_source=args.enhance_model,
sample_rate=args.audio_rate,
chunk_s=args.enhance_chunk_s,
hop_s=args.enhance_hop_s,
beamform=beamform,
torch_num_threads=args.torch_num_threads,
tdoa_every=args.tdoa_every,
enhance_infer=args.enhance_infer,
enhance_socket=args.enhance_socket,
identity=identity,
gate=make_gate(
identity,
args.speaker_camera,
bool(args.audio_gate),
args.audio_gate_open_db,
args.audio_gate_close_db,
args.audio_gate_hold_s,
args.enhance_hop_s,
),
)
clog.info("audio backend=%s enhance=%s beamform=%s gate=%s",
cleaner.backend, enhance_mode, beamform,
"on" if cleaner._gate is not None else "off")
room = rtc.Room()
nbytes = frame_bytes(w, h)
shm_path = os.path.join(args.video_shm_dir, f"{identity}.i420")
shm = LatestFrameReader(shm_path)
scratch = bytearray(nbytes)
video_frames = 0
video_gate = {"in_flight": 0, "dropped": 0, "max_in_flight": 3}
from encoded_pub import capture_encoded
from nal_shm import NalReader
if args.video_codec == "av1":
from av1_obu import is_keyframe
else:
from annexb import is_keyframe
nal = NalReader(os.path.join(args.h264_shm_dir, f"{identity}.h264"))
clog.info("video path=encoded-preencoded identity=%s codec=%s h264=%s",
identity, args.video_codec, nal.path)
audio_card = cam.get("audio_card")
hops = HopReader(os.path.join(args.audio_shm_dir, f"{identity}.pcm")) if audio_card is not None else None
atrack = None
audio_source = None
v_opts, a_opts = _pub_opts(rtc, args, h)
async with connect_gate:
await room.connect(args.url, cam["token"], rtc.RoomOptions(auto_subscribe=False))
clog.info("connected room=%r", room.name)
video_source = rtc.VideoSource(w, h)
video_track = rtc.LocalVideoTrack.create_video_track(
f"{identity}-video", video_source)
if hops is not None:
audio_source = rtc.AudioSource(
sample_rate=args.audio_rate, num_channels=1,
queue_size_ms=audio_queue_size_ms(
args.enhance_hop_s, args.audio_queue_ms))
atrack = rtc.LocalAudioTrack.create_audio_track(
f"{identity}-audio", audio_source)
await room.local_participant.publish_track(video_track, v_opts)
if atrack is not None:
await room.local_participant.publish_track(atrack, a_opts)
clog.info("status: UP identity=%s video_shm=%s audio=%s",
identity, shm_path,
f"hw:{audio_card},0" if audio_card is not None else "none")
async def recycle_video():
nonlocal video_source, video_track
sid = getattr(video_track, "sid", None) or ""
clog.warning("encoder recycle rss_anon_kb=%d sid=%s",
_rss_anon_kb(), sid)
if sid:
try:
await room.local_participant.unpublish_track(sid)
except Exception: # noqa: BLE001
clog.exception("unpublish for encoder recycle failed")
stagger = max(0, int(cam.get("index") or 0)) * 0.05
if stagger:
await asyncio.sleep(stagger)
video_source = rtc.VideoSource(w, h)
video_track = rtc.LocalVideoTrack.create_video_track(
f"{identity}-video", video_source)
await room.local_participant.publish_track(video_track, v_opts)
async def video_loop():
nonlocal video_frames, video_source, video_track
period = 1.0 / max(1, int(args.fps))
last_seq = -1
due = time.monotonic()
local_gen = enc_state["gen"]
loop = asyncio.get_running_loop()
while not stop.is_set():
now = time.monotonic()
capture, due = next_video_capture(now, due, period)
if capture:
au = nal.read()
if au is not None:
last_seq, payload = au
try:
rc = capture_encoded(
video_source,
payload,
w,
h,
is_keyframe(payload),
timestamp_us=int(time.monotonic() * 1_000_000),
codec=args.video_codec,
)
if rc == 0:
video_frames += 1
elif video_frames % 90 == 1:
clog.warning("capture_encoded rc=%s seq=%s", rc, last_seq)
except Exception: # noqa: BLE001
clog.exception("capture_encoded failed")
if args.encoder_rss_limit_kb and video_frames % 90 == 0:
rss = _rss_anon_kb()
async with enc_state["lock"]:
if should_recycle_encoders(
rss, args.encoder_rss_limit_kb, now,
enc_state["last"],
args.encoder_recycle_interval_s):
enc_state["last"] = now
enc_state["gen"] += 1
if local_gen < enc_state["gen"]:
await recycle_video()
local_gen = enc_state["gen"]
delay = due - time.monotonic()
if delay > 0:
await asyncio.sleep(delay)
async def audio_loop():
assert hops is not None and audio_source is not None
hop_n = int(round(args.enhance_hop_s * args.audio_rate))
n = 0
loop = asyncio.get_running_loop()
while not stop.is_set():
blob = hops.read()
if blob is None:
await asyncio.sleep(0.005)
continue
try:
pcm = np.frombuffer(blob, dtype=np.int16)
if pcm.size % 2 == 0:
pcm = pcm.reshape(-1, 2)
cleaned = await loop.run_in_executor(None, cleaner.process, pcm)
await audio_source.capture_frame(
rtc.AudioFrame(
cleaned.tobytes(), args.audio_rate, 1, len(cleaned)))
n += 1
if n % 50 == 0:
clog.info("audio hop process_ms=%.1f", cleaner.last_dt_ms)
except Exception: # noqa: BLE001
clog.exception("audio hop failed")
tasks = [asyncio.create_task(video_loop(), name=f"{identity}-video")]
if hops is not None:
tasks.append(asyncio.create_task(audio_loop(), name=f"{identity}-audio"))
try:
await stop.wait()
finally:
for t in tasks:
t.cancel()
await asyncio.gather(*tasks, return_exceptions=True)
clog.info("stopped; delivered %d video frames", video_frames)
try:
await room.disconnect()
except Exception: # noqa: BLE001
pass
async def amain(args) -> int:
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
os.environ.pop("LIBVA_DRM_DEVICE", None)
import livekit.rtc as rtc
cams = json.loads(args.cameras_json)
stop = asyncio.Event()
loop = asyncio.get_running_loop()
for sig in (signal.SIGTERM, signal.SIGINT):
try:
loop.add_signal_handler(sig, stop.set)
except NotImplementedError:
pass
log.info("publisher start cams=%d shm=%s pcm=%s speaker_only=%s "
"encoder_rss_limit_kb=%d",
len(cams), args.video_shm_dir, args.audio_shm_dir,
args.enhance_speaker_only, args.encoder_rss_limit_kb)
gate = asyncio.Semaphore(1)
enc_state = {"lock": asyncio.Lock(), "last": 0.0, "gen": 0}
tasks = [
asyncio.create_task(
publish_one(rtc, cam, args, stop, gate, enc_state),
name=cam["identity"])
for cam in cams
]
await stop.wait()
for t in tasks:
t.cancel()
await asyncio.gather(*tasks, return_exceptions=True)
return 0
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--cameras-json", required=True)
ap.add_argument("--url", required=True)
ap.add_argument("--room", required=True)
ap.add_argument("--width", type=int, default=640)
ap.add_argument("--height", type=int, default=360)
ap.add_argument("--fps", type=int, default=30)
ap.add_argument("--bitrate", type=int, default=1_200_000)
ap.add_argument("--video-codec", default="h264")
ap.add_argument("--video-encoder", default="vaapi")
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
ap.add_argument("--audio-rate", type=int, default=16000)
ap.add_argument("--enhance-mode", default="auto")
ap.add_argument("--enhance-model",
default="speechbrain/metricgan-plus-voicebank")
ap.add_argument("--enhance-chunk-s", type=float, default=1.0)
ap.add_argument("--enhance-hop-s", type=float, default=0.1)
ap.add_argument("--enhance-infer", default="auto")
ap.add_argument("--beamform-mode", default="auto")
ap.add_argument("--torch-num-threads", type=int, default=1)
ap.add_argument("--tdoa-every", type=int, default=5)
ap.add_argument("--audio-queue-ms", type=int, default=None)
ap.add_argument("--enhance-socket", default="")
ap.add_argument("--video-shm-dir", default="/run/livekit-cameras/raw")
ap.add_argument("--audio-shm-dir", default="/run/livekit-cameras/pcm")
ap.add_argument("--h264-shm-dir", default="/run/livekit-cameras/h264")
ap.add_argument("--enhance-speaker-only", action="store_true")
ap.add_argument("--speaker-camera", default="rally")
ap.add_argument("--audio-gate", action="store_true")
ap.add_argument("--audio-gate-open-db", type=float, default=-28.0)
ap.add_argument("--audio-gate-close-db", type=float, default=-34.0)
ap.add_argument("--audio-gate-hold-s", type=float, default=0.3)
ap.add_argument("--encoder-rss-limit-kb", type=int, default=0,
help="Recycle VAAPI VideoSource tracks when RssAnon exceeds this (0=off)")
ap.add_argument("--encoder-recycle-interval-s", type=float, default=120.0)
args = ap.parse_args()
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [%(name)s] %(levelname)s %(message)s")
try:
return asyncio.run(amain(args))
except Exception: # noqa: BLE001
log.exception("publisher fatal")
return 1
if __name__ == "__main__":
sys.exit(main())
+155 -67
View File
@@ -7,6 +7,7 @@ from __future__ import annotations
import argparse
import asyncio
import concurrent.futures
import contextlib
import json
import logging
@@ -35,10 +36,10 @@ def main() -> int:
ap.add_argument("--room", required=True)
ap.add_argument("--url", required=True)
ap.add_argument("--token", required=True)
ap.add_argument("--width", type=int, default=640)
ap.add_argument("--height", type=int, default=360)
ap.add_argument("--fps", type=int, default=30)
ap.add_argument("--bitrate", type=int, default=800_000)
ap.add_argument("--width", type=int, default=320)
ap.add_argument("--height", type=int, default=180)
ap.add_argument("--fps", type=int, default=15)
ap.add_argument("--bitrate", type=int, default=400_000)
ap.add_argument("--video-codec", default="h264")
ap.add_argument("--video-encoder", default="vaapi")
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
@@ -52,6 +53,17 @@ def main() -> int:
ap.add_argument("--beamform-mode", default="auto")
ap.add_argument("--torch-num-threads", type=int, default=1)
ap.add_argument("--tdoa-every", type=int, default=5)
ap.add_argument("--audio-queue-ms", type=int, default=None)
ap.add_argument("--video-hold-s", type=float, default=None)
ap.add_argument("--split-executor", action="store_true")
ap.add_argument("--enhance-socket", default="")
ap.add_argument("--speaker-camera", default="rally")
ap.add_argument("--audio-gate", action="store_true")
ap.add_argument("--audio-gate-open-db", type=float, default=-28.0)
ap.add_argument("--audio-gate-close-db", type=float, default=-34.0)
ap.add_argument("--audio-gate-hold-s", type=float, default=0.3)
ap.add_argument("--video-shm", default="",
help="latest-frame I420 mmap from capture_daemon")
args = ap.parse_args()
cam = json.loads(args.camera_json)
@@ -62,15 +74,16 @@ def main() -> int:
from config import Config # noqa: F401 (ensures .env is loaded)
from audio_cleanup import AudioCleaner, apply_torch_thread_limits
apply_torch_thread_limits(args.torch_num_threads)
import livekit.rtc as rtc
from audio_gate import make_gate
from latency import audio_queue_size_ms, video_hold_seconds
identity = cam["identity"]
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
os.environ.pop("LIBVA_DRM_DEVICE", None)
if not args.enhance_socket:
apply_torch_thread_limits(args.torch_num_threads)
import livekit.rtc as rtc
video_dev = cam["video_device"]
audio_card = cam.get("audio_card")
if args.vaapi_device:
os.environ.setdefault("LIBVA_DRM_DEVICE", args.vaapi_device)
os.environ.setdefault("LIBVA_DRIVER_NAME", "iHD")
async def run() -> int:
cleaner = AudioCleaner(
@@ -83,76 +96,145 @@ def main() -> int:
torch_num_threads=args.torch_num_threads,
tdoa_every=args.tdoa_every,
enhance_infer=args.enhance_infer,
enhance_socket=args.enhance_socket,
identity=identity,
gate=make_gate(
identity,
args.speaker_camera,
bool(args.audio_gate),
args.audio_gate_open_db,
args.audio_gate_close_db,
args.audio_gate_hold_s,
args.enhance_hop_s,
),
)
log.info("audio backend: %s", cleaner.backend)
log.info("audio backend: %s gate=%s", cleaner.backend,
"on" if cleaner._gate is not None else "off")
io_ex = None
enhance_ex = None
split_owned: list[concurrent.futures.Executor] = []
if args.split_executor:
io_ex = concurrent.futures.ThreadPoolExecutor(
max_workers=2, thread_name_prefix=f"{identity}-io")
enhance_ex = concurrent.futures.ThreadPoolExecutor(
max_workers=1, thread_name_prefix=f"{identity}-enh")
split_owned.extend([io_ex, enhance_ex])
log.info("split executor: ffmpeg io vs enhance")
room = rtc.Room()
await room.connect(args.url, args.token)
log.info("connected to room %r as %r", room.name, identity)
await room.connect(
args.url, args.token, rtc.RoomOptions(auto_subscribe=False))
log.info("connected to room %r as %r (auto_subscribe=False)",
room.name, identity)
# ---------------- video ----------------
video_source = rtc.VideoSource(args.width, args.height)
video_track = rtc.LocalVideoTrack.create_video_track(
f"{identity}-video", video_source)
vproc = subprocess.Popen(
["ffmpeg", "-hide_banner", "-loglevel", "error",
"-fflags", "nobuffer", "-flags", "low_delay",
"-thread_queue_size", "4",
"-f", "v4l2",
"-input_format", "mjpeg",
"-video_size", f"{args.width}x{args.height}",
"-framerate", str(args.fps),
"-i", video_dev,
"-f", "rawvideo", "-pix_fmt", "yuv420p",
"pipe:1"],
stdout=subprocess.PIPE,
)
frame_bytes = args.width * args.height * 3 // 2
video_frames = 0
# Audio is published in enhance hops (default 200 ms). Hold video by
# that same amount so lipsync at the subscriber: a frame captured at
# T is sent when the audio hop covering T is sent.
video_hold_s = float(args.enhance_hop_s) if audio_card is not None else 0.0
vproc: subprocess.Popen | None = None
shm_path = (args.video_shm or "").strip()
video_hold_s = video_hold_seconds(
args.enhance_hop_s,
audio_card is not None,
0.0 if shm_path else args.video_hold_s,
)
pending: deque[tuple[float, bytes]] = deque()
max_pending = max(8, int(args.fps * (video_hold_s + 0.15)) + 4)
if video_hold_s > 0:
log.info("av sync: holding video %.0f ms to match audio hop",
video_hold_s * 1000.0)
async def video_loop():
nonlocal video_frames
loop = asyncio.get_running_loop()
while vproc.poll() is None:
data = await loop.run_in_executor(
None, vproc.stdout.read, frame_bytes)
if not data or len(data) < frame_bytes:
if not data:
break
keep = b""
while len(keep) < frame_bytes:
more = await loop.run_in_executor(
None, vproc.stdout.read,
frame_bytes - len(keep))
if not more:
return
keep += more
data = keep
now = time.monotonic()
pending.append((now, data))
while len(pending) > max_pending:
pending.popleft()
for payload in drain_held_frames(pending, now, video_hold_s):
if shm_path:
from frame_shm import LatestFrameReader
shm_reader = LatestFrameReader(shm_path)
log.info("video from shared capture shm=%s hold=0 (latest frame)",
shm_path)
async def video_loop():
nonlocal video_frames
period = 1.0 / max(1, int(args.fps))
last_seq = -1
next_t = time.monotonic()
scratch = bytearray(frame_bytes)
while True:
seq = shm_reader.copy_into(scratch)
if seq is not None and seq != last_seq:
last_seq = seq
try:
video_source.capture_frame(
rtc.VideoFrame(
args.width, args.height,
rtc.VideoBufferType.I420, scratch),
timestamp_us=int(time.monotonic() * 1_000_000),
)
video_frames += 1
except Exception: # noqa: BLE001
log.exception("video capture_frame failed")
next_t += period
delay = next_t - time.monotonic()
if delay > 0:
await asyncio.sleep(delay)
else:
next_t = time.monotonic()
else:
from capture_va import video_capture_cmd
vproc = subprocess.Popen(
video_capture_cmd(
video_dev, args.width, args.height, args.fps, fd=1),
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
)
log.info("capture gst jpegdec Arc (no sidecar h264)")
if video_hold_s > 0:
log.info("av sync: holding video %.0f ms to match audio hop",
video_hold_s * 1000.0)
async def video_loop():
nonlocal video_frames
assert vproc is not None
loop = asyncio.get_running_loop()
stdout = vproc.stdout
assert stdout is not None
while vproc.poll() is None:
data = await loop.run_in_executor(
io_ex, stdout.read, frame_bytes)
if not data or len(data) < frame_bytes:
if not data:
break
keep = b""
while len(keep) < frame_bytes:
more = await loop.run_in_executor(
io_ex, stdout.read,
frame_bytes - len(keep))
if not more:
return
keep += more
data = keep
now = time.monotonic()
pending.append((now, data))
while len(pending) > max_pending:
pending.popleft()
for payload in drain_held_frames(pending, now, video_hold_s):
try:
video_source.capture_frame(
rtc.VideoFrame(
args.width, args.height,
rtc.VideoBufferType.I420, payload),
timestamp_us=int(time.monotonic() * 1_000_000),
)
video_frames += 1
except Exception: # noqa: BLE001
log.exception("video capture_frame failed")
err = b""
if vproc.stderr:
try:
video_source.capture_frame(
rtc.VideoFrame(
args.width, args.height,
rtc.VideoBufferType.I420, payload),
timestamp_us=int(time.monotonic() * 1_000_000),
)
video_frames += 1
err = vproc.stderr.read() or b""
except Exception: # noqa: BLE001
log.exception("video capture_frame failed")
pass
log.error("capture gst exited rc=%s stderr=%s",
vproc.returncode,
err.decode("utf-8", "replace")[-800:])
# ---------------- audio ----------------
aproc: subprocess.Popen | None = None
@@ -160,9 +242,12 @@ def main() -> int:
if audio_card is not None:
audio_source = rtc.AudioSource(
sample_rate=args.audio_rate, num_channels=1,
queue_size_ms=max(250, int(args.enhance_hop_s * 1000) + 50))
queue_size_ms=audio_queue_size_ms(
args.enhance_hop_s, args.audio_queue_ms))
atrack = rtc.LocalAudioTrack.create_audio_track(
f"{identity}-audio", audio_source)
log.info("audio queue_size_ms=%d", audio_queue_size_ms(
args.enhance_hop_s, args.audio_queue_ms))
hop_s = args.enhance_hop_s
chunk = int(hop_s * args.audio_rate) # samples per hop
@@ -177,7 +262,7 @@ def main() -> int:
hops = 0
while aproc.poll() is None:
data = await loop.run_in_executor(
None, aproc.stdout.read, 4096)
io_ex, aproc.stdout.read, 4096)
if not data:
break
buf += data
@@ -189,7 +274,7 @@ def main() -> int:
usable = (n_frames // chunk) * chunk
if usable:
cleaned = await loop.run_in_executor(
None, cleaner.process,
enhance_ex, cleaner.process,
pcm_stereo[:usable])
t_cap = time.perf_counter()
await audio_source.capture_frame(
@@ -292,6 +377,9 @@ def main() -> int:
for t in tasks:
t.cancel()
await asyncio.gather(*tasks, return_exceptions=True)
for ex in split_owned:
with contextlib.suppress(Exception):
ex.shutdown(wait=False, cancel_futures=True)
for p in (vproc, aproc):
if p is not None and p.poll() is None:
with contextlib.suppress(Exception):
+77
View File
@@ -0,0 +1,77 @@
#!/usr/bin/env python3
"""Throwaway: one software-encoder publisher reading cam-01 I420 SHM.
Does not open /dev/camNN. Identity spike-sw so it does not kick cam-01.
"""
from __future__ import annotations
import json
import os
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
os.chdir(ROOT)
from config import load_config # noqa: E402
from tokens import make_token # noqa: E402
SHM_DIR = "/run/livekit-cameras/raw"
SRC = f"{SHM_DIR}/cam-01.i420"
LINK = f"{SHM_DIR}/spike-sw.i420"
def main() -> int:
cfg = load_config()
if not os.path.exists(SRC):
print("missing", SRC, file=sys.stderr)
return 1
try:
os.remove(LINK)
except FileNotFoundError:
pass
os.symlink(SRC, LINK)
cam = {
"index": 0,
"identity": "spike-sw",
"video_device": "/dev/null",
"audio_card": None,
"usb_port": "",
"serial": "",
"extra_video_devices": [],
"width": 640,
"height": 360,
"token": make_token(cfg.url, cfg.api_key, cfg.api_secret, cfg.room, "spike-sw"),
}
os.environ.setdefault(
"LIVEKIT_LIB_PATH",
str(ROOT / "native" / "liblivekit_ffi.so"),
)
os.environ.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
os.environ.setdefault("LIBVA_DRIVER_NAME", "iHD")
argv = [
sys.executable,
str(ROOT / "run_publisher.py"),
"--cameras-json", json.dumps([cam]),
"--url", cfg.url,
"--room", cfg.room,
"--width", "640",
"--height", "360",
"--fps", "30",
"--bitrate", str(cfg.video_bitrate),
"--video-codec", "h264",
"--video-encoder", "software",
"--vaapi-device", cfg.vaapi_device,
"--audio-rate", str(cfg.audio_rate),
"--enhance-mode", "off",
"--beamform-mode", "off",
"--video-shm-dir", SHM_DIR,
"--audio-shm-dir", cfg.audio_shm_dir,
"--encoder-rss-limit-kb", "99000000",
]
os.execv(sys.executable, argv)
if __name__ == "__main__":
raise SystemExit(main())
Binary file not shown.
+23
View File
@@ -0,0 +1,23 @@
"""Annex-B access-unit split."""
from annexb import AnnexBSplitter, is_keyframe, nal_unit_type
def test_splitter_emits_complete_nal():
s = AnnexBSplitter()
assert s.push(b"") == []
first = b"\x00\x00\x00\x01\x67SPS"
second = b"\x00\x00\x00\x01\x68PPS"
assert s.push(first + second) == [first]
assert s.push(b"\x00\x00\x00\x01\x65IDR") == [second]
def test_is_keyframe_idr():
au = b"\x00\x00\x00\x01\x67\x00\x00\x00\x01\x65\x88"
assert is_keyframe(au) is True
assert nal_unit_type(au) == 7
def test_is_keyframe_delta():
au = b"\x00\x00\x00\x01\x41\x00"
assert is_keyframe(au) is False
assert nal_unit_type(au) == 1
+73
View File
@@ -0,0 +1,73 @@
"""Hop energy gate so neighboring C920s do not publish bleed."""
import numpy as np
from audio_gate import HopGate, gate_enabled_for, rms_dbfs
def _sine(n=1600, amp=8000):
return (np.sin(2 * np.pi * 440 * np.arange(n) / 16000) * amp).astype(np.int16)
def test_rms_dbfs_silence_is_floor():
assert rms_dbfs(np.zeros(1600, dtype=np.int16)) <= -120.0
def test_rms_dbfs_sine_is_near_minus_15():
db = rms_dbfs(_sine())
assert -16.5 < db < -14.0
def test_gate_silences_quiet_hop():
g = HopGate(open_db=-32.0, close_db=-40.0, hold_hops=3)
quiet = (np.random.randn(1600) * 80).astype(np.int16)
out = g.apply(quiet)
assert int(np.max(np.abs(out))) == 0
assert g.open is False
def test_gate_passes_loud_hop():
g = HopGate(open_db=-32.0, close_db=-40.0, hold_hops=3)
loud = _sine()
out = g.apply(loud)
assert int(np.max(np.abs(out))) > 1000
assert g.open is True
def test_gate_hold_then_closes():
g = HopGate(open_db=-32.0, close_db=-40.0, hold_hops=2)
g.apply(_sine())
quiet = np.zeros(1600, dtype=np.int16)
assert int(np.max(np.abs(g.apply(quiet)))) == 0 or g.open is True
g.apply(quiet)
g.apply(quiet)
assert g.open is False
assert int(np.max(np.abs(g.apply(quiet)))) == 0
def test_gate_skips_rally():
assert gate_enabled_for("rally", "rally", True) is False
assert gate_enabled_for("cam-07", "rally", True) is True
assert gate_enabled_for("cam-07", "rally", False) is False
def test_cleaner_gates_quiet_after_dsp():
from audio_cleanup import AudioCleaner
c = AudioCleaner(mode="off", beamform="off", gate=HopGate(open_db=-32.0, close_db=-40.0, hold_hops=1))
quiet = (np.random.randn(1600) * 80).astype(np.int16)
out = c.process(quiet)
assert out.shape == (1600,)
assert int(np.max(np.abs(out))) == 0
loud = _sine(1600, 8000)
out = c.process(loud)
assert int(np.max(np.abs(out))) > 1000
def test_cleaner_gate_uses_raw_energy_not_dsp_makeup():
"""Room floor ~-34 dBFS must stay closed even though DSP makeup is ~+3 dB."""
from audio_cleanup import AudioCleaner
c = AudioCleaner(mode="off", beamform="off", gate=HopGate(open_db=-32.0, close_db=-40.0, hold_hops=1))
# sine RMS -34 dBFS => amplitude ~924
floor = _sine(1600, 924)
assert -35.5 < rms_dbfs(floor) < -32.5
out = c.process(floor)
assert int(np.max(np.abs(out))) == 0
+19
View File
@@ -0,0 +1,19 @@
from audio_gst import grouped_audio, shared_audio_cmd
def test_shared_audio_cmd_no_video_elements():
cams = [("cam-01", 4), ("cam-02", 5)]
cmd = shared_audio_cmd(cams, 16000, [8, 9])
joined = " ".join(cmd)
assert joined.count("alsasrc") == 2
assert "device=hw:4,0" in joined
assert "device=hw:5,0" in joined
assert "h264enc" not in joined
assert "v4l2src" not in joined
def test_grouped_audio_splits():
cams = [(f"cam-{i:02d}", i + 1) for i in range(1, 21)]
fds = list(range(20))
groups = grouped_audio(cams, fds, 5)
assert len(groups) == 4
+50
View File
@@ -0,0 +1,50 @@
from capture_va import grouped_cameras, shared_capture_cmd, video_capture_cmd
def test_video_capture_cmd_latest_frame_no_sidecar_h264():
cmd = video_capture_cmd("/dev/cam01", 640, 360, 30)
joined = " ".join(cmd)
assert "v4l2src" in joined
assert "device=/dev/cam01" in joined
assert "jpegdec" in joined
assert "videoconvert" in joined
assert "I420" in joined
assert "leaky=downstream" in joined
assert "max-size-buffers=1" in joined
assert "max-size-time=0" in joined
assert "fdsink" in joined
assert "h264enc" not in joined
assert "tee" not in joined
def test_capture_does_not_gst_crop_or_box():
"""MCU crop is Python-side on 07/14/15 only; gst stays full 360."""
for dev in ("/dev/cam01", "/dev/cam07"):
joined = " ".join(video_capture_cmd(dev, 640, 360, 30))
assert "videocrop" not in joined
assert "videobox" not in joined
assert "videoscale" not in joined
def test_shared_capture_cmd_one_process_no_sidecar_h264():
cams = [("cam-01", "/dev/cam01"), ("cam-02", "/dev/cam02")]
cmd = shared_capture_cmd(cams, 640, 360, 30, [8, 9])
joined = " ".join(cmd)
assert cmd[0].endswith("gst-launch-1.0") or "gst-launch-1.0" in cmd[0]
assert joined.count("v4l2src") == 2
assert joined.count("jpegdec") == 2
assert joined.count("videoconvert") == 2
assert joined.count("fdsink") == 2
assert "fd=8" in joined and "fd=9" in joined
assert "h264enc" not in joined
assert "tee" not in joined
def test_grouped_cameras_splits_into_small_pool():
cams = [(f"cam-{i:02d}", f"/dev/cam{i:02d}") for i in range(1, 21)]
fds = list(range(20, 40))
groups = grouped_cameras(cams, fds, 5)
assert len(groups) == 4
assert all(len(c) == 5 and len(f) == 5 for c, f in groups)
assert groups[0][0][0][0] == "cam-01"
assert groups[-1][0][-1][0] == "cam-20"
+124 -1
View File
@@ -1,4 +1,4 @@
from config import Config, _parse_cameras
from config import Config, _parse_cameras, load_config
def test_parse_cameras_list():
@@ -17,3 +17,126 @@ def test_torch_num_threads_default_one():
def test_enhance_infer_default_auto():
assert Config().enhance_infer == "auto"
def test_latency_flag_defaults():
c = Config()
assert c.audio_queue_ms is None
assert c.video_hold_s is None
assert c.worker_split_executor is False
assert c.enhance_speaker_only is False
assert c.audio_gate is False
assert c.audio_gate_open_db == -28.0
assert c.audio_gate_close_db == -34.0
assert c.audio_gate_hold_s == 0.3
assert c.display_video_capacity == 0
assert c.display_video_format == ""
assert c.display_gst_queue_buffers == 2
assert c.display_speaker_push == "timer"
assert c.participant_tags == ["uwh"]
assert c.display_hide_tags == []
assert c.display_hand_raise is True
assert c.display_hand_raise_hz == 4.0
assert c.display_hand_raise_hold_s == 1.0
assert c.display_hand_raise_model == "thunder"
assert c.display_speak_min_db == -40.0
assert c.display_speak_margin_db == 6.0
assert c.display_speak_hold_s == 0.6
assert c.display_speak_rise_db == 3.0
assert c.display_speak_corr == 0.4
assert c.display_speak_confirm == 2
assert c.portrait_blur is True
assert c.portrait_shm_dir == "/run/livekit-cameras/portrait"
assert c.portrait_hz == 8.0
assert c.portrait_blur_px == 7
assert c.portrait_hold_s == 0.8
def test_enhance_daemon_defaults():
c = Config()
assert c.enhance_daemon is True
assert c.enhance_socket == "/tmp/livekit-enhance.sock"
assert c.capture_daemon is True
assert c.capture_shm_dir == "/run/livekit-cameras/raw"
assert c.audio_daemon is True
assert c.audio_shm_dir == "/run/livekit-cameras/pcm"
assert c.single_publisher is True
def test_load_config_enhance_daemon_env(monkeypatch, tmp_path):
env = tmp_path / ".env"
env.write_text("")
monkeypatch.setenv("ENHANCE_DAEMON", "0")
monkeypatch.setenv("ENHANCE_SOCKET", "/run/foo.sock")
cfg = load_config(env)
assert cfg.enhance_daemon is False
assert cfg.enhance_socket == "/run/foo.sock"
def test_load_config_latency_env(monkeypatch, tmp_path):
env = tmp_path / ".env"
env.write_text("")
monkeypatch.setenv("AUDIO_QUEUE_MS", "60")
monkeypatch.setenv("VIDEO_HOLD_S", "0")
monkeypatch.setenv("WORKER_SPLIT_EXECUTOR", "1")
monkeypatch.setenv("ENHANCE_SPEAKER_ONLY", "true")
monkeypatch.setenv("AUDIO_GATE", "0")
monkeypatch.setenv("AUDIO_GATE_OPEN_DB", "-28")
monkeypatch.setenv("AUDIO_GATE_HOLD_S", "0.5")
monkeypatch.setenv("DISPLAY_VIDEO_CAPACITY", "1")
monkeypatch.setenv("DISPLAY_VIDEO_FORMAT", "BGRA")
monkeypatch.setenv("DISPLAY_GST_QUEUE_BUFFERS", "1")
monkeypatch.setenv("DISPLAY_SPEAKER_PUSH", "arrival")
cfg = load_config(env)
assert cfg.audio_queue_ms == 60
assert cfg.video_hold_s == 0.0
assert cfg.worker_split_executor is True
assert cfg.enhance_speaker_only is True
assert cfg.audio_gate is False
assert cfg.audio_gate_open_db == -28.0
assert cfg.audio_gate_hold_s == 0.5
assert cfg.display_video_capacity == 1
assert cfg.display_video_format == "bgra"
assert cfg.display_gst_queue_buffers == 1
assert cfg.display_speaker_push == "arrival"
def test_load_config_site_tags(monkeypatch, tmp_path):
env = tmp_path / ".env"
env.write_text("")
monkeypatch.setenv("PARTICIPANT_TAGS", "uwh")
monkeypatch.setenv("DISPLAY_HIDE_TAGS", "")
cfg = load_config(env)
assert cfg.participant_tags == ["uwh"]
assert cfg.display_hide_tags == []
monkeypatch.setenv("DISPLAY_HIDE_TAGS", "uwh")
cfg = load_config(env)
assert cfg.display_hide_tags == ["uwh"]
def test_load_config_hand_raise_env(monkeypatch, tmp_path):
env = tmp_path / ".env"
env.write_text("")
monkeypatch.setenv("DISPLAY_HAND_RAISE", "0")
monkeypatch.setenv("DISPLAY_HAND_RAISE_HZ", "3")
monkeypatch.setenv("DISPLAY_HAND_RAISE_HOLD_S", "1.2")
cfg = load_config(env)
assert cfg.display_hand_raise is False
assert cfg.display_hand_raise_hz == 3.0
assert cfg.display_hand_raise_hold_s == 1.2
def test_load_config_portrait_env(monkeypatch, tmp_path):
env = tmp_path / ".env"
env.write_text("")
monkeypatch.setenv("PORTRAIT_BLUR", "0")
monkeypatch.setenv("PORTRAIT_SHM_DIR", "/tmp/portrait")
monkeypatch.setenv("PORTRAIT_HZ", "5")
monkeypatch.setenv("PORTRAIT_BLUR_PX", "11")
monkeypatch.setenv("PORTRAIT_HOLD_S", "0.4")
cfg = load_config(env)
assert cfg.portrait_blur is False
assert cfg.portrait_shm_dir == "/tmp/portrait"
assert cfg.portrait_hz == 5.0
assert cfg.portrait_blur_px == 11
assert cfg.portrait_hold_s == 0.4
+147
View File
@@ -3,11 +3,18 @@ import numpy as np
from display_grid import (
CameraGrid,
bgra_to_i420,
camera_ids,
describe_livekit_error,
display_name,
format_livekit_status,
grid_shape,
i420_nbytes,
is_camera_identity,
letterbox_bgra,
letterbox_i420,
placeholder,
status_screen,
)
@@ -75,7 +82,147 @@ def test_gap_separates_tiles():
assert tuple(canvas[y, x]) == (10, 10, 10, 255)
def test_inner_rect_cam01_is_top_left_inside_border():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4, gap=8, border=3)
x, y, w, h = g.inner_rect(0)
assert (x, y) == (g.gap + g.border, g.gap + g.border)
assert (w, h) == (g.inner_w, g.inner_h)
x2, y2, _, _ = g.inner_rect(1)
assert x2 == g.gap + (g.tile_w + g.gap) + g.border
assert y2 == y
def test_placeholder_draws_text():
img = placeholder(640, 360, ["Screen share", "stub"])
assert img.shape == (360, 640, 4)
assert img.sum() > 0
def test_letterbox_i420_size_and_even_uv():
src = np.zeros((360, 640, 4), dtype=np.uint8)
src[:] = (0, 255, 0, 255)
blob = bgra_to_i420(src)
out = letterbox_i420(blob, 640, 360, 368, 226)
assert len(out) == i420_nbytes(368, 226)
y = np.frombuffer(out, dtype=np.uint8, count=368 * 226).reshape(226, 368)
assert 0 not in y[0]
assert 0 not in y[-1]
def test_grid_i420_render_and_set_frame():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
buf = g.render_i420()
assert len(buf) == i420_nbytes(1920, 1080)
src = np.zeros((360, 640, 4), dtype=np.uint8)
src[:] = (255, 0, 0, 255)
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
buf2 = g.render_i420()
assert buf2 != buf
g.clear("cam-07")
assert g.render_i420() == buf
def test_ensure_slot_binds_remote_to_empty_tile():
g = CameraGrid(width=640, height=360, cols=2, rows=1, identities=["cam-01", "cam-02"])
assert g.ensure_slot("alice") is True
assert "alice" in g._tiles
assert g.identities[0] == "alice"
frame = np.zeros((90, 160, 4), dtype=np.uint8)
g.set_frame("alice", frame)
assert "alice" in g._live
assert g.ensure_slot("alice") is True
def test_grid_shape_shrinks_with_live_count():
assert grid_shape(0) == (1, 1)
assert grid_shape(1) == (1, 1)
assert grid_shape(2) == (2, 1)
assert grid_shape(4) == (2, 2)
assert grid_shape(5) == (3, 2)
assert grid_shape(6) == (3, 2)
assert grid_shape(9) == (3, 3)
assert grid_shape(12) == (4, 3)
assert grid_shape(20) == (5, 4)
cols, rows = grid_shape(5)
assert cols * rows >= 5
assert cols <= 5 and rows <= 4
def test_set_identities_rebuilds_compact_layout():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
g.set_identities(["cam-08", "cam-09", "cam-10", "cam-11", "cam-13"])
assert g.identities == ["cam-08", "cam-09", "cam-10", "cam-11", "cam-13"]
assert (g.cols, g.rows) == (3, 2)
assert g.tile_w > CameraGrid(width=1920, height=1080, cols=5, rows=4).tile_w
g.set_identities(["cam-08", "cam-09"])
assert (g.cols, g.rows) == (2, 1)
assert g.identities == ["cam-08", "cam-09"]
g.set_identities([])
assert g.identities == []
assert (g.cols, g.rows) == (1, 1)
def test_status_overlay_replaces_mosaic_until_cleared():
g = CameraGrid(width=320, height=180, cols=2, rows=1, identities=["cam-01", "cam-02"])
mosaic = g.render()
g.set_status(["LIVEKIT DISCONNECTED", "ws://10.200.0.21:7880", "host unreachable"])
overlay = g.render()
assert overlay != mosaic
assert len(overlay) == 320 * 180 * 4
i420 = g.render_i420()
assert len(i420) == i420_nbytes(320, 180)
g.set_status(None)
assert g.render() == mosaic
def test_format_livekit_status_and_errors():
assert format_livekit_status(url="ws://10.200.0.21:7880", state="connected") == []
connecting = format_livekit_status(url="ws://10.200.0.21:7880", state="connecting")
assert connecting[0] == "CONNECTING TO LIVEKIT"
assert "10.200.0.21:7880" in connecting[1]
lines = format_livekit_status(
url="ws://10.200.0.21:7880",
state="disconnected",
detail="host unreachable",
retry_s=5,
)
assert lines[0] == "LIVEKIT DISCONNECTED"
assert "host unreachable" in lines
assert any("retry" in ln.lower() for ln in lines)
assert "unreachable" in describe_livekit_error(
OSError("No route to host")).lower() or "host" in describe_livekit_error(
OSError("[Errno 113] No route to host")).lower()
assert "timed out" in describe_livekit_error(TimeoutError("timed out")).lower()
assert "refused" in describe_livekit_error(ConnectionRefusedError("Connection refused")).lower()
def test_status_screen_disconnected_uses_composed_layout():
img = status_screen(1920, 1080, [
"LIVEKIT DISCONNECTED",
"ws://10.200.0.21:7880",
"host unreachable",
"retrying in 5s",
])
assert img.shape == (1080, 1920, 4)
assert img.dtype == np.uint8
# Composed: background, accent rail, card — not a flat fill.
assert len(np.unique(img.reshape(-1, 4), axis=0)) > 12
bg = img[24, 80, :3]
card = img[540, 960, :3]
assert tuple(int(x) for x in bg) != tuple(int(x) for x in card)
# Disconnected accent rail is warm amber (BGRA: R > G > B).
stripe = img[540, :8, :3].mean(axis=0)
b, g, r = (float(x) for x in stripe)
assert r > g > b
assert r > 180
def test_status_screen_connecting_uses_cool_accent():
img = status_screen(1920, 1080, [
"CONNECTING TO LIVEKIT",
"ws://10.200.0.21:7880",
])
stripe = img[540, :8, :3].mean(axis=0)
b, g, r = (float(x) for x in stripe)
assert b > r
assert b > 160
+26
View File
@@ -19,3 +19,29 @@ def test_select_three_prefers_xe_hdmi():
assert three[0].name == "HDMI-A-7" or three[0].connected
ids = [o.connector_id for o in three]
assert len(set(ids)) == 3
def test_named_disconnected_falls_back_to_connected_xe():
from drm_outputs import DrmOutput, pick_outputs
wall = DrmOutput(
name="HDMI-A-7", sysfs="card0-HDMI-A-7", node="/dev/dri/card0",
driver="xe", connector_id=538, status="connected", width=3840, height=2160,
)
stale_grid = DrmOutput(
name="HDMI-A-3", sysfs="card1-HDMI-A-3", node="/dev/dri/card1",
driver="i915", connector_id=530, status="disconnected",
)
stale_spk = DrmOutput(
name="HDMI-A-5", sysfs="card0-HDMI-A-5", node="/dev/dri/card0",
driver="xe", connector_id=520, status="disconnected",
)
picked = pick_outputs(
[stale_grid, stale_spk, wall],
n=2,
names=["HDMI-A-3", "HDMI-A-5"],
)
assert len(picked) == 1
assert picked[0].name == "HDMI-A-7"
assert picked[0].connected
assert picked[0].driver == "xe"
+44
View File
@@ -0,0 +1,44 @@
from encode_daemon import _one_enc
def test_h264_uses_igpu_va_not_x264_or_arc_av1():
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD129")
s = " ".join(cmd)
assert "varenderD129h264enc" in cmd
assert "x264enc" not in cmd
assert "vaav1enc" not in cmd
assert "format=nv12" in s
assert "key-int-max=1" in cmd
assert "b-frames=0" in cmd
def test_h264_factory_follows_render_node():
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD128")
assert "varenderD128h264enc" in cmd
def test_c920_stays_fast_gop1():
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD129")
assert "bitrate=2000" in cmd
assert "key-int-max=1" in cmd
assert "target-usage=7" in cmd
def test_rally_1080_uses_higher_bitrate_and_no_motion_trails():
cmd = _one_enc(1920, 1088, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD129")
assert "bitrate=20000" in cmd
assert "key-int-max=1" in cmd
assert "target-usage=2" in cmd
assert "b-frames=0" in cmd
def test_rally_bitrate_rollbacks_via_hi_kbps():
cmd = _one_enc(1920, 1088, 30, 2000, 3, 4, "h264",
render="/dev/dri/renderD129", hi_kbps=12000)
assert "bitrate=12000" in cmd
def test_av1_stays_on_arc_vaav1enc():
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "av1")
assert "vaav1enc" in cmd
assert "varenderD129h264enc" not in cmd
+135
View File
@@ -0,0 +1,135 @@
import os
import threading
import time
import numpy as np
import pytest
def test_frame_roundtrip_stereo_hop():
from enhance_ipc import dump_frame, load_frame
hop = 1600
t = np.arange(hop, dtype=np.int16)
pcm = np.stack([t, t + 1], axis=1)
blob = dump_frame("cam-07", pcm)
ident, out = load_frame(blob)
assert ident == "cam-07"
assert out.dtype == np.int16
assert out.shape == (hop, 2)
np.testing.assert_array_equal(out, pcm)
def test_unix_client_gets_handler_output(tmp_path):
from enhance_daemon import serve_unix
from enhance_ipc import EnhanceClient
sock = str(tmp_path / "enhance.sock")
stop = threading.Event()
def handler(ident, pcm):
assert ident == "cam-03"
# fake "enhance": average channels, invert
if pcm.ndim == 2:
mono = pcm.mean(axis=1)
else:
mono = pcm
return (-mono).astype(np.int16)
t = threading.Thread(target=serve_unix, args=(sock, handler, stop), daemon=True)
t.start()
deadline = time.time() + 2
while time.time() < deadline and not (tmp_path / "enhance.sock").exists():
time.sleep(0.01)
hop = 800
pcm = np.stack([np.full(hop, 1000, np.int16), np.full(hop, 3000, np.int16)], axis=1)
client = EnhanceClient(sock)
try:
out = client.process("cam-03", pcm)
finally:
client.close()
stop.set()
assert out.dtype == np.int16
assert out.shape == (hop,)
assert int(out[0]) == -2000
def _start_echo_daemon(sock, stop):
from enhance_daemon import serve_unix
def handler(ident, pcm):
if pcm.ndim == 2:
mono = pcm.mean(axis=1)
else:
mono = pcm
return np.asarray(mono, dtype=np.int16)
t = threading.Thread(target=serve_unix, args=(sock, handler, stop), daemon=True)
t.start()
deadline = time.time() + 2
while time.time() < deadline and not os.path.exists(sock):
time.sleep(0.01)
return t
def test_audio_cleaner_remote_does_not_load_model(tmp_path):
from audio_cleanup import AudioCleaner
sock = str(tmp_path / "enhance.sock")
stop = threading.Event()
_start_echo_daemon(sock, stop)
try:
c = AudioCleaner(
mode="force", beamform="force", enhance_socket=sock,
identity="cam-09")
assert c._enhancer is None
assert c._beamformer is None
assert c.backend == "daemon"
hop = 1600
stereo = np.stack([
np.full(hop, 400, np.int16),
np.full(hop, 800, np.int16),
], axis=1)
out = c.process(stereo)
assert out.shape == (hop,)
assert out.dtype == np.int16
assert int(out[0]) == 600
finally:
stop.set()
def test_shared_engine_constructs_one_cleaner_for_two_cams():
from audio_cleanup import AudioCleaner
from enhance_daemon import SharedEnhanceEngine
n = {"c": 0}
def make():
n["c"] += 1
return AudioCleaner(mode="off", beamform="off")
eng = SharedEnhanceEngine(make_cleaner=make)
hop = np.zeros(800, dtype=np.int16)
out_a = eng.process("cam-01", hop)
out_b = eng.process("cam-02", hop)
out_a2 = eng.process("cam-01", hop)
assert n["c"] == 1
assert set(eng.identities()) == {"cam-01", "cam-02"}
assert out_a.shape == (800,)
assert out_b.shape == (800,)
assert out_a2.shape == (800,)
def test_wait_for_socket_false_then_true(tmp_path):
from enhance_daemon import wait_for_socket, serve_unix
sock = str(tmp_path / "missing.sock")
assert wait_for_socket(sock, timeout=0.15) is False
stop = threading.Event()
t = threading.Thread(
target=serve_unix, args=(sock, lambda i, p: p, stop), daemon=True)
t.start()
try:
assert wait_for_socket(sock, timeout=2.0) is True
finally:
stop.set()
+23
View File
@@ -0,0 +1,23 @@
from frame_shm import LatestFrameReader, LatestFrameWriter, frame_bytes
def test_latest_frame_roundtrip(tmp_path):
path = str(tmp_path / "cam-01.i420")
w, h = 16, 16
n = frame_bytes(w, h)
writer = LatestFrameWriter(path, w, h)
payload = bytes(range(256)) * (n // 256 + 1)
payload = payload[:n]
seq = writer.write(payload)
assert seq == 1
reader = LatestFrameReader(path)
got = reader.latest()
assert got is not None
assert got[0] == 1
assert got[1] == payload
seq2 = writer.write(bytes([3]) * n)
got2 = reader.latest()
assert got2 is not None
assert got2[0] == seq2
assert got2[1][0] == 3
writer.close()
+66
View File
@@ -0,0 +1,66 @@
from display_grid import CameraGrid
from drm_outputs import DrmOutput
from gst_sink import appsrc_cmd, va_grid_cmd
def _out():
return DrmOutput(
name="HDMI-A-7",
sysfs="card0-HDMI-A-7",
node="/dev/dri/card0",
driver="xe",
connector_id=538,
status="connected",
)
def test_appsrc_queue_buffers_default_two():
cmd = appsrc_cmd(_out(), 1920, 1080, 30)
assert "max-size-buffers=2" in cmd
def test_appsrc_queue_buffers_one():
cmd = appsrc_cmd(_out(), 1920, 1080, 30, queue_buffers=1)
assert "max-size-buffers=1" in cmd
assert "max-size-buffers=2" not in cmd
def test_appsrc_i420_blocksize():
cmd = appsrc_cmd(_out(), 1920, 1080, 30, pixel_format="i420")
joined = " ".join(cmd)
assert "format=i420" in joined
assert "blocksize=3110400" in joined # 1920*1080*3//2
assert "format=bgra" not in joined
def test_va_grid_cmd_pins_arc_compositor_and_twenty_tiles():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
tile_fds = list(range(4, 24))
cmd = va_grid_cmd(
_out(), g, ingest_w=640, ingest_h=360, fps=15,
tile_fds=tile_fds, chrome_fd=3, queue_buffers=1)
joined = " ".join(cmd)
assert "varenderD129compositor" in joined
assert "varenderD129postproc" in joined
assert "kmssink" in joined
assert "connector-id=538" in joined
assert "driver-name=xe" in joined
assert joined.count("fdsrc") == 21 # chrome + 20 tiles
x, y, w, h = g.inner_rect(0)
assert f"xpos={x}" in joined
assert f"ypos={y}" in joined
assert "format=i420" in joined
assert "blocksize=345600" in joined # 640*360*1.5
def test_va_grid_cmd_h264_uses_arc_decoder():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
paths = [f"/tmp/cam-{i:02d}.h264" for i in range(1, 21)]
cmd = va_grid_cmd(
_out(), g, ingest_w=640, ingest_h=360, fps=15,
tile_h264_paths=paths, chrome_fd=3)
joined = " ".join(cmd)
assert "varenderD129h264dec" in joined
assert joined.count("h264parse") == 20
assert "fdsrc" in joined # chrome only
assert joined.count("filesrc") == 20
+30
View File
@@ -0,0 +1,30 @@
import os
import stat
from h264_fifo import hold_fifo
def test_hold_fifo_creates_rdwr_pipe(tmp_path):
path = str(tmp_path / "cam-01.h264")
fd = hold_fifo(path)
try:
st = os.stat(path)
assert stat.S_ISFIFO(st.st_mode)
os.write(fd, b"\x00\x01")
assert os.read(fd, 2) == b"\x00\x01"
finally:
os.close(fd)
def test_hold_fifo_does_not_replace_existing_fifo(tmp_path):
path = str(tmp_path / "cam-02.h264")
a = hold_fifo(path)
ino = os.stat(path).st_ino
b = hold_fifo(path)
try:
assert os.stat(path).st_ino == ino
os.write(a, b"ab")
assert os.read(b, 2) == b"ab"
finally:
os.close(a)
os.close(b)
+171
View File
@@ -0,0 +1,171 @@
"""Hand-raise rule, latch, and yellow I420 chrome."""
import time
import numpy as np
from display_grid import FRAME_RAISE, CameraGrid, bgra_to_i420, i420_nbytes
from hand_raise import RaiseLatch, letterbox_rgb, wrist_is_raised
def _kpts() -> np.ndarray:
k = np.zeros((17, 3), dtype=np.float32)
k[:, 2] = 0.8
k[0] = [0.35, 0.50, 0.9] # nose
k[5] = [0.50, 0.35, 0.9] # L shoulder
k[6] = [0.50, 0.65, 0.9] # R shoulder
k[7] = [0.62, 0.30, 0.9] # L elbow
k[8] = [0.62, 0.70, 0.9]
k[9] = [0.70, 0.28, 0.9] # L wrist down
k[10] = [0.70, 0.72, 0.9]
return k
def test_wrist_down_is_not_raised():
assert wrist_is_raised(_kpts()) is False
def test_left_wrist_above_shoulder_is_raised():
k = _kpts()
k[7] = [0.32, 0.30, 0.9] # elbow up
k[9] = [0.18, 0.32, 0.9] # wrist above shoulder (y 0.50)
assert wrist_is_raised(k) is True
def test_modest_wrist_lift_is_raised():
"""Webcam framing: hand just above the shoulder still counts."""
k = _kpts()
k[7] = [0.46, 0.30, 0.9]
k[9] = [0.43, 0.32, 0.9] # 0.07 above shoulder y=0.50
assert wrist_is_raised(k) is True
def test_wrist_on_face_rejected():
k = _kpts()
k[7] = [0.38, 0.48, 0.9]
k[9] = [0.28, 0.50, 0.9] # up, but next to the nose
assert wrist_is_raised(k) is False
def test_wrist_above_head_without_shoulders():
"""Close-up webcam: shoulders off-frame, hand still clearly up."""
k = np.zeros((17, 3), dtype=np.float32)
k[0] = [0.55, 0.50, 0.85] # nose
k[9] = [0.18, 0.62, 0.7] # wrist high, not on the face
assert wrist_is_raised(k) is True
def test_low_score_not_raised():
k = _kpts()
k[:, 2] = 0.05
assert wrist_is_raised(k) is False
assert wrist_is_raised(np.zeros((17, 3), dtype=np.float32)) is False
def test_raise_latch_needs_hits_then_holds():
latch = RaiseLatch(hits=2, hold_s=0.4, window=2)
t = 10.0
assert latch.update(True, t) is False
assert latch.update(True, t + 0.1) is True
assert latch.update(False, t + 0.2) is True
assert latch.update(False, t + 0.55) is False
def test_raise_latch_survives_one_miss():
latch = RaiseLatch(hits=3, hold_s=1.0, window=5)
t = 10.0
assert latch.update(True, t) is False
assert latch.update(True, t + 0.1) is False
assert latch.update(False, t + 0.2) is False
assert latch.update(True, t + 0.3) is True
assert latch.update(True, t + 0.4) is True
assert latch.update(False, t + 0.5) is True
def test_letterbox_rgb_pads_16x9():
rgb = np.full((360, 640, 3), 200, dtype=np.uint8)
out = letterbox_rgb(rgb, 192)
assert out.shape == (192, 192, 3)
assert int(out[0, 96, 0]) == 0
assert int(out[96, 96, 0]) == 200
thunder = letterbox_rgb(rgb, 256)
assert thunder.shape == (256, 256, 3)
def test_resolve_movenet_thunder_is_256():
from pathlib import Path
from hand_raise import resolve_movenet
path, size, name = resolve_movenet("thunder")
assert name == "thunder"
assert size == 256
assert path.name == "movenet_singlepose_thunder.onnx"
path, size, name = resolve_movenet("lightning")
assert name == "lightning" and size == 192
assert Path(path).name == "movenet_singlepose_lightning.onnx"
def test_set_highlight_paints_yellow_i420_border():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
src = np.zeros((360, 640, 4), dtype=np.uint8)
src[:] = (40, 40, 40, 255)
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
before = np.frombuffer(g.render_i420(), dtype=np.uint8).copy()
g.set_highlight("cam-07", True)
assert g.highlighted("cam-07") is True
after = np.frombuffer(g.render_i420(), dtype=np.uint8)
assert not np.array_equal(before, after)
idx = g.identities.index("cam-07")
ox, oy = g.tile_origin(idx)
yplane = after[: g.width * g.height].reshape(g.height, g.width)
yv, _, _ = g._yuv_raise
# Top edge of the tile chrome should be yellow-luma, not gray live.
assert int(yplane[oy, ox + g.tile_w // 2]) == yv
assert int(yplane[oy - 2, ox + g.tile_w // 2]) == yv
ix, iy, iw, _ih = g.inner_rect(idx)
assert int(yplane[iy, ix + iw // 2]) != yv
g.set_highlight("cam-07", False)
assert g.highlighted("cam-07") is False
restored = np.frombuffer(g.render_i420(), dtype=np.uint8)
assert int(restored[: g.width * g.height].reshape(g.height, g.width)[oy, ox + g.tile_w // 2]) != yv
def test_clear_drops_yellow():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
src = np.zeros((360, 640, 4), dtype=np.uint8)
g.set_frame_i420("cam-01", bgra_to_i420(src), 640, 360)
g.set_highlight("cam-01", True)
g.clear("cam-01")
assert g.highlighted("cam-01") is False
assert len(g.render_i420()) == i420_nbytes(1920, 1080)
def test_fp32_movenet_sees_person_not_raised():
"""INT8 Xenova weights scored ~0.05 on a real person; FP32 must lock on."""
from pathlib import Path
from hand_raise import HandRaiseMonitor, MODEL_PATH
fixture = Path(__file__).parent / "fixtures" / "movenet_warrior_192.npy"
assert MODEL_PATH.name == "movenet_singlepose_lightning.onnx"
rgb = np.load(fixture)
mon = HandRaiseMonitor(enabled=True, model="lightning")
assert mon.enabled
k = mon.infer_rgb(rgb)
assert float(k[:, 2].max()) > 0.4
assert float(k[0, 2]) > 0.3 # nose
assert wrist_is_raised(k) is False # warrior: arms out, not up
def test_fp32_thunder_sees_person_not_raised():
from pathlib import Path
from hand_raise import HandRaiseMonitor, resolve_movenet
path, size, name = resolve_movenet("thunder")
assert path.is_file()
assert size == 256 and name == "thunder"
rgb = np.load(Path(__file__).parent / "fixtures" / "movenet_warrior_192.npy")
mon = HandRaiseMonitor(enabled=True, model="thunder")
assert mon.enabled
assert mon._input_size == 256
k = mon.infer_rgb(rgb)
assert float(k[:, 2].max()) > 0.4
assert float(k[0, 2]) > 0.3
assert wrist_is_raised(k) is False
+47
View File
@@ -0,0 +1,47 @@
from hop_shm import HopReader, HopWriter, hop_bytes
def test_hop_ring_preserves_order(tmp_path):
path = str(tmp_path / "cam-01.pcm")
n = hop_bytes(16000, 2, 0.1)
w = HopWriter(path, n, nslots=4)
r = HopReader(path)
for i in range(3):
payload = bytes([i]) * n
w.write(payload)
got = [r.read() for _ in range(3)]
assert got[0] is not None and got[0][0] == 0
assert got[1] is not None and got[1][0] == 1
assert got[2] is not None and got[2][0] == 2
assert r.read() is None
w.close()
def test_hop_ring_skips_when_reader_lags(tmp_path):
path = str(tmp_path / "cam-01.pcm")
n = 32
w = HopWriter(path, n, nslots=4)
r = HopReader(path)
for i in range(10):
w.write(bytes([i]) * n)
last = r.read()
assert last is not None
w.close()
def test_reader_follows_writer_restart(tmp_path):
"""Cameras-only restart zeros seq; the wall must not stay stuck on _last."""
path = str(tmp_path / "cam-01.pcm")
n = 32
w = HopWriter(path, n, nslots=4)
r = HopReader(path)
for i in range(8):
w.write(bytes([i]) * n)
assert r.read() is not None
w.close()
w2 = HopWriter(path, n, nslots=4)
w2.write(bytes([99]) * n)
got = r.read()
assert got is not None and got[0] == 99
assert r.read() is None
w2.close()
+56
View File
@@ -0,0 +1,56 @@
from i420_nv12 import (
coded_dim,
drop_va_pad_i420,
i420_to_nv12_into,
nv12_bytes,
va_visible_height,
)
def test_coded_dim_pads_up():
assert coded_dim(360) == 368
assert coded_dim(1080) == 1088
assert coded_dim(640) == 640
assert coded_dim(352) == 352
def test_i420_to_nv12_pads_last_row():
w, h = 640, 360
ysz = w * h
uv = (w // 2) * (h // 2)
src = bytearray(ysz + 2 * uv)
src[:ysz] = b"\x80" * ysz
src[(h - 1) * w:ysz] = b"\x90" * w
src[ysz:ysz + uv] = b"\x40" * uv
src[ysz + uv:] = b"\xc0" * uv
cw, ch = 640, 368
out = bytearray(nv12_bytes(cw, ch))
n = i420_to_nv12_into(src, w, h, out)
assert n == nv12_bytes(cw, ch)
assert out[0] == 0x80
# cloned pad rows 360-367
assert out[360 * cw] == 0x90
assert out[367 * cw] == 0x90
assert out[cw * ch] == 0x40
assert out[cw * ch + 1] == 0xc0
def test_drop_va_pad_restores_360():
w, h = 640, 368
vis = 360
ysz = w * h
uv = (w // 2) * (h // 2)
src = bytearray(ysz + 2 * uv)
src[: w * vis] = b"\x77" * (w * vis)
src[w * vis:ysz] = b"\x11" * (w * 8)
out, ow, oh = drop_va_pad_i420(src, w, h)
assert (ow, oh) == (w, vis)
assert len(out) == w * vis * 3 // 2
assert out[0] == 0x77
assert 0x11 not in out[: w * vis]
def test_va_visible_height():
assert va_visible_height(640, 368) == 360
assert va_visible_height(1920, 1088) == 1080
assert va_visible_height(640, 360) == 360
assert va_visible_height(640, 352) == 352
+99
View File
@@ -0,0 +1,99 @@
from capture_va import (
crop_scale_i420_drop_bottom,
jpeg_mcu_bottom_crop,
needs_jpeg_mcu_fix,
repeat_i420_bottom_from_last_good,
)
def _i420(width, height, y=16, u=128, v=128):
y_size = width * height
uv = (width // 2) * (height // 2)
buf = bytearray(y_size + 2 * uv)
buf[:y_size] = bytes([y]) * y_size
buf[y_size:y_size + uv] = bytes([u]) * uv
buf[y_size + uv:] = bytes([v]) * uv
return buf
def test_repeat_clones_last_good_y_and_chroma_over_black_pad():
w, h, crop = 640, 360, 8
buf = _i420(w, h, y=16, u=128, v=128)
last_good = h - crop - 1
y_stride = w
# last good Y row = 90; last good UV row = 40 (chroma 4:2:0)
src = last_good * y_stride
buf[src:src + y_stride] = bytes([90]) * y_stride
uv_w = w // 2
uv_h = h // 2
uv_crop = crop // 2
last_uv = uv_h - uv_crop - 1
y_size = w * h
u0 = y_size + last_uv * uv_w
v0 = y_size + uv_w * uv_h + last_uv * uv_w
buf[u0:u0 + uv_w] = bytes([40]) * uv_w
buf[v0:v0 + uv_w] = bytes([200]) * uv_w
repeat_i420_bottom_from_last_good(buf, w, h, crop)
for r in range(h - crop, h):
row = buf[r * y_stride:(r + 1) * y_stride]
assert set(row) == {90}, r
for r in range(uv_h - uv_crop, uv_h):
u = buf[y_size + r * uv_w:y_size + (r + 1) * uv_w]
v = buf[y_size + uv_w * uv_h + r * uv_w:
y_size + uv_w * uv_h + (r + 1) * uv_w]
assert set(u) == {40}, r
assert set(v) == {200}, r
# rows above the pad stay put
mid = (h // 2) * y_stride
assert set(buf[mid:mid + y_stride]) == {16}
def test_360p_crops_partial_mcu_and_last_full_mcu():
"""360 % 16 = 8 leftover; last full 16-row MCU is also garbage on 07/14/15."""
assert jpeg_mcu_bottom_crop(360) == 24
assert jpeg_mcu_bottom_crop(480) == 0
assert jpeg_mcu_bottom_crop(352) == 0
assert jpeg_mcu_bottom_crop(8) == 0
buf = _i420(640, 480, y=77)
before = bytes(buf)
repeat_i420_bottom_from_last_good(buf, 640, 480, jpeg_mcu_bottom_crop(480))
assert bytes(buf) == before
def test_mcu_fix_cams_off():
assert not needs_jpeg_mcu_fix("cam-07")
assert not needs_jpeg_mcu_fix("cam-12")
assert not needs_jpeg_mcu_fix("cam-14")
assert not needs_jpeg_mcu_fix("cam-15")
assert not needs_jpeg_mcu_fix("cam-01")
assert not needs_jpeg_mcu_fix("cam-20")
def test_crop_scale_drops_garbage_without_cloned_stripe():
w, h = 640, 360
crop = 24
buf = _i420(w, h, y=40, u=100, v=140)
y_size = w * h
# garbage last 24 Y rows
buf[(h - crop) * w:y_size] = bytes([250]) * (crop * w)
out = crop_scale_i420_drop_bottom(buf, w, h, crop)
assert len(out) == len(buf)
y = memoryview(out)[:y_size]
last = bytes(y[(h - 8) * w:h * w])
assert last.count(250) == 0
assert set(last) == {16}
def test_live_path_clones_not_studio_black():
w, h, crop = 640, 360, 24
buf = _i420(w, h, y=40, u=100, v=140)
last_good = h - crop - 1
buf[last_good * w:(last_good + 1) * w] = bytes([90]) * w
buf[(h - crop) * w:h * w] = bytes([250]) * (crop * w)
repeat_i420_bottom_from_last_good(buf, w, h, crop)
visible = bytes(buf[(h - crop) * w:h * w])
assert set(visible) == {90}
assert 16 not in visible
assert 250 not in visible
+64
View File
@@ -0,0 +1,64 @@
"""Env-flagged low-latency knobs (hop/queue, hold, enhance scope)."""
from latency import (
audio_queue_size_ms,
effective_enhance_mode,
video_hold_seconds,
video_stream_kwargs,
)
def test_audio_queue_defaults_to_legacy_floor():
assert audio_queue_size_ms(0.1, audio_queue_ms=None) == 150
assert audio_queue_size_ms(0.2, audio_queue_ms=None) == 250
def test_audio_queue_env_overrides_floor():
assert audio_queue_size_ms(0.1, audio_queue_ms=60) == 60
assert audio_queue_size_ms(0.05, audio_queue_ms=60) == 60
def test_video_hold_matches_hop_when_audio_present():
assert video_hold_seconds(0.1, has_audio=True, video_hold_s=None) == 0.1
assert video_hold_seconds(0.05, has_audio=True, video_hold_s=None) == 0.05
def test_video_hold_zero_without_audio():
assert video_hold_seconds(0.1, has_audio=False, video_hold_s=None) == 0.0
assert video_hold_seconds(0.1, has_audio=False, video_hold_s=0.2) == 0.0
def test_video_hold_env_overrides_including_zero():
assert video_hold_seconds(0.1, has_audio=True, video_hold_s=0.0) == 0.0
assert video_hold_seconds(0.1, has_audio=True, video_hold_s=0.05) == 0.05
def test_enhance_speaker_only_rally_disables_c920s():
assert effective_enhance_mode(
"auto", "cam-01", speaker_only=True, speaker_camera="rally",
) == "off"
assert effective_enhance_mode(
"auto", "rally", speaker_only=True, speaker_camera="rally",
) == "auto"
def test_enhance_speaker_only_inactive_leaves_mode():
assert effective_enhance_mode(
"auto", "cam-07", speaker_only=False, speaker_camera="cam-01",
) == "auto"
def test_enhance_speaker_only_ignored_when_speaker_is_active():
assert effective_enhance_mode(
"auto", "cam-07", speaker_only=True, speaker_camera="active",
) == "auto"
def test_video_stream_legacy_unbounded_native():
assert video_stream_kwargs(capacity=0, format_name="") == {}
def test_video_stream_latest_bgra():
assert video_stream_kwargs(capacity=1, format_name="bgra") == {
"capacity": 1,
"format": "bgra",
}
+46
View File
@@ -0,0 +1,46 @@
"""Latest-AU H.264 mmap: overwrite unread payloads; seq is monotonic."""
from nal_shm import AU_MAX, NalReader, NalWriter
def test_writer_seq_starts_at_one_and_increments(tmp_path):
path = str(tmp_path / "cam-01.h264")
w = NalWriter(path, max_payload=64)
assert w.write(b"\x00\x00\x00\x01\x65" + b"x" * 10) == 1
assert w.write(b"\x00\x00\x00\x01\x41" + b"y" * 10) == 2
w.close()
def test_reader_returns_latest_payload(tmp_path):
path = str(tmp_path / "cam-01.h264")
w = NalWriter(path, max_payload=64)
w.write(b"first-access-unit-xxxx")
w.write(b"second-access-unit-yyy")
r = NalReader(path)
seq, payload = r.read()
assert seq == 2
assert payload.startswith(b"second-access-unit")
w.close()
r.close()
def test_reader_none_when_no_new_au(tmp_path):
path = str(tmp_path / "cam-01.h264")
w = NalWriter(path, max_payload=64)
w.write(b"only-once")
r = NalReader(path)
assert r.read()[0] == 1
assert r.read() is None
w.close()
r.close()
def test_write_drops_oversize_payload(tmp_path):
path = str(tmp_path / "cam-01.h264")
w = NalWriter(path, max_payload=8)
assert w.write(b"123456789") == 0
w.close()
def test_au_max_is_bounded():
assert AU_MAX >= 64 * 1024
assert AU_MAX <= 2 * 1024 * 1024
+88
View File
@@ -0,0 +1,88 @@
from participant_tags import (
should_hide_participant,
tags_from_fields,
token_attributes,
want_display_video,
)
def test_tags_from_attribute_tag():
assert tags_from_fields({"tag": "uwh"}, "") == {"uwh"}
def test_tags_from_comma_separated_tags_attr():
assert tags_from_fields({"tags": "uwh,lab"}, None) == {"uwh", "lab"}
def test_tags_from_json_metadata():
assert tags_from_fields({}, '{"tag": "uwh"}') == {"uwh"}
def test_hide_disabled_when_hide_tags_empty():
assert should_hide_participant({"tag": "uwh"}, "", []) is False
def test_hide_uwh_when_filter_enabled():
assert should_hide_participant({"tag": "uwh"}, "", ["uwh"]) is True
assert should_hide_participant({"tag": "nyc"}, "", ["uwh"]) is False
assert should_hide_participant({}, "", ["uwh"]) is False
def test_token_attributes_sets_tag_and_tags():
assert token_attributes(["uwh"]) == {"tag": "uwh", "tags": "uwh"}
assert token_attributes([]) == {}
def test_want_display_keeps_cameras_when_filter_off():
assert want_display_video(
identity="cam-01",
attributes={"tag": "uwh"},
metadata="",
hide_tags=[],
participant_prefix="cam",
display_identity="display-wall",
) is True
assert want_display_video(
identity="alice",
attributes={},
metadata="",
hide_tags=[],
participant_prefix="cam",
display_identity="display-wall",
) is False
assert want_display_video(
identity="rally",
attributes={},
metadata="",
hide_tags=[],
participant_prefix="cam",
display_identity="display-wall",
speaker_identity="rally",
) is True
assert want_display_video(
identity="rally",
attributes={},
metadata="",
hide_tags=[],
participant_prefix="cam",
display_identity="display-wall",
) is False
def test_want_display_hides_uwh_and_shows_remote_when_filter_on():
assert want_display_video(
identity="cam-01",
attributes={"tag": "uwh"},
metadata="",
hide_tags=["uwh"],
participant_prefix="cam",
display_identity="display-wall",
) is False
assert want_display_video(
identity="alice",
attributes={"tag": "nyc"},
metadata="",
hide_tags=["uwh"],
participant_prefix="cam",
display_identity="display-wall",
) is True
+209
View File
@@ -0,0 +1,209 @@
"""Person-aware background blur: composite, hold, I420, encode path."""
import numpy as np
from portrait import (
MaskHold,
apply_i420,
apply_portrait,
bgr_to_i420,
composite_bgr,
i420_nbytes,
i420_source_path,
i420_to_bgr,
person_present,
soft_blur_bgr,
)
def _checker(h: int, w: int) -> np.ndarray:
img = np.zeros((h, w, 3), dtype=np.uint8)
img[::4, :] = 255
img[:, ::4] = 255
return img
def test_composite_keeps_person_pixels():
h, w = 80, 128
bg = _checker(h, w)
person = bg.copy()
person[16:64, 32:96] = (40, 180, 40)
mask = np.zeros((h, w), dtype=np.float32)
mask[16:64, 32:96] = 1.0
out = composite_bgr(person, soft_blur_bgr(person, 9), mask)
roi = out[24:56, 40:88]
assert abs(int(roi[:, :, 1].mean()) - 180) < 8
assert abs(int(roi[:, :, 0].mean()) - 40) < 8
def test_composite_blurs_background_checker():
h, w = 80, 128
bg = _checker(h, w)
person = bg.copy()
person[16:64, 32:96] = (40, 180, 40)
mask = np.zeros((h, w), dtype=np.float32)
mask[16:64, 32:96] = 1.0
sharp_var = float(bg[0:12, :].var())
out = composite_bgr(person, soft_blur_bgr(person, 9), mask)
blur_var = float(out[0:12, :].var())
assert sharp_var > 1000
assert blur_var < sharp_var * 0.5
def test_empty_mask_is_fully_blurred():
img = _checker(40, 64)
mask = np.zeros((40, 64), dtype=np.float32)
out = apply_portrait(img, mask)
assert person_present(mask) is False
assert float(out.var()) < float(img.var()) * 0.5
assert not np.array_equal(out, img)
def test_tiny_blob_is_not_a_person():
mask = np.zeros((40, 64), dtype=np.float32)
mask[10:12, 10:12] = 0.9
assert person_present(mask) is False
def test_mask_hold_keeps_last_on_miss():
hold = MaskHold(hold_s=0.5, ema=1.0)
mask = np.ones((8, 8), dtype=np.float32)
got = hold.update(mask, True, 1.0)
assert got is not None
missed = hold.update(None, False, 1.2)
assert missed is not None
assert float(missed.mean()) == 1.0
assert hold.update(None, False, 1.6) is None
def test_ema_smooths_mask_flicker():
hold = MaskHold(hold_s=1.0, ema=0.5)
a = np.zeros((4, 4), dtype=np.float32)
b = np.ones((4, 4), dtype=np.float32)
hold.update(a, True, 0.0)
out = hold.update(b, True, 0.1)
assert 0.4 < float(out.mean()) < 0.6
def test_i420_roundtrip_keeps_640x360():
w, h = 640, 360
bgr = np.zeros((h, w, 3), dtype=np.uint8)
bgr[20:200, 80:400] = (30, 90, 200)
payload = bgr_to_i420(bgr)
assert len(payload) == i420_nbytes(w, h)
mask = np.zeros((h, w), dtype=np.float32)
mask[20:200, 80:400] = 1.0
out = apply_i420(payload, w, h, mask, radius=9)
assert len(out) == i420_nbytes(w, h)
back = i420_to_bgr(out, w, h)
assert back.shape == (h, w, 3)
def test_missing_mask_blurs_whole_frame():
w, h = 48, 32
bgr = _checker(h, w)
payload = bgr_to_i420(bgr)
out = apply_i420(payload, w, h, None, radius=7)
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
y1 = np.frombuffer(out, dtype=np.uint8, count=w * h).reshape(h, w)
assert float(y1.var()) < float(y0.var()) * 0.6
def test_apply_i420_blurs_y_background():
w, h = 64, 48
bgr = np.zeros((h, w, 3), dtype=np.uint8)
bgr[::3, :] = 255
bgr[:, ::3] = 255
bgr[8:40, 16:48] = (20, 200, 20)
payload = bgr_to_i420(bgr)
mask = np.zeros((h, w), dtype=np.float32)
mask[8:40, 16:48] = 1.0
out = apply_i420(payload, w, h, mask, radius=9)
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
y1 = np.frombuffer(out, dtype=np.uint8, count=w * h).reshape(h, w)
assert float(y1[0:6, :].var()) < float(y0[0:6, :].var()) * 0.6
assert abs(int(y1[12:36, 20:44].mean()) - int(y0[12:36, 20:44].mean())) < 12
def test_c920_uses_portrait_dir_rally_stays_raw():
assert i420_source_path("cam-01", "/run/raw", "/run/por") == "/run/por/cam-01.i420"
assert i420_source_path("cam-15", "/run/raw", "/run/por") == "/run/por/cam-15.i420"
assert i420_source_path("rally", "/run/raw", "/run/por") == "/run/raw/rally.i420"
assert i420_source_path("cam-01", "/run/raw", "") == "/run/raw/cam-01.i420"
def test_letterbox_mask_maps_back_to_16x9():
from portrait import letterbox_bgr, unletterbox_mask
bgr = np.zeros((360, 640, 3), dtype=np.uint8)
canvas, box = letterbox_bgr(bgr, 256)
assert canvas.shape == (256, 256, 3)
x0, y0, nw, nh = box
assert nw > nh
m = np.zeros((256, 256), dtype=np.float32)
cy, cx = y0 + nh // 2, x0 + nw // 2
m[cy - 2:cy + 3, cx - 2:cx + 3] = 1.0
full = unletterbox_mask(m, box, 360, 640)
assert full.shape == (360, 640)
assert float(full[180, 320]) > 0.5
assert float(full[2, 2]) < 0.1
def test_selfie_cpu_sees_warrior_not_empty():
import cv2
from pathlib import Path
from portrait import PersonSeg
rgb = np.load(Path(__file__).parent / "fixtures" / "movenet_warrior_192.npy")
bgr = cv2.cvtColor(rgb, cv2.COLOR_RGB2BGR)
seg = PersonSeg()
assert seg.device == "CPU"
mask = seg.infer_bgr(bgr)
assert person_present(mask)
empty = np.full((360, 640, 3), 40, dtype=np.uint8)
assert person_present(seg.infer_bgr(empty)) is False
def test_pose_protect_keeps_wrists_and_face():
from portrait import pose_protect_mask
k = np.zeros((17, 3), dtype=np.float32)
k[0] = [0.30, 0.50, 0.9] # nose
k[9] = [0.12, 0.18, 0.9] # L wrist
k[10] = [0.12, 0.82, 0.9] # R wrist
m = pose_protect_mask(100, 200, k)
assert float(m[12, 36]) > 0.7
assert float(m[12, 164]) > 0.7
assert float(m[30, 100]) > 0.7
assert float(m[90, 10]) < 0.2
def test_union_masks_keeps_face_and_body():
from portrait import union_masks
body = np.zeros((40, 40), dtype=np.float32)
body[20:36, 10:30] = 1.0
face = np.zeros((40, 40), dtype=np.float32)
face[4:14, 14:26] = 1.0
out = union_masks(body, face)
assert float(out[8, 20]) == 1.0
assert float(out[28, 20]) == 1.0
assert float(out[2, 2]) == 0.0
def test_lighter_blur_keeps_more_background_detail():
w, h = 64, 48
bgr = np.zeros((h, w, 3), dtype=np.uint8)
bgr[::3, :] = 255
bgr[:, ::3] = 255
payload = bgr_to_i420(bgr)
mask = np.zeros((h, w), dtype=np.float32)
mask[8:40, 16:48] = 1.0
heavy = apply_i420(payload, w, h, mask, radius=15)
light = apply_i420(payload, w, h, mask, radius=7)
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
yh = np.frombuffer(heavy, dtype=np.uint8, count=w * h).reshape(h, w)
yl = np.frombuffer(light, dtype=np.uint8, count=w * h).reshape(h, w)
bg0 = float(y0[0:6, :].var())
assert float(yl[0:6, :].var()) > float(yh[0:6, :].var())
assert float(yl[0:6, :].var()) < bg0
+87
View File
@@ -0,0 +1,87 @@
"""SHM round-trip for portrait daemon with a fake mask."""
import time
import numpy as np
from frame_shm import LatestFrameReader, LatestFrameWriter, wait_for_ready
from portrait import bgr_to_i420, i420_nbytes, i420_to_bgr, person_present
from portrait_daemon import CamSlot, tick_once
def test_tick_blurs_bg_and_keeps_person(tmp_path):
w, h = 64, 48
bgr = np.zeros((h, w, 3), dtype=np.uint8)
bgr[::3, :] = 255
bgr[:, ::3] = 255
bgr[8:40, 16:48] = (20, 200, 20)
payload = bgr_to_i420(bgr)
src = tmp_path / "raw" / "cam-01.i420"
dst = tmp_path / "por" / "cam-01.i420"
writer = LatestFrameWriter(str(src), w, h)
writer.write(payload)
mask = np.zeros((h, w), dtype=np.float32)
mask[8:40, 16:48] = 1.0
def infer(_bgr):
return mask
slot = CamSlot(
identity="cam-01",
width=w,
height=h,
reader=LatestFrameReader(str(src)),
writer=LatestFrameWriter(str(dst), w, h),
infer_hz=30.0,
hold_s=0.8,
blur_px=9,
)
n = tick_once([slot], infer, time.monotonic(), infer_budget_s=0.05)
assert n == 1
out_r = LatestFrameReader(str(dst))
buf = bytearray(i420_nbytes(w, h))
seq = out_r.copy_into(buf)
assert seq == 1
out = i420_to_bgr(buf, w, h)
roi = out[12:36, 20:44]
assert abs(int(roi[:, :, 1].mean()) - 200) < 25
bg = out[0:6, :]
assert float(bg.var()) < float(bgr[0:6, :].var()) * 0.6
def test_tick_blurs_when_empty(tmp_path):
w, h = 32, 32
bgr = np.zeros((h, w, 3), dtype=np.uint8)
bgr[::3, :] = 255
bgr[:, ::3] = 255
payload = bgr_to_i420(bgr)
src = tmp_path / "raw" / "cam-02.i420"
dst = tmp_path / "por" / "cam-02.i420"
LatestFrameWriter(str(src), w, h).write(payload)
def infer(_bgr):
return np.zeros((h, w), dtype=np.float32)
slot = CamSlot(
identity="cam-02",
width=w,
height=h,
reader=LatestFrameReader(str(src)),
writer=LatestFrameWriter(str(dst), w, h),
infer_hz=30.0,
hold_s=0.2,
blur_px=9,
)
tick_once([slot], infer, time.monotonic(), infer_budget_s=0.05)
buf = bytearray(i420_nbytes(w, h))
assert LatestFrameReader(str(dst)).copy_into(buf) == 1
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
y1 = np.frombuffer(buf, dtype=np.uint8, count=w * h).reshape(h, w)
assert float(y1.var()) < float(y0.var()) * 0.6
assert person_present(np.zeros((h, w), dtype=np.float32)) is False
def test_wait_ready_helper(tmp_path):
p = tmp_path / "ready"
assert wait_for_ready(str(p), timeout=0.05) is False
p.write_text("ok\n")
assert wait_for_ready(str(p), timeout=0.2) is True
+36
View File
@@ -0,0 +1,36 @@
import pytest
from pathlib import Path
from discovery import Camera
from rally import dante_capture_card, discover_rally, rally_video_node
from run import fleet_cameras
from usb_ids import RALLY_IDENTITY
_HAVE_RALLY = rally_video_node() is not None
@pytest.mark.skipif(not _HAVE_RALLY, reason="Rally V-R0010 not on USB")
def test_rally_node_is_pinned_or_uvc():
node = rally_video_node()
assert node is not None
assert Path(node).exists()
@pytest.mark.skipif(not _HAVE_RALLY, reason="Rally V-R0010 not on USB")
def test_discover_rally_not_a_cam_slot():
cam = discover_rally()
assert cam is not None
assert cam.identity == RALLY_IDENTITY
assert cam.width == 1920
assert cam.height == 1080
assert cam.identity != "cam-01"
card = dante_capture_card()
assert cam.audio_card == card
def test_fleet_cameras_drops_rally():
cams = [
Camera(index=0, identity="cam-01", video_device="/dev/cam01", audio_card=1),
Camera(index=20, identity="rally", video_device="/dev/rally", audio_card=0),
]
assert [c.identity for c in fleet_cameras(cams)] == ["cam-01"]
+172
View File
@@ -0,0 +1,172 @@
"""YuNet face gates for Rally follow — reject ceiling/stand/noise."""
from __future__ import annotations
import numpy as np
from rally_follow import FaceGate, accept_yunet_face, select_yunet_face
def _face(
*,
x=400.0,
y=200.0,
w=80.0,
h=100.0,
score=0.93,
re=None,
le=None,
nose=None,
rm=None,
lm=None,
) -> np.ndarray:
"""YuNet 15-vector: box + 5 landmarks + score, in 960x540 detect space."""
if re is None:
re = (x + 0.30 * w, y + 0.38 * h)
if le is None:
le = (x + 0.70 * w, y + 0.38 * h)
if nose is None:
nose = (x + 0.50 * w, y + 0.55 * h)
if rm is None:
rm = (x + 0.35 * w, y + 0.78 * h)
if lm is None:
lm = (x + 0.65 * w, y + 0.78 * h)
row = np.zeros(15, dtype=np.float32)
row[0:4] = (x, y, w, h)
row[4:6] = re
row[6:8] = le
row[8:10] = nose
row[10:12] = rm
row[12:14] = lm
row[14] = score
return row
FW, FH = 960, 540
def test_good_face_accepted():
assert accept_yunet_face(_face(), FW, FH) is True
def test_tiny_blob_rejected():
# Live FP: 25x34 @ 0.805, nose not between the eyes.
row = _face(x=512, y=250, w=25, h=34, score=0.805,
re=(528.5, 264), le=(532.1, 263), nose=(536.1, 271),
rm=(529.7, 277), lm=(532.1, 278))
assert accept_yunet_face(row, FW, FH) is False
def test_low_score_rejected():
assert accept_yunet_face(_face(score=0.61), FW, FH) is False
def test_ceiling_band_rejected():
row = _face(x=400, y=2, w=80, h=100)
assert accept_yunet_face(row, FW, FH) is False
def test_stand_band_rejected():
row = _face(x=400, y=470, w=80, h=100)
assert accept_yunet_face(row, FW, FH) is False
def test_nose_not_between_eyes_rejected():
row = _face(re=(430, 238), le=(450, 238), nose=(470, 255))
assert accept_yunet_face(row, FW, FH) is False
def test_upside_down_landmarks_rejected():
row = _face(
re=(424, 270), le=(456, 270), nose=(440, 240),
rm=(428, 220), lm=(452, 220),
)
assert accept_yunet_face(row, FW, FH) is False
def test_select_skips_large_invalid_for_smaller_valid():
junk = _face(x=10, y=2, w=80, h=100, score=0.99) # ceiling
good = _face(x=400, y=200, w=80, h=100, score=0.88)
picked = select_yunet_face(np.stack([junk, good]), FW, FH)
assert picked is not None
assert float(picked[0]) == 400.0
def test_select_empty():
assert select_yunet_face(None, FW, FH) is None
assert select_yunet_face(np.zeros((0, 15)), FW, FH) is None
def test_face_gate_needs_confirm():
gate = FaceGate(need=3, max_jump=0.18)
box = (400, 200, 80, 100)
assert gate.update(box, 960, 540) is None
assert gate.update(box, 960, 540) is None
assert gate.update(box, 960, 540) == box
def test_face_gate_jump_resets():
gate = FaceGate(need=3, max_jump=0.18)
a = (400, 200, 80, 100)
b = (800, 400, 80, 100)
assert gate.update(a, 960, 540) is None
assert gate.update(a, 960, 540) is None
assert gate.update(b, 960, 540) is None # new target
assert gate.update(b, 960, 540) is None
assert gate.update(b, 960, 540) == b
def test_face_gate_loss_clears():
gate = FaceGate(need=2, max_jump=0.18)
box = (400, 200, 80, 100)
gate.update(box, 960, 540)
assert gate.update(box, 960, 540) == box
assert gate.update(None, 960, 540) is None
assert gate.update(box, 960, 540) is None
def test_profile_face_accepted():
"""Head turned: eyes cluster, nose not between them — still a face."""
row = _face(
x=400, y=200, w=80, h=100, score=0.90,
re=(418, 238), le=(426, 237),
nose=(410, 255),
rm=(416, 278), lm=(428, 277),
)
assert accept_yunet_face(row, FW, FH) is True
def test_head_hold_keeps_box_on_roi_motion():
from rally_follow import HeadHold
hold = HeadHold(hold_s=1.0)
box = (400, 200, 80, 100)
t = 10.0
assert hold.update(box, moving=False, now=t) == box
assert hold.update(None, moving=True, now=t + 0.3) == box
assert hold.update(None, moving=False, now=t + 0.4) == box # coast
assert hold.update(None, moving=False, now=t + 2.0) is None
def test_roi_motion_ignores_outside_last_box():
from rally_follow import roi_motion
prev = np.zeros((540, 960), dtype=np.uint8)
now = prev.copy()
now[10:40, 10:40] = 200 # motion far from the face
assert roi_motion(prev, now, (400, 200, 80, 100)) is False
now2 = prev.copy()
now2[210:260, 420:460] = 200
assert roi_motion(prev, now2, (400, 200, 80, 100)) is True
def test_stale_hold_freezes_does_not_keep_panning():
"""Last image-space box is not a pan target — that derails as the camera moves."""
from rally_follow import lock_action, steer_speeds
left = (80, 400, 160, 200)
pan, tilt = steer_speeds(left, 1920, 1080, deadband=0.14)
assert pan == -1
assert lock_action(confirmed=left, held=left) == "steer"
assert lock_action(confirmed=None, held=left) == "freeze"
assert lock_action(confirmed=None, held=None) == "lost"
cx = 1920 / 2 - 40
mid = (int(cx - 80), 440, 160, 200)
pan, tilt = steer_speeds(mid, 1920, 1080, deadband=0.14)
assert pan == 0 and tilt == 0
+11
View File
@@ -0,0 +1,11 @@
from rally_follow import video_cmd
def test_video_cmd_without_preview_has_no_kmssink():
cmd = video_cmd("/dev/rally", 1920, 1080, 30, 8, preview=None)
s = " ".join(cmd)
assert "kmssink" not in s
assert "tee" not in cmd
assert "fdsink" in cmd
assert "fd=8" in cmd
assert "device=/dev/rally" in s
+103
View File
@@ -0,0 +1,103 @@
from config import Config
from discovery import Camera
from run import worker_args_for
import json
import jwt
def _cfg(**kw) -> Config:
return Config(url="ws://127.0.0.1:7880", api_key="k", api_secret="s", **kw)
def _cam(identity: str = "cam-07") -> Camera:
return Camera(index=0, identity=identity, video_device="/dev/cam07", audio_card=2)
def test_worker_args_legacy_omits_optional_latency_flags():
args = worker_args_for(_cfg(), _cam())
assert "--audio-queue-ms" not in args
assert "--video-hold-s" not in args
assert "--split-executor" not in args
assert args[args.index("--enhance-mode") + 1] == "auto"
def test_worker_args_latency_flags_and_non_speaker_enhance_off():
cfg = _cfg(
audio_queue_ms=60,
video_hold_s=0.0,
worker_split_executor=True,
enhance_speaker_only=True,
display_speaker_camera="cam-01",
enhance_mode="auto",
)
args = worker_args_for(cfg, _cam("cam-07"))
assert args[args.index("--audio-queue-ms") + 1] == "60"
assert args[args.index("--video-hold-s") + 1] == "0.0"
assert "--split-executor" in args
assert args[args.index("--enhance-mode") + 1] == "off"
assert args[args.index("--beamform-mode") + 1] == "off"
def test_worker_args_speaker_keeps_enhance_mode():
cfg = _cfg(
enhance_mode="force",
enhance_speaker_only=True,
display_speaker_camera="cam-07",
)
args = worker_args_for(cfg, _cam("cam-07"))
assert args[args.index("--enhance-mode") + 1] == "force"
assert args[args.index("--beamform-mode") + 1] == "auto"
def test_worker_token_carries_uwh_tag():
args = worker_args_for(_cfg(), _cam("cam-07"))
token = args[args.index("--token") + 1]
claims = jwt.decode(token, "s", algorithms=["HS256"])
assert claims["attributes"]["tag"] == "uwh"
assert claims["attributes"]["tags"] == "uwh"
def test_worker_args_passes_enhance_socket():
args = worker_args_for(_cfg(enhance_socket="/tmp/e.sock"), _cam())
assert args[args.index("--enhance-socket") + 1] == "/tmp/e.sock"
def test_worker_args_omits_enhance_socket_when_empty():
args = worker_args_for(_cfg(enhance_socket=""), _cam())
assert "--enhance-socket" not in args
def test_worker_args_passes_video_shm():
args = worker_args_for(_cfg(), _cam("cam-07"))
assert args[args.index("--video-shm") + 1] == "/run/livekit-cameras/raw/cam-07.i420"
def test_worker_args_omits_video_shm_when_capture_daemon_off():
args = worker_args_for(_cfg(capture_daemon=False), _cam())
assert "--video-shm" not in args
def test_publisher_args_single_process_and_speaker_only():
from run import publisher_args_for
cfg = _cfg(enhance_speaker_only=True, display_speaker_camera="cam-01")
args = publisher_args_for(cfg, [_cam("cam-07"), _cam("cam-01")])
assert "--enhance-speaker-only" in args
assert "--audio-gate" not in args
payload = json.loads(args[args.index("--cameras-json") + 1])
assert {p["identity"] for p in payload} == {"cam-07", "cam-01"}
assert all("token" in p for p in payload)
assert args[args.index("--video-shm-dir") + 1] == "/run/livekit-cameras/raw"
assert args[args.index("--audio-shm-dir") + 1] == "/run/livekit-cameras/pcm"
def test_publisher_args_rally_keeps_1080p_dims():
from run import publisher_args_for
rally = Camera(index=20, identity="rally", video_device="/dev/rally",
audio_card=0, width=1920, height=1080)
args = publisher_args_for(_cfg(), [_cam("cam-01"), rally])
payload = json.loads(args[args.index("--cameras-json") + 1])
by_id = {p["identity"]: p for p in payload}
assert by_id["rally"]["width"] == 1920
assert by_id["rally"]["height"] == 1080
assert by_id["cam-01"]["width"] is None
+199
View File
@@ -0,0 +1,199 @@
"""Green speaking chrome on the KMS grid (LiveKit active-speaker)."""
import numpy as np
from display_grid import FRAME_RAISE, FRAME_SPEAK, CameraGrid, bgra_to_i420, bgra_to_yuv_pixel
def test_set_speakers_paints_green_i420_border():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
src = np.zeros((360, 640, 4), dtype=np.uint8)
src[:] = (40, 40, 40, 255)
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
before = np.frombuffer(g.render_i420(), dtype=np.uint8).copy()
g.set_speakers(["cam-07"])
assert g.speaking("cam-07") is True
after = np.frombuffer(g.render_i420(), dtype=np.uint8)
assert not np.array_equal(before, after)
idx = g.identities.index("cam-07")
ox, oy = g.tile_origin(idx)
yplane = after[: g.width * g.height].reshape(g.height, g.width)
yv, _, _ = bgra_to_yuv_pixel(FRAME_SPEAK)
assert int(yplane[oy, ox + g.tile_w // 2]) == yv
# Extra thickness lives in the gutter, not on the feed.
assert int(yplane[oy - 2, ox + g.tile_w // 2]) == yv
ix, iy, iw, _ih = g.inner_rect(idx)
assert int(yplane[iy, ix + iw // 2]) != yv
def test_speaking_overrides_yellow_then_yellow_returns():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
src = np.zeros((360, 640, 4), dtype=np.uint8)
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
g.set_highlight("cam-07", True)
g.set_speakers(["cam-07"])
idx = g.identities.index("cam-07")
ox, oy = g.tile_origin(idx)
yplane = np.frombuffer(g.render_i420(), dtype=np.uint8)[: g.width * g.height].reshape(
g.height, g.width)
y_speak, _, _ = bgra_to_yuv_pixel(FRAME_SPEAK)
y_raise, _, _ = bgra_to_yuv_pixel(FRAME_RAISE)
assert int(yplane[oy, ox + g.tile_w // 2]) == y_speak
g.set_speakers([])
yplane = np.frombuffer(g.render_i420(), dtype=np.uint8)[: g.width * g.height].reshape(
g.height, g.width)
assert g.speaking("cam-07") is False
assert g.highlighted("cam-07") is True
assert int(yplane[oy, ox + g.tile_w // 2]) == y_raise
def test_clear_drops_speaking():
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
src = np.zeros((360, 640, 4), dtype=np.uint8)
g.set_frame_i420("cam-01", bgra_to_i420(src), 640, 360)
g.set_speakers(["cam-01"])
g.clear("cam-01")
assert g.speaking("cam-01") is False
class _P:
def __init__(self, identity, keep=True):
self.identity = identity
self.keep = keep
def test_active_speaker_ids_keeps_order_and_filter():
from display_grid import active_speaker_ids, primary_grid_speaker
speakers = [_P("cam-07"), _P("rally", False), _P("cam-03"), _P("")]
assert active_speaker_ids(speakers, want=lambda p: p.keep) == ["cam-07", "cam-03"]
ids = ["rally", "cam-03", "cam-07"]
assert primary_grid_speaker(ids, ["cam-01", "cam-03", "cam-07"]) == ["cam-03"]
assert primary_grid_speaker(["rally"], ["cam-01"]) == []
def test_select_grid_speakers_drops_floor_noise():
from display_grid import select_grid_speakers
assert select_grid_speakers(
[("cam-01", -34.0), ("cam-02", -36.0)], min_db=-24.0, margin_db=6.0,
) == []
def test_select_grid_speakers_bleed_keeps_louder_only():
from display_grid import select_grid_speakers
assert select_grid_speakers(
[("cam-07", -12.0), ("cam-08", -22.0)], min_db=-24.0, margin_db=6.0,
) == ["cam-07"]
def test_select_grid_speakers_two_talkers_both_kept():
from display_grid import select_grid_speakers
assert select_grid_speakers(
[("cam-07", -12.0), ("cam-08", -14.0)], min_db=-24.0, margin_db=6.0,
) == ["cam-07", "cam-08"]
def test_select_grid_speakers_normal_speech_above_mic_floor():
"""Conversational C920 speech is often ~-24 dBFS; absolute -20 missed it."""
from display_grid import select_grid_speakers
floors = {"cam-07": -34.0, "cam-08": -34.0}
assert select_grid_speakers(
[("cam-07", -24.0), ("cam-08", -33.0)],
min_db=-40.0, margin_db=6.0, floors=floors, rise_db=6.0,
) == ["cam-07"]
def test_select_grid_speakers_two_talkers_each_above_own_floor():
from display_grid import select_grid_speakers
floors = {"cam-07": -34.0, "cam-08": -36.0}
assert select_grid_speakers(
[("cam-07", -24.0), ("cam-08", -26.0)],
min_db=-40.0, margin_db=6.0, floors=floors, rise_db=6.0,
) == ["cam-07", "cam-08"]
def test_select_quiet_mic_speech_not_cut_by_abs_min():
"""Quiet C920 idle is ~-45; +3 dB speech is still below -40 dBFS."""
from display_grid import select_grid_speakers
floors = {"cam-10": -45.0, "cam-01": -37.0}
assert select_grid_speakers(
[("cam-10", -41.5), ("cam-01", -36.5)],
min_db=-40.0, margin_db=6.0, floors=floors, rise_db=3.0,
) == ["cam-10"]
def test_update_noise_floor_tracks_idle_not_speech():
from display_grid import update_noise_floor
fl = update_noise_floor(-34.0, -33.5, speaking=False, rise_db=4.0)
assert -34.0 < fl < -33.5
stuck = update_noise_floor(-34.0, -20.0, speaking=True, rise_db=4.0)
assert stuck == -34.0
def test_confirm_speakers_needs_two_hops_new_talker():
from display_grid import confirm_speakers
first, counts = confirm_speakers({}, ["cam-07"], need=2)
assert first == []
second, counts = confirm_speakers(counts, ["cam-07"], need=2)
assert second == ["cam-07"]
# already on: one hop of continued speech keeps chrome
stay, _ = confirm_speakers(counts, ["cam-07"], need=2, already=["cam-07"])
assert stay == ["cam-07"]
# a one-hop idle spike on another seat does not light
spike, _ = confirm_speakers({}, ["cam-01"], need=2, already=["cam-07"])
assert spike == []
def test_waveform_similarity_high_for_delayed_copy():
from display_grid import waveform_similarity
t = np.linspace(0, 1, 1600, endpoint=False)
a = np.sin(2 * np.pi * 220 * t).astype(np.float32)
b = np.roll(a * 0.4, 80)
assert waveform_similarity(a, b) > 0.85
def test_waveform_similarity_low_for_independent_signals():
from display_grid import waveform_similarity
rng = np.random.default_rng(0)
a = rng.normal(size=1600).astype(np.float32)
b = rng.normal(size=1600).astype(np.float32)
assert waveform_similarity(a, b) < 0.35
def test_collapse_bleed_drops_similar_quieter_mic():
from display_grid import collapse_bleed
t = np.linspace(0, 1, 1600, endpoint=False)
src = np.sin(2 * np.pi * 180 * t).astype(np.float32)
waves = {"cam-07": src, "cam-08": np.roll(src * 0.3, 40)}
levels = {"cam-07": -12.0, "cam-08": -18.0}
assert collapse_bleed(["cam-07", "cam-08"], waves, levels, corr_min=0.4) == ["cam-07"]
def test_collapse_bleed_keeps_two_unlike_talkers():
from display_grid import collapse_bleed
t = np.linspace(0, 1, 1600, endpoint=False)
a = np.sin(2 * np.pi * 180 * t).astype(np.float32)
b = np.sin(2 * np.pi * 440 * t + 1.3).astype(np.float32)
waves = {"cam-07": a, "cam-08": b}
levels = {"cam-07": -12.0, "cam-08": -13.0}
assert collapse_bleed(["cam-07", "cam-08"], waves, levels, corr_min=0.6) == [
"cam-07", "cam-08",
]
def test_hold_speakers_adds_second_talker_immediately():
from display_grid import hold_speakers
cur, until = hold_speakers(
["cam-07"], ["cam-07", "cam-08"], prev_peak=-12.0, new_peak=-12.0,
now=10.0, hold_until=10.5, hold_s=0.6, switch_db=3.0)
assert cur == ["cam-07", "cam-08"]
def test_hold_speakers_ignores_similar_bleed_flicker():
from display_grid import hold_speakers
cur, until = hold_speakers(
["cam-07"], ["cam-08"], prev_peak=-12.0, new_peak=-12.5,
now=10.2, hold_until=10.8, hold_s=0.6, switch_db=3.0)
assert cur == ["cam-07"]
cur, _ = hold_speakers(
["cam-07"], ["cam-08"], prev_peak=-12.0, new_peak=-8.0,
now=10.2, hold_until=10.8, hold_s=0.6, switch_db=3.0)
assert cur == ["cam-08"]
+24
View File
@@ -0,0 +1,24 @@
import jwt
from tokens import make_token
def test_make_token_includes_uwh_attributes():
token = make_token(
"ws://127.0.0.1:7880", "devkey", "secret", "cameras", "cam-01",
attributes={"tag": "uwh", "tags": "uwh"},
)
claims = jwt.decode(token, "secret", algorithms=["HS256"])
assert claims["sub"] == "cam-01"
video = claims.get("video") or {}
assert video.get("canPublish") is True
assert video.get("canSubscribe") is False
assert claims["attributes"]["tag"] == "uwh"
assert claims["attributes"]["tags"] == "uwh"
def test_make_token_omits_attributes_when_empty():
token = make_token(
"ws://127.0.0.1:7880", "devkey", "secret", "cameras", "cam-01")
claims = jwt.decode(token, "secret", algorithms=["HS256"])
assert "attributes" not in claims or not claims.get("attributes")
+15
View File
@@ -0,0 +1,15 @@
from usb_ids import (
HUDDLY_VID,
LOGITECH_VID,
is_c920_pid,
is_fleet_video,
is_rally_pid,
)
def test_c920_is_fleet_huddly_and_rally_are_not():
assert is_c920_pid(0x08E5)
assert is_fleet_video(LOGITECH_VID, 0x08E5)
assert not is_fleet_video(HUDDLY_VID, 0x0021)
assert not is_fleet_video(LOGITECH_VID, 0x0881)
assert is_rally_pid(0x0881)
+16
View File
@@ -0,0 +1,16 @@
from uvc_exposure import pin_cmd
def test_pin_cmd_order_manual_then_time_then_dynfps():
cmd = pin_cmd("/dev/cam07")
assert cmd[:3] == ["v4l2-ctl", "-d", "/dev/cam07"]
assert cmd[3] == "--set-ctrl=auto_exposure=1"
assert cmd[4] == "--set-ctrl=exposure_time_absolute=333"
assert cmd[5] == "--set-ctrl=exposure_dynamic_framerate=0"
def test_pin_cmd_auto_skips_inactive_time():
cmd = pin_cmd("/dev/cam07", auto_exposure=3, dynamic_framerate=1)
assert "--set-ctrl=auto_exposure=3" in cmd
assert "--set-ctrl=exposure_dynamic_framerate=1" in cmd
assert not any(c.startswith("--set-ctrl=exposure_time_absolute=") for c in cmd)
+38
View File
@@ -0,0 +1,38 @@
from vaapi_pin import (
classify_render_nodes,
igpu_render,
pin_encode_env,
should_pin_vaapi,
)
def test_classify_by_pci_even_if_card_numbers_swap():
def pci_id_of(node: str) -> str:
return {
"/dev/dri/renderD128": "0xe20b",
"/dev/dri/renderD129": "0xa780",
}[node]
c = classify_render_nodes(
["/dev/dri/renderD128", "/dev/dri/renderD129"],
pci_id_of=pci_id_of,
)
assert c["arc"] == "/dev/dri/renderD128"
assert c["igpu"] == "/dev/dri/renderD129"
def test_pin_encode_env_points_at_igpu_not_arc():
env = pin_encode_env({}, igpu="/dev/dri/renderD129")
assert env["LIBVA_DRIVER_NAME"] == "iHD"
assert env["LIVEKIT_ENCODE_RENDER"] == "/dev/dri/renderD129"
assert env.get("LIBVA_DRM_DEVICE") != "/dev/dri/renderD128"
def test_should_not_pin_vaapi_when_preencoded():
assert should_pin_vaapi(preencoded=True) is False
assert should_pin_vaapi(preencoded=False) is True
def test_igpu_render_matches_live_a780():
node = igpu_render()
assert node.startswith("/dev/dri/renderD")
+69
View File
@@ -0,0 +1,69 @@
"""VideoSource in-flight cap + VAAPI stale-frame drop (encoder behind)."""
from latency import (
should_drop_encoder_queue_depth,
should_drop_stale_vaapi_frame,
video_capture_gate_acquire,
video_capture_gate_release,
)
def test_vaapi_keeps_fresh_delta_frame():
# 30 fps, 33 ms old — still inside the 3-frame budget.
assert should_drop_stale_vaapi_frame(
now_us=1_000_000, capture_us=967_000, fps=30, max_frames_behind=3,
is_keyframe=False,
) is False
def test_vaapi_drops_delta_older_than_three_frame_periods():
# 30 fps × 3 = 100 ms. 150 ms on EncoderQueue → drop before ToI420.
assert should_drop_stale_vaapi_frame(
now_us=1_000_000, capture_us=850_000, fps=30, max_frames_behind=3,
is_keyframe=False,
) is True
def test_vaapi_never_drops_keyframe():
assert should_drop_stale_vaapi_frame(
now_us=1_000_000, capture_us=1, fps=30, max_frames_behind=3,
is_keyframe=True,
) is False
def test_vaapi_missing_timestamp_is_not_dropped():
assert should_drop_stale_vaapi_frame(
now_us=1_000_000, capture_us=0, fps=30, max_frames_behind=3,
is_keyframe=False,
) is False
def test_gate_admits_until_max_in_flight():
gate = {"in_flight": 0, "dropped": 0, "max_in_flight": 3}
assert video_capture_gate_acquire(gate) is True
assert video_capture_gate_acquire(gate) is True
assert video_capture_gate_acquire(gate) is True
assert gate["in_flight"] == 3
assert video_capture_gate_acquire(gate) is False
assert gate["dropped"] == 1
assert gate["in_flight"] == 3
def test_gate_release_allows_another_capture():
gate = {"in_flight": 3, "dropped": 0, "max_in_flight": 3}
video_capture_gate_release(gate)
assert gate["in_flight"] == 2
assert video_capture_gate_acquire(gate) is True
assert gate["in_flight"] == 3
def test_queue_depth_keeps_slot_when_empty():
assert should_drop_encoder_queue_depth(queued=0, max_queued=1) is False
def test_queue_depth_drops_when_at_max():
# Latest-frame cap: one in-flight OnFrame/Encode, overwrite don't push.
assert should_drop_encoder_queue_depth(queued=1, max_queued=1) is True
def test_queue_depth_drops_when_over_max():
assert should_drop_encoder_queue_depth(queued=3, max_queued=1) is True
+52
View File
@@ -0,0 +1,52 @@
"""Publisher must not catch up frames and must recycle native encoders on RSS."""
from latency import (
next_video_capture,
should_recycle_encoders,
)
def test_next_video_capture_sleeps_until_due():
capture, due = next_video_capture(now=10.0, due=10.05, period=1 / 30)
assert capture is False
assert due == 10.05
def test_next_video_capture_fires_on_due_and_advances_one_period():
capture, due = next_video_capture(now=10.0, due=10.0, period=0.05)
assert capture is True
assert due == 10.05
def test_next_video_capture_skips_catchup_when_behind():
# 80 ms late at 30 fps would be two extra ticks if we kept adding period.
capture, due = next_video_capture(now=10.08, due=10.0, period=1 / 30)
assert capture is True
assert due == 10.08 + (1 / 30)
def test_should_recycle_when_anon_rss_exceeds_limit_and_interval_elapsed():
assert should_recycle_encoders(
rss_anon_kb=4_000_000, limit_kb=3_000_000,
now=100.0, last_recycle_at=0.0, min_interval_s=60.0,
) is True
def test_should_not_recycle_below_limit():
assert should_recycle_encoders(
rss_anon_kb=1_000_000, limit_kb=3_000_000,
now=100.0, last_recycle_at=0.0, min_interval_s=60.0,
) is False
def test_should_not_recycle_inside_min_interval():
assert should_recycle_encoders(
rss_anon_kb=5_000_000, limit_kb=3_000_000,
now=100.0, last_recycle_at=80.0, min_interval_s=60.0,
) is False
def test_should_not_recycle_when_limit_disabled():
assert should_recycle_encoders(
rss_anon_kb=9_000_000, limit_kb=0,
now=100.0, last_recycle_at=0.0, min_interval_s=60.0,
) is False
+25
View File
@@ -0,0 +1,25 @@
from visca_xu_bridge import parse_visca, visca_frames
def test_visca_split_frames():
frames, rest = visca_frames(bytes([0x81, 0x01, 0x06, 0x04, 0xFF, 0x81]))
assert frames == [bytes([0x81, 0x01, 0x06, 0x04, 0xFF])]
assert rest == bytes([0x81])
def test_visca_stop_and_move():
stop = parse_visca(bytes([0x81, 0x01, 0x06, 0x01, 0x15, 0x15, 0x03, 0x03, 0xFF]))
assert stop["action"] == "stop"
left = parse_visca(bytes([0x81, 0x01, 0x06, 0x01, 0x08, 0x08, 0x01, 0x03, 0xFF]))
assert left == {"action": "move", "pan": -1.0, "tilt": 0.0, "zoom": 0.0}
up = parse_visca(bytes([0x81, 0x01, 0x06, 0x01, 0x08, 0x08, 0x03, 0x01, 0xFF]))
assert up["tilt"] == 1.0
def test_visca_zoom_home_preset():
assert parse_visca(bytes([0x81, 0x01, 0x04, 0x07, 0x00, 0xFF]))["action"] == "zoom_stop"
tele = parse_visca(bytes([0x81, 0x01, 0x04, 0x07, 0x27, 0xFF]))
assert tele == {"action": "zoom", "zoom": 1.0}
assert parse_visca(bytes([0x81, 0x01, 0x06, 0x04, 0xFF]))["action"] == "home"
rec = parse_visca(bytes([0x81, 0x01, 0x04, 0x3F, 0x02, 0x01, 0xFF]))
assert rec == {"action": "preset_recall", "index": 1}
+5 -2
View File
@@ -4,7 +4,8 @@ from livekit.api import AccessToken, VideoGrants
def make_token(url: str, api_key: str, api_secret: str,
room: str, identity: str) -> str:
room: str, identity: str,
attributes: dict[str, str] | None = None) -> str:
token = (
AccessToken(api_key, api_secret)
.with_identity(identity)
@@ -13,10 +14,12 @@ def make_token(url: str, api_key: str, api_secret: str,
room_join=True,
room=room,
can_publish=True,
can_subscribe=True,
can_subscribe=False,
can_publish_data=True,
))
)
if attributes:
token = token.with_attributes(attributes)
return token.to_jwt()
+62
View File
@@ -0,0 +1,62 @@
"""USB identities for the fleet vs the (future) speaker Rally."""
from __future__ import annotations
from pathlib import Path
LOGITECH_VID = 0x046D
# HD Pro C920 (this fleet) and the older 082d id.
C920_PIDS = frozenset({0x08E5, 0x082D})
# Logi Rally Camera 0881; Rally Bar / Mini / Huddle variants.
RALLY_PIDS = frozenset({0x0881, 0x087C, 0x089B, 0x08D3})
RALLY_IDENTITY = "rally"
# Huddly IQ on the PCH/Dante hub — not a cam-NN slot.
HUDDLY_VID = 0x2BD9
HUDDLY_PIDS = frozenset({0x0021})
def usb_vid_pid_from_sysfs(sysfs_device: str) -> tuple[int, int] | None:
p = Path(sysfs_device)
for _ in range(12):
vend = p / "idVendor"
prod = p / "idProduct"
if vend.is_file() and prod.is_file():
try:
return int(vend.read_text().strip(), 16), int(prod.read_text().strip(), 16)
except ValueError:
return None
if str(p).startswith("/sys/devices"):
p = p.parent
else:
return None
return None
def usb_vid_pid_of_video_node(video_node: str) -> tuple[int, int] | None:
link = Path(f"/sys/class/video4linux/{Path(video_node).name}/device")
try:
real = str(link.resolve())
except OSError:
return None
return usb_vid_pid_from_sysfs(real)
def is_c920_pid(pid: int) -> bool:
return int(pid) in C920_PIDS
def is_huddly_iq(vid: int, pid: int) -> bool:
return int(vid) == HUDDLY_VID and int(pid) in HUDDLY_PIDS
def is_fleet_video(vid: int, pid: int) -> bool:
"""True only for the 20 C920s. Rally, Huddly IQ, etc. stay out of cam-NN."""
return int(vid) == LOGITECH_VID and int(pid) in C920_PIDS
def is_rally_pid(pid: int) -> bool:
return int(pid) in RALLY_PIDS
def looks_like_rally_name(name: str) -> bool:
n = (name or "").lower()
return "rally" in n
+63
View File
@@ -0,0 +1,63 @@
"""Pin Logitech C920 exposure so 360p stays 30 fps.
Aperture-priority + exposure_dynamic_framerate=1 lets the sensor sit at
~66 ms / gain 255 and drop to 15 fps. Manual 333 (0.1 ms units = 1/30 s)
with dyn-fps off holds the gst 30 fps request.
"""
from __future__ import annotations
import logging
import os
import subprocess
log = logging.getLogger("cameras.uvc-exposure")
# UVC: 1 = Manual Mode, 3 = Aperture Priority. Time unit is 0.1 ms.
AUTO_EXPOSURE_MANUAL = 1
EXPOSURE_TIME_30FPS = 333
def pin_cmd(
device: str,
*,
auto_exposure: int = AUTO_EXPOSURE_MANUAL,
exposure_time: int = EXPOSURE_TIME_30FPS,
dynamic_framerate: int = 0,
) -> list[str]:
cmd = [
"v4l2-ctl",
"-d",
str(device),
f"--set-ctrl=auto_exposure={int(auto_exposure)}",
f"--set-ctrl=exposure_dynamic_framerate={int(dynamic_framerate)}",
]
# Time is inactive under aperture-priority (3); setting it fails the whole pin.
if int(auto_exposure) == AUTO_EXPOSURE_MANUAL:
cmd.insert(4, f"--set-ctrl=exposure_time_absolute={int(exposure_time)}")
return cmd
def pin_from_env() -> tuple[int, int, int]:
ae = int(os.environ.get("VIDEO_AUTO_EXPOSURE", str(AUTO_EXPOSURE_MANUAL)))
exp = int(os.environ.get("VIDEO_EXPOSURE_TIME", str(EXPOSURE_TIME_30FPS)))
dyn = int(os.environ.get("VIDEO_EXPOSURE_DYNAMIC", "0"))
return ae, exp, dyn
def pin_c920_exposure(device: str) -> None:
ae, exp, dyn = pin_from_env()
cmd = pin_cmd(device, auto_exposure=ae, exposure_time=exp, dynamic_framerate=dyn)
try:
r = subprocess.run(cmd, capture_output=True, text=True, timeout=5)
except OSError as exc:
log.warning("pin %s failed: %s", device, exc)
return
if r.returncode:
log.warning("pin %s rc=%s %s", device, r.returncode, (r.stderr or "").strip())
return
log.info("pinned %s ae=%s time=%s dynfps=%s", device, ae, exp, dyn)
def pin_devices(devices: list[str]) -> None:
for dev in devices:
pin_c920_exposure(dev)
+104
View File
@@ -0,0 +1,104 @@
"""DRM render-node identity for this host.
Arc B580 PCI id 0xe20b (today often renderD128). iGPU UHD 770 PCI id 0xa780
(today often renderD129). Card indexes swap; never hardcode which is Arc.
Encode belongs on the iGPU. The Arc is kmssink wall only.
LiveKit PreEncoded publish does not need a VA bind-mount.
"""
from __future__ import annotations
import logging
import os
from pathlib import Path
log = logging.getLogger("cameras.vaapi")
ARC_PCI = "0xe20b"
IGPU_PCI = "0xa780"
def _pci_id_of_render(node: str) -> str:
name = Path(node).name
p = Path("/sys/class/drm") / name / "device" / "device"
try:
return p.read_text().strip().lower()
except OSError:
return ""
def classify_render_nodes(
nodes: list[str] | None = None,
pci_id_of=None,
) -> dict:
reader = pci_id_of or _pci_id_of_render
if nodes is None:
dri = Path("/dev/dri")
nodes = sorted(str(p) for p in dri.glob("renderD*")) if dri.is_dir() else []
arc = igpu = None
ids: dict[str, str] = {}
for n in nodes:
pid = (reader(n) or "").strip().lower()
if not pid.startswith("0x"):
try:
pid = hex(int(pid, 16))
except ValueError:
pass
ids[n] = pid
if pid == ARC_PCI:
arc = n
elif pid == IGPU_PCI:
igpu = n
return {"arc": arc, "igpu": igpu, "ids": ids}
def igpu_render() -> str:
found = classify_render_nodes()["igpu"]
if found:
return found
fallback = "/dev/dri/renderD129"
log.warning("iGPU PCI %s not found; falling back to %s", IGPU_PCI, fallback)
return fallback
def arc_render() -> str:
found = classify_render_nodes()["arc"]
if found:
return found
fallback = "/dev/dri/renderD128"
log.warning("Arc PCI %s not found; falling back to %s", ARC_PCI, fallback)
return fallback
def pin_encode_env(env: dict | None = None, igpu: str | None = None) -> dict:
out = dict(os.environ if env is None else env)
node = igpu or igpu_render()
out["LIBVA_DRIVER_NAME"] = "iHD"
out["GST_VA_ALL_DRIVERS"] = "1"
out["LIVEKIT_ENCODE_RENDER"] = node
out.pop("LIBVA_DRM_DEVICE", None)
return out
def should_pin_vaapi(preencoded: bool) -> bool:
return not bool(preencoded)
def pin_arc_vaapi_env(env: dict | None = None) -> dict:
"""Deprecated: PreEncoded does not encode in libwebrtc."""
out = dict(os.environ if env is None else env)
out["LIBVA_DRIVER_NAME"] = "iHD"
out["GST_VA_ALL_DRIVERS"] = "1"
return out
def apply_env(env: dict) -> None:
for k, v in env.items():
os.environ[k] = v
def bind_arc_over_igpu_render(arc: str | None = None, igpu: str | None = None) -> bool:
"""Deprecated leftover. Do not call from capture or publisher."""
del arc, igpu
log.warning("bind_arc_over_igpu_render is deprecated; skipped")
return False
+222
View File
@@ -0,0 +1,222 @@
"""VISCA-IP façade for Logitech Rally (UVC + XU), not native VISCA.
Listens on TCP (default 127.0.0.1:5678), accepts raw VISCA frames ending in
0xFF, and drives pan/tilt/zoom through v4l2. AutoPTZ should use backend
visca_ip, address 127.0.0.1:5678, raw framing. visca_usb is pyserial — do
not point it at this camera.
"""
from __future__ import annotations
import argparse
import logging
import select
import socket
import subprocess
import threading
from typing import Callable
log = logging.getLogger("cameras.visca_xu")
# Sony/PTZOptics ACK + completion for a command addressed as 81.
_ACK = bytes([0x90, 0x41, 0xFF])
_COMP = bytes([0x90, 0x51, 0xFF])
_SYNTAX = bytes([0x90, 0x60, 0x02, 0xFF])
def visca_frames(buf: bytes) -> tuple[list[bytes], bytes]:
"""Split a byte stream into 0xFF-terminated frames; return (frames, rest)."""
frames: list[bytes] = []
start = 0
for i, b in enumerate(buf):
if b == 0xFF:
frames.append(buf[start:i + 1])
start = i + 1
return frames, buf[start:]
def parse_visca(frame: bytes) -> dict:
"""Parse one VISCA command. Unknown frames get action='unknown'."""
f = bytes(frame)
if len(f) < 5 or f[-1] != 0xFF:
return {"action": "invalid"}
# 81 01 06 01 VV WW PP QQ FF pan/tilt drive
if len(f) >= 9 and f[1:4] == bytes([0x01, 0x06, 0x01]):
pan_dir, tilt_dir = f[6], f[7]
pan = 0.0
tilt = 0.0
if pan_dir == 0x01:
pan = -1.0
elif pan_dir == 0x02:
pan = 1.0
if tilt_dir == 0x01:
tilt = 1.0
elif tilt_dir == 0x02:
tilt = -1.0
if pan_dir == 0x03 and tilt_dir == 0x03:
return {"action": "stop"}
return {"action": "move", "pan": pan, "tilt": tilt, "zoom": 0.0}
# 81 01 04 07 ZZ FF zoom
if len(f) >= 6 and f[1:4] == bytes([0x01, 0x04, 0x07]):
z = f[4]
if z == 0x00:
return {"action": "zoom_stop"}
if (z & 0xF0) == 0x20:
return {"action": "zoom", "zoom": 1.0}
if (z & 0xF0) == 0x30:
return {"action": "zoom", "zoom": -1.0}
return {"action": "zoom_stop"}
# 81 01 06 04 FF home
if f[1:4] == bytes([0x01, 0x06, 0x04]):
return {"action": "home"}
# 81 01 04 3F 01/02 MM FF preset set/recall
if len(f) >= 7 and f[1:4] == bytes([0x01, 0x04, 0x3F]):
if f[4] == 0x02:
return {"action": "preset_recall", "index": int(f[5])}
if f[4] == 0x01:
return {"action": "preset_set", "index": int(f[5])}
# inquiries: answer none (get_position returns None on AutoPTZ)
if len(f) >= 5 and f[1] == 0x09:
return {"action": "inquiry"}
return {"action": "unknown"}
class RallyV4L2:
"""Thin v4l2-ctl wrapper. Rally exposes generic pan/tilt CIDs on this host."""
def __init__(self, device: str = "/dev/rally"):
self.device = device
def _ctl(self, *args: str) -> str:
r = subprocess.run(
["v4l2-ctl", "-d", self.device, *args],
capture_output=True, text=True, check=False)
if r.returncode != 0:
log.warning("v4l2-ctl rc=%s %s", r.returncode, r.stderr.strip())
return r.stdout
def get(self) -> dict[str, int]:
out = self._ctl("--get-ctrl=pan_absolute,tilt_absolute,zoom_absolute")
vals: dict[str, int] = {}
for line in out.splitlines():
if ":" not in line:
continue
k, v = line.split(":", 1)
try:
vals[k.strip()] = int(v.strip())
except ValueError:
pass
return vals
def set_speed(self, pan: int, tilt: int) -> None:
pan = max(-1, min(1, int(pan)))
tilt = max(-1, min(1, int(tilt)))
self._ctl(f"--set-ctrl=pan_speed={pan},tilt_speed={tilt}")
def stop(self) -> None:
self.set_speed(0, 0)
def nudge_zoom(self, direction: int, step: int = 20) -> None:
cur = self.get().get("zoom_absolute", 100)
nxt = max(100, min(1500, cur + (step if direction > 0 else -step)))
self._ctl(f"--set-ctrl=zoom_absolute={nxt}")
def home(self) -> None:
self.stop()
self._ctl("--set-ctrl=pan_absolute=0,tilt_absolute=0,zoom_absolute=100")
def apply(self, cmd: dict) -> None:
action = cmd.get("action")
if action in ("stop", "zoom_stop"):
self.stop()
elif action == "move":
pan = cmd.get("pan") or 0.0
tilt = cmd.get("tilt") or 0.0
self.set_speed(
-1 if pan < -0.01 else (1 if pan > 0.01 else 0),
-1 if tilt < -0.01 else (1 if tilt > 0.01 else 0),
)
elif action == "zoom":
z = cmd.get("zoom") or 0.0
if abs(z) < 0.01:
return
self.nudge_zoom(1 if z > 0 else -1)
elif action == "home":
self.home()
def serve(
host: str,
port: int,
apply: Callable[[dict], None],
stop: threading.Event | None = None,
) -> None:
srv = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
srv.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
srv.bind((host, port))
srv.listen(4)
srv.settimeout(0.5)
log.info("VISCA-IP façade %s:%d (raw) -> Rally v4l2", host, port)
clients: dict[socket.socket, bytearray] = {}
try:
while stop is None or not stop.is_set():
rlist = [srv, *clients]
try:
ready, _, _ = select.select(rlist, [], [], 0.5)
except InterruptedError:
continue
for s in ready:
if s is srv:
conn, addr = srv.accept()
conn.setblocking(False)
clients[conn] = bytearray()
log.info("visca client %s", addr)
continue
try:
data = s.recv(4096)
except BlockingIOError:
continue
except OSError:
data = b""
if not data:
clients.pop(s, None)
s.close()
continue
clients[s].extend(data)
frames, rest = visca_frames(bytes(clients[s]))
clients[s] = bytearray(rest)
for fr in frames:
cmd = parse_visca(fr)
if cmd.get("action") in ("invalid", "inquiry", "unknown"):
if cmd.get("action") == "invalid":
s.sendall(_SYNTAX)
continue
try:
apply(cmd)
except Exception: # noqa: BLE001
log.exception("apply failed %s", cmd)
try:
s.sendall(_ACK + _COMP)
except OSError:
pass
finally:
for c in list(clients):
c.close()
srv.close()
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--device", default="/dev/rally")
ap.add_argument("--host", default="127.0.0.1")
ap.add_argument("--port", type=int, default=5678)
args = ap.parse_args(argv)
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [visca-xu] %(levelname)s %(message)s")
rally = RallyV4L2(args.device)
serve(args.host, args.port, rally.apply)
return 0
if __name__ == "__main__":
raise SystemExit(main())