feat: point the fleet at wss LiveKit room uwh-telhai
Publish and subscribe over wss://livekit.uni-wh.de:7800 and refuse cleartext ws://. Conference room is uwh-telhai. Includes the uncommitted encoded H.264 publish path, Rally hairpin, and KMS wall overlay.
This commit is contained in:
+64
-8
@@ -1,13 +1,19 @@
|
||||
# LiveKit server connection
|
||||
# Local TEST server (autostarted). Switch this URL to the remote
|
||||
# LiveKit later; then: systemctl disable --now livekit-server
|
||||
LIVEKIT_URL=ws://127.0.0.1:7880
|
||||
# Remote LiveKit. Local livekit-server.service is disabled.
|
||||
# Signaling must be wss://. The client verifies the TLS cert (do not use ws://).
|
||||
LIVEKIT_URL=wss://livekit.uni-wh.de:7800
|
||||
LIVEKIT_API_KEY=APIKey_xxx
|
||||
LIVEKIT_API_SECRET=APIsecret_xxx
|
||||
|
||||
# Room / naming
|
||||
LIVEKIT_ROOM=cameras
|
||||
# Conference room. Publishers and the KMS wall must use this same name.
|
||||
LIVEKIT_ROOM=uwh-telhai
|
||||
PARTICIPANT_PREFIX=cam
|
||||
# Published on local camera participants as LiveKit attributes tag/tags.
|
||||
PARTICIPANT_TAGS=uwh
|
||||
# Display wall: empty = show local cameras (filter OFF). Set to uwh to hide
|
||||
# local tagged participants so remotes fill the wall instead.
|
||||
DISPLAY_HIDE_TAGS=
|
||||
|
||||
# Streams
|
||||
# "all" = every discovered camera, "1,3,5" or "0-19" = subset
|
||||
@@ -16,7 +22,9 @@ VIDEO_WIDTH=640
|
||||
VIDEO_HEIGHT=360
|
||||
VIDEO_FPS=30
|
||||
VIDEO_MIN_FPS=5
|
||||
VIDEO_BITRATE=800000
|
||||
VIDEO_BITRATE=1200000
|
||||
# Rally 1080p encode (bps). Rollback if encode/gst dies: 12000000 then restart cameras.
|
||||
VIDEO_RALLY_BITRATE=20000000
|
||||
VIDEO_CODEC=h264
|
||||
VIDEO_ENCODER=vaapi
|
||||
LIBVA_DRM_DEVICE=/dev/dri/renderD129
|
||||
@@ -43,8 +51,33 @@ ENHANCE_INFER=auto
|
||||
BEAMFORM_MODE=auto
|
||||
# One BLAS/Torch thread per worker (20 workers vs 28 cores)
|
||||
TORCH_NUM_THREADS=1
|
||||
# Recompute GCC-PHAT TDOA every N hops (5 * 200 ms = 1 s)
|
||||
# Recompute GCC-PHAT TDOA every N hops (5 * hop)
|
||||
BEAMFORM_TDOA_EVERY=5
|
||||
# One MetricGAN process for all cameras (workers send hops over a unix socket).
|
||||
# ENHANCE_DAEMON=0 reverts to per-worker model load (~10 GiB extra RAM).
|
||||
ENHANCE_DAEMON=1
|
||||
ENHANCE_SOCKET=/tmp/livekit-enhance.sock
|
||||
|
||||
# --- optional latency knobs (defaults keep current behavior) ---
|
||||
# AUDIO_QUEUE_MS=60 # LiveKit AudioSource queue; unset = max(150, hop_ms+50)
|
||||
# VIDEO_HOLD_S=0 # hold published video this many seconds; unset = match hop; 0 = no hold
|
||||
# WORKER_SPLIT_EXECUTOR=1 # dedicated thread pools: ffmpeg reads vs enhance
|
||||
# ENHANCE_SPEAKER_ONLY=1 # MetricGAN+beamform only on DISPLAY_SPEAKER_CAMERA (rally)
|
||||
# AUDIO_GATE=1 # optional hop mute; off by default so two talkers both publish
|
||||
# AUDIO_GATE_OPEN_DB=-28
|
||||
# AUDIO_GATE_CLOSE_DB=-34
|
||||
# AUDIO_GATE_HOLD_S=0.3
|
||||
# DISPLAY_VIDEO_CAPACITY=1 # VideoStream ring; 0 = unbounded, 1 = latest frame
|
||||
# DISPLAY_VIDEO_FORMAT=bgra # request BGRA from SDK (skip extra convert)
|
||||
# DISPLAY_GST_QUEUE_BUFFERS=1
|
||||
# DISPLAY_SPEAKER_PUSH=arrival # timer | arrival (push speaker pane on decode)
|
||||
#
|
||||
# Suggested A/B (wall, ~hop 50 ms, speaker-only enhance):
|
||||
# ENHANCE_HOP_S=0.05 AUDIO_QUEUE_MS=60 WORKER_SPLIT_EXECUTOR=1
|
||||
# ENHANCE_SPEAKER_ONLY=1 DISPLAY_VIDEO_CAPACITY=1 DISPLAY_VIDEO_FORMAT=bgra
|
||||
# DISPLAY_GST_QUEUE_BUFFERS=1 DISPLAY_SPEAKER_PUSH=arrival
|
||||
# Aggressive (no neural enhance, 20 ms hops):
|
||||
# ENHANCE_MODE=off BEAMFORM_MODE=off ENHANCE_HOP_S=0.02 AUDIO_QUEUE_MS=40 VIDEO_HOLD_S=0
|
||||
|
||||
# Misc
|
||||
PUBLISH_TIMEOUT_S=60
|
||||
@@ -57,12 +90,35 @@ DISPLAY_ROLES=grid,speaker,screenshare
|
||||
# Empty = auto (connected heads first, prefer Arc HDMI-A-*)
|
||||
DISPLAY_CONNECTORS=
|
||||
DISPLAY_IDENTITY=display-wall
|
||||
# Speaker camera pane: cam-NN or "active"
|
||||
DISPLAY_SPEAKER_CAMERA=cam-01
|
||||
# Speaker HDMI pane: LiveKit identity of the Logitech Rally (not a C920).
|
||||
# The Rally is not on this host yet — pane shows a placeholder until `rally` joins.
|
||||
DISPLAY_SPEAKER_CAMERA=rally
|
||||
DISPLAY_GRID_COLS=5
|
||||
DISPLAY_GRID_ROWS=4
|
||||
DISPLAY_GRID_WIDTH=1920
|
||||
DISPLAY_GRID_HEIGHT=1080
|
||||
DISPLAY_GRID_FPS=30
|
||||
# Yellow tile border when a participant raises a hand (MoveNet on the wall).
|
||||
# DISPLAY_HAND_RAISE=0 to disable. Displays-only restart.
|
||||
DISPLAY_HAND_RAISE=1
|
||||
DISPLAY_HAND_RAISE_HZ=4
|
||||
DISPLAY_HAND_RAISE_HOLD_S=1.0
|
||||
# DISPLAY_HAND_RAISE_MODEL=thunder # thunder (256) or lightning (192); displays-only
|
||||
# Green speaker chrome: LiveKit candidates, hop RMS picks louder vs two talkers.
|
||||
# DISPLAY_SPEAK_MIN_DB=-40
|
||||
# DISPLAY_SPEAK_RISE_DB=3
|
||||
# DISPLAY_SPEAK_MARGIN_DB=6
|
||||
# DISPLAY_SPEAK_HOLD_S=0.6
|
||||
# DISPLAY_SPEAK_CORR=0.6 # hop xcorr; similar waveforms = bleed, keep louder
|
||||
# DISPLAY_SPEAK_CONFIRM=2 # consecutive hops before a new seat greens
|
||||
# Screen share is a stub until production LiveKit is online.
|
||||
LIVEKIT_PUBLIC_URL=
|
||||
|
||||
# Person-aware background blur on C920 publish (OpenVINO CPU, not iGPU/Arc).
|
||||
# Empty seats passthrough. Rally is not blurred. Rollback: PORTRAIT_BLUR=0 then
|
||||
# cameras-only restart.
|
||||
PORTRAIT_BLUR=1
|
||||
# PORTRAIT_HZ=8
|
||||
# PORTRAIT_BLUR_PX=7
|
||||
# PORTRAIT_HOLD_S=0.8
|
||||
# PORTRAIT_SHM_DIR=/run/livekit-cameras/portrait
|
||||
|
||||
@@ -5,9 +5,13 @@ LiveKit room.
|
||||
|
||||
Each camera becomes one LiveKit participant (`cam-01` .. `cam-20`) that
|
||||
publishes:
|
||||
- a camera video track (640x360 @ 30 fps MJPG, H.264 VAAPI — sized
|
||||
for low-latency wall decode, not max quality)
|
||||
- a microphone audio track (16 kHz mono, cleaned with speechbrain)
|
||||
- a camera video track (640x360 @ 30 fps MJPG, H.264 VAAPI on Arc).
|
||||
1280x720@30 and 1920x1080@30 negotiate, but UVC on the 6–7 camera
|
||||
USB 2.0 hubs fails buffer-pool allocation for most devices. Keep 30
|
||||
fps; drop resolution if USB isoc is exhausted. YUYV 1080p is 5 fps
|
||||
only — always MJPEG.
|
||||
- a microphone audio track (16 kHz mono, cleaned with one shared
|
||||
speechbrain MetricGAN process; workers do not each load the model)
|
||||
|
||||
Identity is the USB serial, pinned by udev (`/dev/camNN` + `slots.json`).
|
||||
`--cameras 1,5` means participants `cam-01` and `cam-05`, not discovery order.
|
||||
@@ -96,7 +100,10 @@ journalctl -u livekit-displays -f
|
||||
```
|
||||
|
||||
`.env`: `DISPLAY_ROLES`, `DISPLAY_CONNECTORS`, `DISPLAY_SPEAKER_CAMERA`,
|
||||
`DISPLAY_GRID_*`. kmssink takes the DRM master (framebuffer console on
|
||||
`DISPLAY_GRID_*`. Local cameras publish LiveKit attributes `tag=uwh`
|
||||
(`PARTICIPANT_TAGS`). The wall can hide those with `DISPLAY_HIDE_TAGS=uwh`
|
||||
so remotes fill the mosaic instead; leave `DISPLAY_HIDE_TAGS` empty to keep
|
||||
showing the local fleet (current default). kmssink takes the DRM master (framebuffer console on
|
||||
HDMI-A-7 is replaced by the grid).
|
||||
|
||||
## Audio cleanup
|
||||
@@ -116,6 +123,13 @@ stereo through a SpeechBrain preprocessor, then published as mono:
|
||||
The 1 s window is **past context**, not +1 s delay: only the latest hop
|
||||
is published. Algorithmic delay ≈ hop + compute.
|
||||
|
||||
Optional latency flags (all default to current behavior). Suggested wall
|
||||
A/B: `ENHANCE_HOP_S=0.05 AUDIO_QUEUE_MS=60 WORKER_SPLIT_EXECUTOR=1
|
||||
ENHANCE_SPEAKER_ONLY=1 DISPLAY_VIDEO_CAPACITY=1 DISPLAY_VIDEO_FORMAT=bgra
|
||||
DISPLAY_GST_QUEUE_BUFFERS=1 DISPLAY_SPEAKER_PUSH=arrival`. Aggressive:
|
||||
`ENHANCE_MODE=off BEAMFORM_MODE=off ENHANCE_HOP_S=0.02 AUDIO_QUEUE_MS=40
|
||||
VIDEO_HOLD_S=0`. See `.env.example`.
|
||||
|
||||
Each worker caps Torch/BLAS/OpenVINO to `TORCH_NUM_THREADS=1` (default)
|
||||
so 14–20 processes do not oversubscribe the i7-14700. This host has an
|
||||
Intel Arc B580; the venv Torch is NVIDIA CUDA and **cannot** use it. Do
|
||||
@@ -144,20 +158,15 @@ This box is meant to behave like an appliance:
|
||||
|
||||
## systemd (enabled)
|
||||
|
||||
- `livekit-server.service` — **local TEST** LiveKit on :7880 (autostart for now)
|
||||
- `livekit-cameras.service` — publishers. `Wants=` the test server, does not
|
||||
`Require` it. Destination is `LIVEKIT_URL` in `.env` (currently
|
||||
`ws://127.0.0.1:7880`).
|
||||
- `livekit-displays.service` — optional. Subscribes to `cam-01`/`cam-02`/`cam-03`
|
||||
- `livekit-server.service` — **local TEST** LiveKit on :7880 (**disabled**;
|
||||
cameras no longer `Wants=` it). Do not re-enable unless falling back
|
||||
to localhost.
|
||||
- `livekit-cameras.service` — publishers. Destination is `LIVEKIT_URL` in
|
||||
`.env` (`wss://livekit.uni-wh.de:7800`, room `uwh-telhai`). Signaling
|
||||
is WSS only; the client verifies the TLS certificate.
|
||||
- `livekit-displays.service` — optional. Subscribes to the camera room
|
||||
and drives 3 DRM connectors with GStreamer `kmssink` (no X11/Wayland).
|
||||
|
||||
When the real/remote LiveKit exists, change `.env` `LIVEKIT_URL` and:
|
||||
|
||||
```bash
|
||||
systemctl disable --now livekit-server
|
||||
systemctl restart livekit-cameras
|
||||
```
|
||||
|
||||
```bash
|
||||
systemctl status livekit-cameras livekit-server
|
||||
journalctl -u livekit-cameras -f
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
"""Split Annex-B H.264 into access units."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def _start_at(buf: bytes, i: int) -> int:
|
||||
n = len(buf)
|
||||
while i + 2 < n:
|
||||
if buf[i] == 0 and buf[i + 1] == 0:
|
||||
if buf[i + 2] == 1:
|
||||
return i
|
||||
if i + 3 < n and buf[i + 2] == 0 and buf[i + 3] == 1:
|
||||
return i
|
||||
i += 1
|
||||
return -1
|
||||
|
||||
|
||||
def nal_unit_type(nal: bytes) -> int:
|
||||
"""First NAL type in an AU (5 = IDR)."""
|
||||
i = _start_at(nal, 0)
|
||||
if i < 0:
|
||||
return 0
|
||||
hdr = i + (4 if i + 3 < len(nal) and nal[i + 2] == 0 else 3)
|
||||
if hdr >= len(nal):
|
||||
return 0
|
||||
return nal[hdr] & 0x1F
|
||||
|
||||
|
||||
def is_keyframe(au: bytes) -> bool:
|
||||
i = 0
|
||||
n = len(au)
|
||||
while True:
|
||||
s = _start_at(au, i)
|
||||
if s < 0:
|
||||
return False
|
||||
hdr = s + (4 if s + 3 < n and au[s + 2] == 0 else 3)
|
||||
if hdr >= n:
|
||||
return False
|
||||
ntype = au[hdr] & 0x1F
|
||||
if ntype == 5:
|
||||
return True
|
||||
i = hdr + 1
|
||||
|
||||
|
||||
class AnnexBSplitter:
|
||||
"""Feed byte-stream; yield complete AUs (including start codes)."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._buf = bytearray()
|
||||
|
||||
def push(self, data: bytes) -> list[bytes]:
|
||||
if data:
|
||||
self._buf.extend(data)
|
||||
out: list[bytes] = []
|
||||
buf = self._buf
|
||||
first = _start_at(bytes(buf), 0)
|
||||
if first < 0:
|
||||
if len(buf) > 1_000_000:
|
||||
del buf[: len(buf) - 8]
|
||||
return out
|
||||
if first:
|
||||
del buf[:first]
|
||||
pos = 0
|
||||
while True:
|
||||
nxt = _start_at(bytes(buf), pos + 3)
|
||||
if nxt < 0:
|
||||
break
|
||||
au = bytes(buf[:nxt])
|
||||
if au:
|
||||
out.append(au)
|
||||
del buf[:nxt]
|
||||
pos = 0
|
||||
return out
|
||||
+52
-1
@@ -32,6 +32,8 @@ import time
|
||||
|
||||
import numpy as np
|
||||
|
||||
from audio_gate import hop_mono_int16
|
||||
|
||||
log = logging.getLogger("audio.cleanup")
|
||||
|
||||
SAMPLE_RATE = 16000
|
||||
@@ -66,7 +68,10 @@ class AudioCleaner:
|
||||
beamform: str = "auto",
|
||||
torch_num_threads: int = 1,
|
||||
tdoa_every: int = 5,
|
||||
enhance_infer: str = "auto"):
|
||||
enhance_infer: str = "auto",
|
||||
enhance_socket: str = "",
|
||||
identity: str = "",
|
||||
gate=None):
|
||||
self.mode = mode
|
||||
self.beamform_mode = beamform
|
||||
self.sample_rate = sample_rate
|
||||
@@ -93,6 +98,19 @@ class AudioCleaner:
|
||||
self._ov_compiled = None
|
||||
self._ov_request = None
|
||||
self._ov_input = None
|
||||
self.enhance_socket = (enhance_socket or "").strip()
|
||||
self.identity = identity or ""
|
||||
self._client = None
|
||||
self._gate = gate
|
||||
|
||||
if self.enhance_socket and mode != "off":
|
||||
self._connect_daemon()
|
||||
if self._client is not None:
|
||||
return
|
||||
if mode == "force":
|
||||
raise RuntimeError(
|
||||
f"ENHANCE_MODE=force but enhance daemon "
|
||||
f"{self.enhance_socket!r} is unreachable")
|
||||
|
||||
self._load_torch()
|
||||
if beamform == "force":
|
||||
@@ -121,6 +139,19 @@ class AudioCleaner:
|
||||
|
||||
self._refresh_backend()
|
||||
|
||||
def _connect_daemon(self) -> None:
|
||||
from enhance_ipc import EnhanceClient
|
||||
|
||||
try:
|
||||
self._client = EnhanceClient(self.enhance_socket)
|
||||
except Exception as e: # noqa: BLE001
|
||||
log.warning("enhance daemon %s unreachable (%s)", self.enhance_socket, e)
|
||||
self._client = None
|
||||
return
|
||||
self.backend = "daemon"
|
||||
log.info("audio enhance via daemon %s identity=%s",
|
||||
self.enhance_socket, self.identity or "-")
|
||||
|
||||
def _refresh_backend(self) -> None:
|
||||
parts = []
|
||||
if self._enhancer is not None:
|
||||
@@ -323,6 +354,11 @@ class AudioCleaner:
|
||||
pcm = np.asarray(pcm)
|
||||
t0 = time.perf_counter()
|
||||
try:
|
||||
if self._gate is not None:
|
||||
n = pcm.shape[0]
|
||||
self._gate.apply(hop_mono_int16(pcm))
|
||||
if not self._gate.open:
|
||||
return np.zeros(n, dtype=np.int16)
|
||||
return self._process_inner(pcm)
|
||||
finally:
|
||||
self.last_dt_ms = (time.perf_counter() - t0) * 1000.0
|
||||
@@ -340,6 +376,21 @@ class AudioCleaner:
|
||||
n = frames.shape[0]
|
||||
x = frames.astype(np.float32) / 32768.0
|
||||
|
||||
if self._client is not None:
|
||||
try:
|
||||
out = self._client.process(self.identity, frames.astype(np.int16))
|
||||
out = np.asarray(out, dtype=np.int16)
|
||||
if out.shape[0] != n:
|
||||
if out.shape[0] > n:
|
||||
out = out[:n]
|
||||
else:
|
||||
out = np.pad(out, (0, n - out.shape[0]))
|
||||
return out
|
||||
except Exception as e: # noqa: BLE001
|
||||
log.warning("enhance daemon failed (%s); using fallback", e)
|
||||
mono_hop = frames.astype(np.float32).reshape(n, -1).mean(axis=1) / 32768.0
|
||||
return self._to_int16(self._simple_clean(mono_hop))
|
||||
|
||||
# Delay-and-Sum the *hop* (200 ms is ~60 ms here). Beamforming the
|
||||
# full 1 s window every hop is ~330 ms — not 20-camera real-time.
|
||||
# Same tutorial modules; TDOA is estimated per hop via GCC-PHAT.
|
||||
|
||||
+158
@@ -0,0 +1,158 @@
|
||||
"""Shared C920 mic capture: small gst-launch pool, PCM hops in mmap rings."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import select
|
||||
import signal
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
from audio_gst import GST_GROUP_SIZE, grouped_audio, shared_audio_cmd
|
||||
from hop_shm import HopWriter, hop_bytes
|
||||
|
||||
log = logging.getLogger("cameras.audio_daemon")
|
||||
READY_NAME = "ready"
|
||||
|
||||
|
||||
def _open_pipes(n: int) -> tuple[list[int], list[int]]:
|
||||
reads: list[int] = []
|
||||
writes: list[int] = []
|
||||
for _ in range(n):
|
||||
r, w = os.pipe()
|
||||
try:
|
||||
import fcntl
|
||||
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 65536)
|
||||
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 65536)
|
||||
except OSError:
|
||||
pass
|
||||
os.set_inheritable(w, True)
|
||||
os.set_inheritable(r, False)
|
||||
reads.append(r)
|
||||
writes.append(w)
|
||||
return reads, writes
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser(description="Shared ALSA capture daemon")
|
||||
ap.add_argument("--cameras-json", required=True)
|
||||
ap.add_argument("--rate", type=int, default=16000)
|
||||
ap.add_argument("--hop-s", type=float, default=0.1)
|
||||
ap.add_argument("--shm-dir", default="/run/livekit-cameras/pcm")
|
||||
ap.add_argument("--group-size", type=int, default=GST_GROUP_SIZE)
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [audio-daemon] %(levelname)s %(message)s")
|
||||
|
||||
cams = json.loads(args.cameras_json)
|
||||
pairs = [(c["identity"], int(c["audio_card"]))
|
||||
for c in cams if c.get("audio_card") is not None]
|
||||
if not pairs:
|
||||
log.error("no audio cards")
|
||||
return 2
|
||||
|
||||
need = hop_bytes(args.rate, 2, args.hop_s)
|
||||
os.makedirs(args.shm_dir, exist_ok=True)
|
||||
writers = [
|
||||
HopWriter(os.path.join(args.shm_dir, f"{ident}.pcm"), need)
|
||||
for ident, _card in pairs
|
||||
]
|
||||
reads, writes = _open_pipes(len(pairs))
|
||||
env = dict(os.environ)
|
||||
|
||||
groups = grouped_audio(pairs, writes, args.group_size)
|
||||
log.info("starting %d gst-launch for %d mics rate=%d hop=%.3f (group=%d)",
|
||||
len(groups), len(pairs), args.rate, args.hop_s, args.group_size)
|
||||
procs: list[subprocess.Popen] = []
|
||||
for g_cams, g_fds in groups:
|
||||
cmd = shared_audio_cmd(g_cams, args.rate, g_fds)
|
||||
procs.append(subprocess.Popen(
|
||||
cmd,
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=None,
|
||||
pass_fds=tuple(g_fds),
|
||||
env=env,
|
||||
close_fds=True,
|
||||
))
|
||||
for w in writes:
|
||||
os.close(w)
|
||||
for r in reads:
|
||||
os.set_blocking(r, False)
|
||||
|
||||
bufs = [bytearray() for _ in pairs]
|
||||
seqs = [0] * len(pairs)
|
||||
stop = False
|
||||
|
||||
def _stop(_signum=None, _frame=None):
|
||||
nonlocal stop
|
||||
stop = True
|
||||
|
||||
signal.signal(signal.SIGTERM, _stop)
|
||||
signal.signal(signal.SIGINT, _stop)
|
||||
|
||||
ready_path = os.path.join(args.shm_dir, READY_NAME)
|
||||
if os.path.exists(ready_path):
|
||||
os.unlink(ready_path)
|
||||
marked = False
|
||||
last_log = time.monotonic()
|
||||
fd_index = {r: i for i, r in enumerate(reads)}
|
||||
|
||||
try:
|
||||
while not stop:
|
||||
dead = [p for p in procs if p.poll() is not None]
|
||||
if dead:
|
||||
log.error("gst exited rc=%s", [p.returncode for p in dead])
|
||||
break
|
||||
ready, _, _ = select.select(reads, [], [], 0.2)
|
||||
for r in ready:
|
||||
i = fd_index[r]
|
||||
while True:
|
||||
try:
|
||||
chunk = os.read(r, max(need, need - len(bufs[i])))
|
||||
except BlockingIOError:
|
||||
break
|
||||
if not chunk:
|
||||
break
|
||||
bufs[i].extend(chunk)
|
||||
while len(bufs[i]) >= need:
|
||||
hop = bytes(bufs[i][:need])
|
||||
del bufs[i][:need]
|
||||
seqs[i] = writers[i].write(hop)
|
||||
if not marked and any(s > 0 for s in seqs):
|
||||
open(ready_path, "w").write("ok\n")
|
||||
marked = True
|
||||
log.info("ready first hops %s",
|
||||
",".join(f"{pairs[i][0]}={seqs[i]}" for i in range(len(pairs))))
|
||||
now = time.monotonic()
|
||||
if now - last_log >= 5.0:
|
||||
last_log = now
|
||||
log.info("seqs %s",
|
||||
" ".join(f"{pairs[i][0]}:{seqs[i]}" for i in range(len(pairs))))
|
||||
finally:
|
||||
for proc in procs:
|
||||
if proc.poll() is None:
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
for r in reads:
|
||||
try:
|
||||
os.close(r)
|
||||
except OSError:
|
||||
pass
|
||||
for w in writers:
|
||||
w.close()
|
||||
if os.path.exists(ready_path):
|
||||
os.unlink(ready_path)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,91 @@
|
||||
"""Per-hop noise gate for C920 mics (LiveKit active-speaker needs distinct energy)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
|
||||
FLOOR_DB = -120.0
|
||||
|
||||
|
||||
def rms_dbfs(pcm: np.ndarray) -> float:
|
||||
"""RMS of int16 PCM in dBFS. Empty/zero -> FLOOR_DB."""
|
||||
x = np.asarray(pcm)
|
||||
if x.size == 0:
|
||||
return FLOOR_DB
|
||||
rms = float(np.sqrt(np.mean(np.square(x.astype(np.float64) / 32768.0))))
|
||||
if rms <= 1e-12:
|
||||
return FLOOR_DB
|
||||
return 20.0 * float(np.log10(rms))
|
||||
|
||||
|
||||
def hop_mono_int16(pcm: np.ndarray) -> np.ndarray:
|
||||
x = np.asarray(pcm)
|
||||
if x.ndim == 2:
|
||||
return np.mean(x.astype(np.float32), axis=1).astype(np.int16)
|
||||
return np.asarray(x, dtype=np.int16)
|
||||
|
||||
|
||||
def gate_enabled_for(identity: str, speaker_camera: str, enabled: bool) -> bool:
|
||||
"""Dante/rally stays ungated; C920 seats are gated when enabled."""
|
||||
if not enabled:
|
||||
return False
|
||||
ident = (identity or "").strip()
|
||||
skip = (speaker_camera or "").strip()
|
||||
if ident and skip and ident == skip:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def hold_hops(hold_s: float, hop_s: float) -> int:
|
||||
hop = max(1e-6, float(hop_s))
|
||||
return max(1, int(round(float(hold_s) / hop)))
|
||||
|
||||
|
||||
def make_gate(
|
||||
identity: str,
|
||||
speaker_camera: str,
|
||||
enabled: bool,
|
||||
open_db: float,
|
||||
close_db: float,
|
||||
hold_s: float,
|
||||
hop_s: float,
|
||||
) -> HopGate | None:
|
||||
if not gate_enabled_for(identity, speaker_camera, enabled):
|
||||
return None
|
||||
return HopGate(
|
||||
open_db=open_db,
|
||||
close_db=close_db,
|
||||
hold_hops=hold_hops(hold_s, hop_s),
|
||||
)
|
||||
|
||||
|
||||
class HopGate:
|
||||
"""Open on close-talk energy; hangover then mute so neighbors stay silent."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
open_db: float = -32.0,
|
||||
close_db: float = -40.0,
|
||||
hold_hops: int = 3,
|
||||
) -> None:
|
||||
self.open_db = float(open_db)
|
||||
self.close_db = min(float(close_db), self.open_db)
|
||||
self.hold_hops = max(0, int(hold_hops))
|
||||
self.open = False
|
||||
self.hold = 0
|
||||
|
||||
def apply(self, pcm: np.ndarray) -> np.ndarray:
|
||||
pcm = np.asarray(pcm, dtype=np.int16)
|
||||
db = rms_dbfs(pcm)
|
||||
if db >= self.open_db:
|
||||
self.open = True
|
||||
self.hold = self.hold_hops
|
||||
return pcm
|
||||
if self.open:
|
||||
if db >= self.close_db:
|
||||
self.hold = self.hold_hops
|
||||
return pcm
|
||||
if self.hold > 0:
|
||||
self.hold -= 1
|
||||
return pcm
|
||||
self.open = False
|
||||
return np.zeros_like(pcm)
|
||||
@@ -0,0 +1,40 @@
|
||||
"""GStreamer ALSA capture graphs for the shared audio daemon."""
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
|
||||
GST_LAUNCH = shutil.which("gst-launch-1.0") or "gst-launch-1.0"
|
||||
GST_GROUP_SIZE = 5
|
||||
|
||||
|
||||
def shared_audio_cmd(
|
||||
cameras: list[tuple[str, int]],
|
||||
rate: int,
|
||||
fds: list[int],
|
||||
) -> list[str]:
|
||||
if len(cameras) != len(fds):
|
||||
raise ValueError("cameras and fds length mismatch")
|
||||
cmd: list[str] = [GST_LAUNCH, "-q"]
|
||||
caps = f"audio/x-raw,format=S16LE,rate={int(rate)},channels=2"
|
||||
for (_ident, card), fd in zip(cameras, fds):
|
||||
cmd += [
|
||||
"alsasrc", f"device=hw:{int(card)},0", "do-timestamp=false",
|
||||
"!", caps,
|
||||
"!", "queue", "max-size-buffers=2", "max-size-bytes=0",
|
||||
"max-size-time=0", "leaky=downstream",
|
||||
"!", "fdsink", f"fd={int(fd)}", "sync=false",
|
||||
]
|
||||
return cmd
|
||||
|
||||
|
||||
def grouped_audio(
|
||||
cameras: list[tuple[str, int]],
|
||||
fds: list[int],
|
||||
group_size: int = GST_GROUP_SIZE,
|
||||
) -> list[tuple[list[tuple[str, int]], list[int]]]:
|
||||
if group_size < 1:
|
||||
raise ValueError("group_size must be >= 1")
|
||||
out = []
|
||||
for i in range(0, len(cameras), group_size):
|
||||
out.append((cameras[i:i + group_size], fds[i:i + group_size]))
|
||||
return out
|
||||
+123
@@ -0,0 +1,123 @@
|
||||
"""Split AV1 low-overhead OBU streams into temporal units."""
|
||||
from __future__ import annotations
|
||||
|
||||
OBU_SEQUENCE_HEADER = 1
|
||||
OBU_TEMPORAL_DELIMITER = 2
|
||||
|
||||
|
||||
def _leb128(data: bytes, i: int) -> tuple[int, int] | None:
|
||||
n = len(data)
|
||||
result = 0
|
||||
shift = 0
|
||||
while i < n:
|
||||
byte = data[i]
|
||||
i += 1
|
||||
result |= (byte & 0x7F) << shift
|
||||
if (byte & 0x80) == 0:
|
||||
return result, i
|
||||
shift += 7
|
||||
if shift > 56:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def parse_obus(data: bytes) -> list[tuple[int, int, int]]:
|
||||
"""Return (type, start, end) for each complete OBU. Empty if malformed."""
|
||||
out: list[tuple[int, int, int]] = []
|
||||
i = 0
|
||||
n = len(data)
|
||||
while i < n:
|
||||
start = i
|
||||
hdr = data[i]
|
||||
i += 1
|
||||
if hdr & 0x80:
|
||||
return []
|
||||
obu_type = (hdr >> 3) & 0x0F
|
||||
has_ext = (hdr & 0x04) != 0
|
||||
has_size = (hdr & 0x02) != 0
|
||||
if has_ext:
|
||||
if i >= n:
|
||||
return []
|
||||
i += 1
|
||||
if has_size:
|
||||
leb = _leb128(data, i)
|
||||
if leb is None:
|
||||
return []
|
||||
size, i = leb
|
||||
end = i + size
|
||||
if end > n:
|
||||
return []
|
||||
i = end
|
||||
else:
|
||||
i = n
|
||||
out.append((obu_type, start, i))
|
||||
return out
|
||||
|
||||
|
||||
def is_keyframe(tu: bytes) -> bool:
|
||||
return any(t == OBU_SEQUENCE_HEADER for t, _s, _e in parse_obus(tu))
|
||||
|
||||
|
||||
class Av1TuSplitter:
|
||||
"""Feed obu-stream bytes; yield complete temporal units."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._buf = bytearray()
|
||||
|
||||
def push(self, data: bytes) -> list[bytes]:
|
||||
if data:
|
||||
self._buf.extend(data)
|
||||
out: list[bytes] = []
|
||||
buf = self._buf
|
||||
while True:
|
||||
n = len(buf)
|
||||
if n < 1:
|
||||
break
|
||||
i = 0
|
||||
tu_end = -1
|
||||
saw_coded = False
|
||||
ok = True
|
||||
while i < n:
|
||||
start = i
|
||||
hdr = buf[i]
|
||||
i += 1
|
||||
if hdr & 0x80:
|
||||
ok = False
|
||||
break
|
||||
obu_type = (hdr >> 3) & 0x0F
|
||||
has_ext = (hdr & 0x04) != 0
|
||||
has_size = (hdr & 0x02) != 0
|
||||
if has_ext:
|
||||
if i >= n:
|
||||
ok = False
|
||||
break
|
||||
i += 1
|
||||
if has_size:
|
||||
leb = _leb128(bytes(buf), i)
|
||||
if leb is None:
|
||||
ok = False
|
||||
break
|
||||
size, i = leb
|
||||
if i + size > n:
|
||||
ok = False
|
||||
break
|
||||
i = i + size
|
||||
else:
|
||||
ok = False
|
||||
break
|
||||
if obu_type == OBU_TEMPORAL_DELIMITER and saw_coded:
|
||||
tu_end = start
|
||||
break
|
||||
if obu_type != OBU_TEMPORAL_DELIMITER:
|
||||
saw_coded = True
|
||||
if not ok:
|
||||
if n > 2_000_000:
|
||||
del buf[: n - 16]
|
||||
break
|
||||
if tu_end < 0:
|
||||
break
|
||||
tu = bytes(buf[:tu_end])
|
||||
if tu:
|
||||
out.append(tu)
|
||||
del buf[:tu_end]
|
||||
return out
|
||||
@@ -0,0 +1,176 @@
|
||||
"""Shared C920 capture: small gst-launch pool, latest I420 frames in mmap."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import select
|
||||
import signal
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
from capture_va import (
|
||||
GST_GROUP_SIZE,
|
||||
grouped_cameras,
|
||||
shared_capture_cmd,
|
||||
)
|
||||
from frame_shm import LatestFrameWriter, frame_bytes
|
||||
from uvc_exposure import pin_devices
|
||||
|
||||
log = logging.getLogger("cameras.capture_daemon")
|
||||
|
||||
READY_NAME = "ready"
|
||||
|
||||
|
||||
def _open_pipes(n: int) -> tuple[list[int], list[int]]:
|
||||
reads: list[int] = []
|
||||
writes: list[int] = []
|
||||
for _ in range(n):
|
||||
r, w = os.pipe()
|
||||
try:
|
||||
import fcntl
|
||||
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 1048576)
|
||||
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 1048576)
|
||||
except OSError:
|
||||
pass
|
||||
os.set_inheritable(w, True)
|
||||
os.set_inheritable(r, False)
|
||||
reads.append(r)
|
||||
writes.append(w)
|
||||
return reads, writes
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser(description="Shared JPEG capture daemon")
|
||||
ap.add_argument("--cameras-json", required=True,
|
||||
help="JSON list of {identity, video_device}")
|
||||
ap.add_argument("--width", type=int, default=640)
|
||||
ap.add_argument("--height", type=int, default=360)
|
||||
ap.add_argument("--fps", type=int, default=30)
|
||||
ap.add_argument("--shm-dir", default="/run/livekit-cameras/raw")
|
||||
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
|
||||
ap.add_argument("--group-size", type=int, default=GST_GROUP_SIZE)
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [capture-daemon] %(levelname)s %(message)s")
|
||||
|
||||
cams = json.loads(args.cameras_json)
|
||||
pairs = [(c["identity"], c["video_device"]) for c in cams]
|
||||
if not pairs:
|
||||
log.error("no cameras")
|
||||
return 2
|
||||
|
||||
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
|
||||
|
||||
need = frame_bytes(args.width, args.height)
|
||||
os.makedirs(args.shm_dir, exist_ok=True)
|
||||
writers = [
|
||||
LatestFrameWriter(os.path.join(args.shm_dir, f"{ident}.i420"),
|
||||
args.width, args.height)
|
||||
for ident, _dev in pairs
|
||||
]
|
||||
reads, writes = _open_pipes(len(pairs))
|
||||
env = dict(os.environ)
|
||||
env["LIBVA_DRM_DEVICE"] = args.vaapi_device
|
||||
env["LIBVA_DRIVER_NAME"] = "iHD"
|
||||
env["GST_REGISTRY_UPDATE"] = "no"
|
||||
|
||||
groups = grouped_cameras(pairs, writes, args.group_size)
|
||||
log.info("starting %d gst-launch for %d cameras %dx%d@%d (group=%d)",
|
||||
len(groups), len(pairs), args.width, args.height, args.fps,
|
||||
args.group_size)
|
||||
procs: list[subprocess.Popen] = []
|
||||
for g_cams, g_fds in groups:
|
||||
cmd = shared_capture_cmd(g_cams, args.width, args.height, args.fps, g_fds)
|
||||
procs.append(subprocess.Popen(
|
||||
cmd,
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=None,
|
||||
pass_fds=tuple(g_fds),
|
||||
env=env,
|
||||
close_fds=True,
|
||||
))
|
||||
for w in writes:
|
||||
os.close(w)
|
||||
for r in reads:
|
||||
os.set_blocking(r, False)
|
||||
pin_devices([dev for _ident, dev in pairs])
|
||||
|
||||
bufs = [bytearray() for _ in pairs]
|
||||
seqs = [0] * len(pairs)
|
||||
stop = False
|
||||
|
||||
def _stop(_signum=None, _frame=None):
|
||||
nonlocal stop
|
||||
stop = True
|
||||
|
||||
signal.signal(signal.SIGTERM, _stop)
|
||||
signal.signal(signal.SIGINT, _stop)
|
||||
|
||||
ready_path = os.path.join(args.shm_dir, READY_NAME)
|
||||
if os.path.exists(ready_path):
|
||||
os.unlink(ready_path)
|
||||
marked = False
|
||||
last_log = time.monotonic()
|
||||
fd_index = {r: i for i, r in enumerate(reads)}
|
||||
|
||||
try:
|
||||
while not stop:
|
||||
dead = [p for p in procs if p.poll() is not None]
|
||||
if dead:
|
||||
log.error("gst exited rc=%s", [p.returncode for p in dead])
|
||||
break
|
||||
ready, _, _ = select.select(reads, [], [], 0.2)
|
||||
now_ns = time.time_ns()
|
||||
for r in ready:
|
||||
i = fd_index[r]
|
||||
while True:
|
||||
try:
|
||||
chunk = os.read(r, max(need, need - len(bufs[i])))
|
||||
except BlockingIOError:
|
||||
break
|
||||
if not chunk:
|
||||
break
|
||||
bufs[i].extend(chunk)
|
||||
while len(bufs[i]) >= need:
|
||||
frame = bytearray(bufs[i][:need])
|
||||
del bufs[i][:need]
|
||||
seqs[i] = writers[i].write(frame, now_ns)
|
||||
if not marked and any(s > 0 for s in seqs):
|
||||
pin_devices([dev for _ident, dev in pairs])
|
||||
open(ready_path, "w").write("ok\n")
|
||||
marked = True
|
||||
log.info("ready first frames %s",
|
||||
",".join(f"{pairs[i][0]}={seqs[i]}" for i in range(len(pairs))))
|
||||
now = time.monotonic()
|
||||
if now - last_log >= 5.0:
|
||||
last_log = now
|
||||
log.info("seqs %s",
|
||||
" ".join(f"{pairs[i][0]}:{seqs[i]}" for i in range(len(pairs))))
|
||||
finally:
|
||||
for proc in procs:
|
||||
if proc.poll() is None:
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
for r in reads:
|
||||
try:
|
||||
os.close(r)
|
||||
except OSError:
|
||||
pass
|
||||
for w in writers:
|
||||
w.close()
|
||||
if os.path.exists(ready_path):
|
||||
os.unlink(ready_path)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
+175
@@ -0,0 +1,175 @@
|
||||
"""GStreamer capture: C920 MJPEG -> jpegdec -> I420.
|
||||
|
||||
Shared capture daemon hosts a small pool of gst-launch processes (not
|
||||
one per worker). No sidecar H.264: LiveKit VAAPI is the only encoder.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
|
||||
GST_LAUNCH = shutil.which("gst-launch-1.0") or "gst-launch-1.0"
|
||||
|
||||
# 5 cameras per gst-launch. group=1 does not raise fps (UVC still 15);
|
||||
# a small pool still cuts 20 worker pipelines.
|
||||
GST_GROUP_SIZE = 5
|
||||
|
||||
_QUEUE = (
|
||||
"queue",
|
||||
"max-size-buffers=1",
|
||||
"max-size-bytes=0",
|
||||
"max-size-time=0",
|
||||
"leaky=downstream",
|
||||
)
|
||||
|
||||
|
||||
def _caps(width: int, height: int, fps: int) -> str:
|
||||
fps = max(1, int(fps))
|
||||
return f"image/jpeg,width={int(width)},height={int(height)},framerate={fps}/1"
|
||||
|
||||
|
||||
JPEG_MCU_FLICKER_CAMS = frozenset()
|
||||
|
||||
|
||||
def jpeg_mcu_bottom_crop(height: int) -> int:
|
||||
"""Rows in the last 16-tall 4:2:0 JPEG MCU (0 if height is aligned).
|
||||
|
||||
C920 MJPEG 640x360 leaves 8 leftover rows. Some sensors fill that
|
||||
pad with garbage that flickers on the wall (cam-07/14/15).
|
||||
"""
|
||||
h = int(height)
|
||||
if h < 32:
|
||||
return 0
|
||||
rem = h % 16
|
||||
if rem == 0:
|
||||
return 0
|
||||
return rem + 16
|
||||
|
||||
|
||||
def needs_jpeg_mcu_fix(ident: str) -> bool:
|
||||
return ident in JPEG_MCU_FLICKER_CAMS
|
||||
|
||||
|
||||
def crop_scale_i420_drop_bottom(
|
||||
buf: bytearray | bytes, width: int, height: int, crop: int
|
||||
) -> bytearray:
|
||||
"""Drop the bottom `crop` I420 rows; fill with studio black (Y=16).
|
||||
|
||||
Unused on the live path: Y=16 is a raised bar once encode is 16-aligned.
|
||||
Clone is the live fix on JPEG_MCU_FLICKER_CAMS. Kept for tests.
|
||||
"""
|
||||
crop = int(crop)
|
||||
width, height = int(width), int(height)
|
||||
need = width * height * 3 // 2
|
||||
out = bytearray(buf[:need] if len(buf) >= need else buf)
|
||||
if crop <= 0 or height <= crop or width < 2 or height % 2 or crop % 2:
|
||||
return out
|
||||
y_stride = width
|
||||
y_size = width * height
|
||||
uv_w = width // 2
|
||||
uv_h = height // 2
|
||||
for r in range(height - crop, height):
|
||||
dst = r * y_stride
|
||||
out[dst:dst + y_stride] = bytes([16]) * y_stride
|
||||
uv_crop = crop // 2
|
||||
u_base = y_size
|
||||
v_base = y_size + uv_w * uv_h
|
||||
z = bytes([128]) * uv_w
|
||||
for r in range(uv_h - uv_crop, uv_h):
|
||||
out[u_base + r * uv_w:u_base + (r + 1) * uv_w] = z
|
||||
out[v_base + r * uv_w:v_base + (r + 1) * uv_w] = z
|
||||
return out
|
||||
|
||||
|
||||
def repeat_i420_bottom_from_last_good(
|
||||
buf: bytearray, width: int, height: int, crop: int
|
||||
) -> None:
|
||||
"""Overwrite the bottom `crop` I420 rows by cloning the last good line.
|
||||
|
||||
Used after videocrop+videobox (studio-black pad). Cloning beats a
|
||||
visible 8px bar and beats videoscale (which re-edged cam-07).
|
||||
"""
|
||||
crop = int(crop)
|
||||
if crop <= 0:
|
||||
return
|
||||
width, height = int(width), int(height)
|
||||
if width <= 0 or height <= crop or width % 2 or height % 2 or crop % 2:
|
||||
return
|
||||
y_stride = width
|
||||
y_size = width * height
|
||||
uv_w = width // 2
|
||||
uv_h = height // 2
|
||||
uv_crop = crop // 2
|
||||
last_y = height - crop - 1
|
||||
src_y = last_y * y_stride
|
||||
src_row = buf[src_y:src_y + y_stride]
|
||||
for r in range(height - crop, height):
|
||||
dst = r * y_stride
|
||||
buf[dst:dst + y_stride] = src_row
|
||||
last_uv = uv_h - uv_crop - 1
|
||||
u_base = y_size
|
||||
v_base = y_size + uv_w * uv_h
|
||||
src_u = buf[u_base + last_uv * uv_w:u_base + (last_uv + 1) * uv_w]
|
||||
src_v = buf[v_base + last_uv * uv_w:v_base + (last_uv + 1) * uv_w]
|
||||
for r in range(uv_h - uv_crop, uv_h):
|
||||
buf[u_base + r * uv_w:u_base + (r + 1) * uv_w] = src_u
|
||||
buf[v_base + r * uv_w:v_base + (r + 1) * uv_w] = src_v
|
||||
|
||||
|
||||
def _one_cam(device: str, width: int, height: int, fps: int, fd: int) -> list[str]:
|
||||
width, height = int(width), int(height)
|
||||
tail: list[str] = [
|
||||
"videoconvert", "qos=false", "n-threads=2",
|
||||
"!", "video/x-raw,format=I420",
|
||||
"!", *_QUEUE,
|
||||
"!", "fdsink", f"fd={int(fd)}", "sync=false",
|
||||
]
|
||||
return [
|
||||
"v4l2src", f"device={device}", "do-timestamp=false",
|
||||
"!", _caps(width, height, fps),
|
||||
"!", "jpegparse",
|
||||
"!", "jpegdec", "qos=false",
|
||||
"!", *tail,
|
||||
]
|
||||
|
||||
|
||||
def video_capture_cmd(
|
||||
device: str,
|
||||
width: int,
|
||||
height: int,
|
||||
fps: int,
|
||||
h264_fifo: str | None = None, # ignored: no sidecar encode
|
||||
bitrate_kbps: int = 800,
|
||||
fd: int = 1,
|
||||
) -> list[str]:
|
||||
"""Single-camera pipeline: jpegdec -> I420, latest-frame queue, fdsink."""
|
||||
del h264_fifo, bitrate_kbps
|
||||
return [GST_LAUNCH, "-q", *_one_cam(device, width, height, fps, fd)]
|
||||
|
||||
|
||||
def shared_capture_cmd(
|
||||
cameras: list[tuple[str, str]],
|
||||
width: int,
|
||||
height: int,
|
||||
fps: int,
|
||||
fds: list[int],
|
||||
) -> list[str]:
|
||||
"""One gst-launch with N jpegdec->I420->fdsink graphs (no h264enc)."""
|
||||
if len(cameras) != len(fds):
|
||||
raise ValueError("cameras and fds length mismatch")
|
||||
cmd: list[str] = [GST_LAUNCH, "-q"]
|
||||
for (_ident, device), fd in zip(cameras, fds):
|
||||
cmd += _one_cam(device, width, height, fps, fd)
|
||||
return cmd
|
||||
|
||||
|
||||
def grouped_cameras(
|
||||
cameras: list[tuple[str, str]],
|
||||
fds: list[int],
|
||||
group_size: int = GST_GROUP_SIZE,
|
||||
) -> list[tuple[list[tuple[str, str]], list[int]]]:
|
||||
if group_size < 1:
|
||||
raise ValueError("group_size must be >= 1")
|
||||
out: list[tuple[list[tuple[str, str]], list[int]]] = []
|
||||
for i in range(0, len(cameras), group_size):
|
||||
out.append((cameras[i:i + group_size], fds[i:i + group_size]))
|
||||
return out
|
||||
@@ -26,11 +26,12 @@ class Config:
|
||||
room: str = "cameras"
|
||||
participant_prefix: str = "cam"
|
||||
# video
|
||||
width: int = 640
|
||||
height: int = 360
|
||||
fps: int = 30
|
||||
width: int = 320
|
||||
height: int = 180
|
||||
fps: int = 15
|
||||
min_fps: int = 5
|
||||
video_bitrate: int = 800_000
|
||||
video_bitrate: int = 400_000
|
||||
rally_video_bitrate: int = 20_000_000
|
||||
video_codec: str = "h264"
|
||||
video_encoder: str = "vaapi" # auto|software|hardware|nvenc|vaapi
|
||||
vaapi_device: str = "/dev/dri/renderD129"
|
||||
@@ -47,6 +48,29 @@ class Config:
|
||||
beamform_mode: str = "auto" # auto | force | off
|
||||
torch_num_threads: int = 1
|
||||
beamform_tdoa_every: int = 5
|
||||
# latency knobs (unset / False = legacy behavior)
|
||||
audio_queue_ms: Optional[int] = None # AUDIO_QUEUE_MS; None = max(150, hop_ms+50)
|
||||
video_hold_s: Optional[float] = None # VIDEO_HOLD_S; None = match hop
|
||||
worker_split_executor: bool = False
|
||||
enhance_speaker_only: bool = False
|
||||
audio_gate: bool = False
|
||||
audio_gate_open_db: float = -28.0
|
||||
audio_gate_close_db: float = -34.0
|
||||
audio_gate_hold_s: float = 0.3
|
||||
enhance_daemon: bool = True
|
||||
enhance_socket: str = "/tmp/livekit-enhance.sock"
|
||||
capture_daemon: bool = True
|
||||
capture_shm_dir: str = "/run/livekit-cameras/raw"
|
||||
audio_daemon: bool = True
|
||||
audio_shm_dir: str = "/run/livekit-cameras/pcm"
|
||||
encode_daemon: bool = True
|
||||
encode_shm_dir: str = "/run/livekit-cameras/h264"
|
||||
single_publisher: bool = True
|
||||
display_video_capacity: int = 0 # 0 = unbounded VideoStream
|
||||
display_video_format: str = "" # "" = SDK default, "bgra" | "i420"
|
||||
display_gst_queue_buffers: int = 2
|
||||
display_speaker_push: str = "timer" # timer | arrival
|
||||
display_pump_workers: int = 4
|
||||
# misc
|
||||
publish_timeout_s: float = 60.0
|
||||
log_level: str = "INFO"
|
||||
@@ -59,7 +83,7 @@ class Config:
|
||||
display_identity: str = "display-wall"
|
||||
display_roles: list[str] = field(
|
||||
default_factory=lambda: ["grid", "speaker", "screenshare"])
|
||||
display_speaker_camera: str = "cam-01" # cam-NN or "active"
|
||||
display_speaker_camera: str = "rally" # LiveKit identity; not a C920
|
||||
display_speaker_participant: str = "speaker"
|
||||
display_grid_cols: int = 5
|
||||
display_grid_rows: int = 4
|
||||
@@ -67,6 +91,28 @@ class Config:
|
||||
display_grid_height: int = 1080
|
||||
display_grid_fps: int = 30
|
||||
livekit_public_url: str = ""
|
||||
# Site tag published on local cameras (LiveKit participant attributes).
|
||||
participant_tags: list[str] = field(default_factory=lambda: ["uwh"])
|
||||
# Display wall: skip participants carrying these tags. Empty = off
|
||||
# (show local cameras). Set to ["uwh"] to show remotes instead.
|
||||
display_hide_tags: list[str] = field(default_factory=list)
|
||||
# Yellow tile chrome when MoveNet sees a raised hand (display process only).
|
||||
display_hand_raise: bool = True
|
||||
display_hand_raise_hz: float = 4.0
|
||||
display_hand_raise_hold_s: float = 1.0
|
||||
display_hand_raise_model: str = "thunder"
|
||||
display_speak_min_db: float = -40.0
|
||||
display_speak_margin_db: float = 6.0
|
||||
display_speak_hold_s: float = 0.6
|
||||
display_speak_rise_db: float = 3.0
|
||||
display_speak_corr: float = 0.4
|
||||
display_speak_confirm: int = 2
|
||||
# Person-aware background blur (C920s only; OpenVINO CPU). Rollback: PORTRAIT_BLUR=0.
|
||||
portrait_blur: bool = True
|
||||
portrait_shm_dir: str = "/run/livekit-cameras/portrait"
|
||||
portrait_hz: float = 8.0
|
||||
portrait_blur_px: int = 7
|
||||
portrait_hold_s: float = 0.8
|
||||
|
||||
|
||||
def _parse_cameras(spec: str) -> list[str]:
|
||||
@@ -95,6 +141,31 @@ def _parse_list(spec: str) -> list[str]:
|
||||
return [p.strip() for p in spec.split(",") if p.strip()]
|
||||
|
||||
|
||||
def _parse_bool(spec: str, default: bool = False) -> bool:
|
||||
raw = (spec or "").strip().lower()
|
||||
if not raw:
|
||||
return default
|
||||
if raw in ("1", "true", "yes", "on"):
|
||||
return True
|
||||
if raw in ("0", "false", "no", "off"):
|
||||
return False
|
||||
raise ValueError(f"invalid bool {spec!r}")
|
||||
|
||||
|
||||
def _opt_int(spec: str) -> Optional[int]:
|
||||
raw = (spec or "").strip()
|
||||
if raw == "":
|
||||
return None
|
||||
return int(raw)
|
||||
|
||||
|
||||
def _opt_float(spec: str) -> Optional[float]:
|
||||
raw = (spec or "").strip()
|
||||
if raw == "":
|
||||
return None
|
||||
return float(raw)
|
||||
|
||||
|
||||
def load_config(env_path: Optional[Path] = None) -> Config:
|
||||
_load_dotenv(Path(env_path or Path(__file__).parent / ".env"))
|
||||
|
||||
@@ -102,13 +173,14 @@ def load_config(env_path: Optional[Path] = None) -> Config:
|
||||
url=os.environ.get("LIVEKIT_URL", ""),
|
||||
api_key=os.environ.get("LIVEKIT_API_KEY", ""),
|
||||
api_secret=os.environ.get("LIVEKIT_API_SECRET", ""),
|
||||
room=os.environ.get("LIVEKIT_ROOM", "cameras"),
|
||||
room=os.environ.get("LIVEKIT_ROOM", "uwh-telhai"),
|
||||
participant_prefix=os.environ.get("PARTICIPANT_PREFIX", "cam"),
|
||||
width=int(os.environ.get("VIDEO_WIDTH", "640")),
|
||||
height=int(os.environ.get("VIDEO_HEIGHT", "360")),
|
||||
fps=int(os.environ.get("VIDEO_FPS", "30")),
|
||||
width=int(os.environ.get("VIDEO_WIDTH", "320")),
|
||||
height=int(os.environ.get("VIDEO_HEIGHT", "180")),
|
||||
fps=int(os.environ.get("VIDEO_FPS", "15")),
|
||||
min_fps=int(os.environ.get("VIDEO_MIN_FPS", "5")),
|
||||
video_bitrate=int(os.environ.get("VIDEO_BITRATE", "800000")),
|
||||
video_bitrate=int(os.environ.get("VIDEO_BITRATE", "400000")),
|
||||
rally_video_bitrate=int(os.environ.get("VIDEO_RALLY_BITRATE", "20000000")),
|
||||
video_codec=os.environ.get("VIDEO_CODEC", "h264").lower(),
|
||||
video_encoder=os.environ.get("VIDEO_ENCODER", "vaapi").lower(),
|
||||
vaapi_device=os.environ.get("LIBVA_DRM_DEVICE", "/dev/dri/renderD129"),
|
||||
@@ -124,6 +196,32 @@ def load_config(env_path: Optional[Path] = None) -> Config:
|
||||
beamform_mode=os.environ.get("BEAMFORM_MODE", "auto").lower(),
|
||||
torch_num_threads=int(os.environ.get("TORCH_NUM_THREADS", "1")),
|
||||
beamform_tdoa_every=int(os.environ.get("BEAMFORM_TDOA_EVERY", "5")),
|
||||
audio_queue_ms=_opt_int(os.environ.get("AUDIO_QUEUE_MS", "")),
|
||||
video_hold_s=_opt_float(os.environ.get("VIDEO_HOLD_S", "")),
|
||||
worker_split_executor=_parse_bool(os.environ.get("WORKER_SPLIT_EXECUTOR", "")),
|
||||
enhance_speaker_only=_parse_bool(os.environ.get("ENHANCE_SPEAKER_ONLY", "")),
|
||||
audio_gate=_parse_bool(os.environ.get("AUDIO_GATE", ""), default=False),
|
||||
audio_gate_open_db=float(os.environ.get("AUDIO_GATE_OPEN_DB", "-28")),
|
||||
audio_gate_close_db=float(os.environ.get("AUDIO_GATE_CLOSE_DB", "-34")),
|
||||
audio_gate_hold_s=float(os.environ.get("AUDIO_GATE_HOLD_S", "0.3")),
|
||||
enhance_daemon=_parse_bool(os.environ.get("ENHANCE_DAEMON", ""), default=True),
|
||||
enhance_socket=os.environ.get("ENHANCE_SOCKET", "/tmp/livekit-enhance.sock").strip()
|
||||
or "/tmp/livekit-enhance.sock",
|
||||
capture_daemon=_parse_bool(os.environ.get("CAPTURE_DAEMON", ""), default=True),
|
||||
capture_shm_dir=os.environ.get("CAPTURE_SHM_DIR", "/run/livekit-cameras/raw").strip()
|
||||
or "/run/livekit-cameras/raw",
|
||||
audio_daemon=_parse_bool(os.environ.get("AUDIO_DAEMON", ""), default=True),
|
||||
audio_shm_dir=os.environ.get("AUDIO_SHM_DIR", "/run/livekit-cameras/pcm").strip()
|
||||
or "/run/livekit-cameras/pcm",
|
||||
encode_daemon=_parse_bool(os.environ.get("ENCODE_DAEMON", ""), default=True),
|
||||
encode_shm_dir=os.environ.get("ENCODE_SHM_DIR", "/run/livekit-cameras/h264").strip()
|
||||
or "/run/livekit-cameras/h264",
|
||||
single_publisher=_parse_bool(os.environ.get("SINGLE_PUBLISHER", ""), default=True),
|
||||
display_video_capacity=int(os.environ.get("DISPLAY_VIDEO_CAPACITY", "0")),
|
||||
display_video_format=os.environ.get("DISPLAY_VIDEO_FORMAT", "").strip().lower(),
|
||||
display_gst_queue_buffers=int(os.environ.get("DISPLAY_GST_QUEUE_BUFFERS", "2")),
|
||||
display_speaker_push=os.environ.get("DISPLAY_SPEAKER_PUSH", "timer").strip().lower(),
|
||||
display_pump_workers=int(os.environ.get("DISPLAY_PUMP_WORKERS", "4")),
|
||||
publish_timeout_s=float(os.environ.get("PUBLISH_TIMEOUT_S", "60")),
|
||||
log_level=os.environ.get("LOG_LEVEL", "INFO").upper(),
|
||||
cameras=_parse_cameras(os.environ.get("CAMERAS", "all")),
|
||||
@@ -135,7 +233,8 @@ def load_config(env_path: Optional[Path] = None) -> Config:
|
||||
display_roles=_parse_list(
|
||||
os.environ.get("DISPLAY_ROLES", "grid,speaker,screenshare"))
|
||||
or ["grid", "speaker", "screenshare"],
|
||||
display_speaker_camera=os.environ.get("DISPLAY_SPEAKER_CAMERA", "cam-01"),
|
||||
display_speaker_camera=os.environ.get("DISPLAY_SPEAKER_CAMERA", "rally").strip()
|
||||
or "rally",
|
||||
display_speaker_participant=os.environ.get(
|
||||
"DISPLAY_SPEAKER_PARTICIPANT", "speaker"),
|
||||
display_grid_cols=int(os.environ.get("DISPLAY_GRID_COLS", "5")),
|
||||
@@ -144,6 +243,25 @@ def load_config(env_path: Optional[Path] = None) -> Config:
|
||||
display_grid_height=int(os.environ.get("DISPLAY_GRID_HEIGHT", "1080")),
|
||||
display_grid_fps=int(os.environ.get("DISPLAY_GRID_FPS", "30")),
|
||||
livekit_public_url=os.environ.get("LIVEKIT_PUBLIC_URL", ""),
|
||||
participant_tags=_parse_list(os.environ.get("PARTICIPANT_TAGS", "uwh")),
|
||||
display_hide_tags=_parse_list(os.environ.get("DISPLAY_HIDE_TAGS", "")),
|
||||
display_hand_raise=_parse_bool(os.environ.get("DISPLAY_HAND_RAISE", ""), default=True),
|
||||
display_hand_raise_hz=float(os.environ.get("DISPLAY_HAND_RAISE_HZ", "4")),
|
||||
display_hand_raise_hold_s=float(os.environ.get("DISPLAY_HAND_RAISE_HOLD_S", "1.0")),
|
||||
display_hand_raise_model=os.environ.get("DISPLAY_HAND_RAISE_MODEL", "thunder").strip().lower(),
|
||||
display_speak_min_db=float(os.environ.get("DISPLAY_SPEAK_MIN_DB", "-40")),
|
||||
display_speak_margin_db=float(os.environ.get("DISPLAY_SPEAK_MARGIN_DB", "6")),
|
||||
display_speak_hold_s=float(os.environ.get("DISPLAY_SPEAK_HOLD_S", "0.6")),
|
||||
display_speak_rise_db=float(os.environ.get("DISPLAY_SPEAK_RISE_DB", "3")),
|
||||
display_speak_corr=float(os.environ.get("DISPLAY_SPEAK_CORR", "0.4")),
|
||||
display_speak_confirm=int(os.environ.get("DISPLAY_SPEAK_CONFIRM", "2")),
|
||||
portrait_blur=_parse_bool(os.environ.get("PORTRAIT_BLUR", ""), default=True),
|
||||
portrait_shm_dir=os.environ.get(
|
||||
"PORTRAIT_SHM_DIR", "/run/livekit-cameras/portrait"
|
||||
).strip() or "/run/livekit-cameras/portrait",
|
||||
portrait_hz=float(os.environ.get("PORTRAIT_HZ", "8")),
|
||||
portrait_blur_px=int(os.environ.get("PORTRAIT_BLUR_PX", "7")),
|
||||
portrait_hold_s=float(os.environ.get("PORTRAIT_HOLD_S", "0.8")),
|
||||
)
|
||||
|
||||
if cfg.enhance_mode not in ("auto", "force", "off"):
|
||||
@@ -158,4 +276,45 @@ def load_config(env_path: Optional[Path] = None) -> Config:
|
||||
raise ValueError("TORCH_NUM_THREADS must be >= 1")
|
||||
if cfg.beamform_tdoa_every < 1:
|
||||
raise ValueError("BEAMFORM_TDOA_EVERY must be >= 1")
|
||||
if cfg.audio_queue_ms is not None and cfg.audio_queue_ms < 1:
|
||||
raise ValueError("AUDIO_QUEUE_MS must be >= 1")
|
||||
if cfg.audio_gate_hold_s < 0:
|
||||
raise ValueError("AUDIO_GATE_HOLD_S must be >= 0")
|
||||
if cfg.video_hold_s is not None and cfg.video_hold_s < 0:
|
||||
raise ValueError("VIDEO_HOLD_S must be >= 0")
|
||||
if cfg.display_video_capacity < 0:
|
||||
raise ValueError("DISPLAY_VIDEO_CAPACITY must be >= 0")
|
||||
if cfg.display_video_format not in ("", "bgra", "i420", "yuv420p"):
|
||||
raise ValueError(
|
||||
f"DISPLAY_VIDEO_FORMAT must be empty, bgra or i420, got {cfg.display_video_format}")
|
||||
if cfg.display_gst_queue_buffers < 1:
|
||||
raise ValueError("DISPLAY_GST_QUEUE_BUFFERS must be >= 1")
|
||||
if cfg.display_speaker_push not in ("timer", "arrival"):
|
||||
raise ValueError(
|
||||
f"DISPLAY_SPEAKER_PUSH must be timer|arrival, got {cfg.display_speaker_push}")
|
||||
if cfg.display_pump_workers < 1:
|
||||
raise ValueError("DISPLAY_PUMP_WORKERS must be >= 1")
|
||||
if cfg.display_hand_raise_hz <= 0:
|
||||
raise ValueError("DISPLAY_HAND_RAISE_HZ must be > 0")
|
||||
if cfg.display_hand_raise_hold_s < 0:
|
||||
raise ValueError("DISPLAY_HAND_RAISE_HOLD_S must be >= 0")
|
||||
if cfg.display_hand_raise_model not in ("lightning", "thunder"):
|
||||
raise ValueError(
|
||||
f"DISPLAY_HAND_RAISE_MODEL must be lightning|thunder, got {cfg.display_hand_raise_model}")
|
||||
if cfg.display_speak_margin_db < 0:
|
||||
raise ValueError("DISPLAY_SPEAK_MARGIN_DB must be >= 0")
|
||||
if cfg.display_speak_hold_s < 0:
|
||||
raise ValueError("DISPLAY_SPEAK_HOLD_S must be >= 0")
|
||||
if cfg.display_speak_rise_db < 0:
|
||||
raise ValueError("DISPLAY_SPEAK_RISE_DB must be >= 0")
|
||||
if not 0.0 <= cfg.display_speak_corr <= 1.0:
|
||||
raise ValueError("DISPLAY_SPEAK_CORR must be in 0..1")
|
||||
if cfg.display_speak_confirm < 1:
|
||||
raise ValueError("DISPLAY_SPEAK_CONFIRM must be >= 1")
|
||||
if cfg.portrait_hz <= 0:
|
||||
raise ValueError("PORTRAIT_HZ must be > 0")
|
||||
if cfg.portrait_blur_px < 1:
|
||||
raise ValueError("PORTRAIT_BLUR_PX must be >= 1")
|
||||
if cfg.portrait_hold_s < 0:
|
||||
raise ValueError("PORTRAIT_HOLD_S must be >= 0")
|
||||
return cfg
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
# Logitech Rally / Rally Bar as the speaker camera (not a cam-NN C920 slot).
|
||||
# Product ids: Rally Camera 0881, Rally Bar Huddle 087c, Rally Bar 089b,
|
||||
# Rally Bar Mini 08d3. Install when the device is on site:
|
||||
# cp deploy/99-rally-pin.rules /etc/udev/rules.d/
|
||||
# udevadm control --reload-rules && udevadm trigger
|
||||
#
|
||||
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="0881", ATTR{index}=="0", SYMLINK+="rally"
|
||||
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="087c", ATTR{index}=="0", SYMLINK+="rally"
|
||||
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="089b", ATTR{index}=="0", SYMLINK+="rally"
|
||||
SUBSYSTEM=="video4linux", ATTRS{idVendor}=="046d", ATTRS{idProduct}=="08d3", ATTR{index}=="0", SYMLINK+="rally"
|
||||
@@ -1,11 +1,9 @@
|
||||
[Unit]
|
||||
Description=Publish C920 cameras (speechbrain audio) to LiveKit
|
||||
Documentation=file:///home/fdenkena/livekit-cameras/README.md
|
||||
After=network-online.target sound.target livekit-server.service
|
||||
After=network-online.target sound.target
|
||||
Wants=network-online.target
|
||||
# Wants, not Requires: local livekit-server is a test stand-in.
|
||||
# Point LIVEKIT_URL at a remote server later and you can disable this unit.
|
||||
Wants=livekit-server.service
|
||||
# Local livekit-server is disabled; publishers use LIVEKIT_URL in .env.
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
[Unit]
|
||||
Description=Drive 3 DRM displays with LiveKit camera output (GStreamer kmssink)
|
||||
Documentation=file:///home/fdenkena/livekit-cameras/README.md
|
||||
After=network-online.target livekit-server.service livekit-cameras.service
|
||||
Wants=network-online.target livekit-cameras.service
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
# Do not After=/Wants= livekit-cameras: the wall must stay up when cameras
|
||||
# are missing and when the SFU is down (on-screen error, then reconnect).
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
@@ -10,7 +12,7 @@ WorkingDirectory=/home/fdenkena/livekit-cameras
|
||||
EnvironmentFile=-/home/fdenkena/livekit-cameras/.env
|
||||
Environment=PYTHONUNBUFFERED=1
|
||||
# Exclusive KMS access — no X/Wayland on these connectors.
|
||||
ExecStartPre=/home/fdenkena/livekit-cameras/deploy/wait-for-livekit.sh
|
||||
# Start kmssink immediately; run_displays.py shows LiveKit errors on the wall.
|
||||
ExecStart=/home/fdenkena/livekit-cameras/.venv/bin/python /home/fdenkena/livekit-cameras/run_displays.py
|
||||
ExecStop=/bin/kill -TERM $MAINPID
|
||||
Restart=always
|
||||
|
||||
@@ -2,6 +2,9 @@
|
||||
# Wait until LIVEKIT_URL (from .env) accepts TCP. Used so publishers do not
|
||||
# race the local test livekit-server on boot. When LIVEKIT_URL points at a
|
||||
# remote host this waits on that host instead.
|
||||
#
|
||||
# bash /dev/tcp to an unreachable host can hang until the kernel SYN timeout
|
||||
# (longer than systemd TimeoutStartSec). Cap each probe at 1s.
|
||||
set -u
|
||||
ENV_FILE=/home/fdenkena/livekit-cameras/.env
|
||||
if [ -f "$ENV_FILE" ]; then
|
||||
@@ -29,7 +32,7 @@ if [ "$host" = "localhost" ]; then
|
||||
fi
|
||||
echo "wait-for-livekit: $host:$port"
|
||||
for _ in $(seq 1 30); do
|
||||
if (echo >/dev/tcp/"$host"/"$port") 2>/dev/null; then
|
||||
if timeout 1 bash -c "echo >/dev/tcp/${host}/${port}" 2>/dev/null; then
|
||||
echo "wait-for-livekit: ready"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
@@ -18,6 +18,13 @@ import subprocess
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from usb_ids import (
|
||||
is_fleet_video,
|
||||
is_rally_pid,
|
||||
looks_like_rally_name,
|
||||
usb_vid_pid_of_video_node,
|
||||
)
|
||||
|
||||
log = logging.getLogger("cameras.discovery")
|
||||
|
||||
_SLOTS_FILE = Path(__file__).parent / "slots.json"
|
||||
@@ -32,6 +39,9 @@ class Camera:
|
||||
usb_port: str = ""
|
||||
serial: str = ""
|
||||
extra_video_devices: list[str] = field(default_factory=list)
|
||||
# None = use fleet VIDEO_WIDTH / VIDEO_HEIGHT (C920 360p). Rally is 720p.
|
||||
width: int | None = None
|
||||
height: int | None = None
|
||||
|
||||
|
||||
def select_cameras(cameras: list[Camera], spec: list[str]) -> list[Camera]:
|
||||
@@ -192,8 +202,18 @@ def discover_cameras() -> list[Camera]:
|
||||
for name, nodes in groups.items():
|
||||
if not nodes:
|
||||
continue
|
||||
if looks_like_rally_name(name):
|
||||
log.info("skipping Rally/speaker device %r (not a cam-NN slot)", name)
|
||||
continue
|
||||
nodes.sort(key=lambda n: int(Path(n).name.replace("video", "")))
|
||||
primary = nodes[0]
|
||||
ids = usb_vid_pid_of_video_node(primary)
|
||||
if ids is not None and not is_fleet_video(ids[0], ids[1]):
|
||||
log.info("skipping non-C920 USB %04x:%04x at %s", ids[0], ids[1], primary)
|
||||
continue
|
||||
if ids is not None and is_rally_pid(ids[1]):
|
||||
log.info("skipping Rally USB %04x:%04x at %s", ids[0], ids[1], primary)
|
||||
continue
|
||||
serial = _serial_of_video_node(primary)
|
||||
port = _usb_port_of_video_node(primary)
|
||||
|
||||
|
||||
+866
-7
@@ -1,11 +1,302 @@
|
||||
"""20-camera mosaic + framed tiles with name placeholders for the KMS wall."""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import threading
|
||||
from typing import Iterable, Optional
|
||||
|
||||
import cv2
|
||||
import numpy as np
|
||||
|
||||
# Compact mosaic for n live tiles (cols, rows). Landscape-first, max 5x4.
|
||||
_GRID_SHAPES = {
|
||||
0: (1, 1),
|
||||
1: (1, 1),
|
||||
2: (2, 1),
|
||||
3: (2, 2),
|
||||
4: (2, 2),
|
||||
5: (3, 2),
|
||||
6: (3, 2),
|
||||
7: (4, 2),
|
||||
8: (4, 2),
|
||||
9: (3, 3),
|
||||
10: (4, 3),
|
||||
11: (4, 3),
|
||||
12: (4, 3),
|
||||
13: (5, 3),
|
||||
14: (5, 3),
|
||||
15: (5, 3),
|
||||
16: (4, 4),
|
||||
17: (5, 4),
|
||||
18: (5, 4),
|
||||
19: (5, 4),
|
||||
20: (5, 4),
|
||||
}
|
||||
|
||||
|
||||
def grid_shape(n: int, max_cols: int = 5, max_rows: int = 4) -> tuple[int, int]:
|
||||
"""Cols/rows for `n` live tiles. Empty room still occupies 1x1."""
|
||||
cap = max(1, int(max_cols) * int(max_rows))
|
||||
n = max(0, min(int(n), cap))
|
||||
if max_cols == 5 and max_rows == 4 and n in _GRID_SHAPES:
|
||||
return _GRID_SHAPES[n]
|
||||
if n <= 1:
|
||||
return 1, 1
|
||||
cols = min(int(max_cols), n)
|
||||
rows = min(int(max_rows), math.ceil(n / cols))
|
||||
while cols * rows < n and (cols < max_cols or rows < max_rows):
|
||||
if cols < max_cols:
|
||||
cols += 1
|
||||
else:
|
||||
rows += 1
|
||||
return cols, rows
|
||||
|
||||
|
||||
def describe_livekit_error(exc: BaseException) -> str:
|
||||
"""Short on-screen reason for a LiveKit connect/session failure."""
|
||||
if isinstance(exc, TimeoutError):
|
||||
return "connection timed out"
|
||||
if isinstance(exc, ConnectionRefusedError):
|
||||
return "connection refused"
|
||||
raw = str(exc) or type(exc).__name__
|
||||
low = raw.lower()
|
||||
if "113" in raw or "no route" in low or "unreachable" in low:
|
||||
return "host unreachable"
|
||||
if "111" in raw or "refused" in low:
|
||||
return "connection refused"
|
||||
if "timed out" in low or "timeout" in low:
|
||||
return "connection timed out"
|
||||
if "name or service not known" in low or "nodename nor servname" in low:
|
||||
return "DNS lookup failed"
|
||||
if "connection reset" in low or "broken pipe" in low:
|
||||
return "connection reset"
|
||||
if "websocket" in low or "ws://" in low or "wss://" in low:
|
||||
return raw[:96]
|
||||
return raw[:96]
|
||||
|
||||
|
||||
def format_livekit_status(
|
||||
*,
|
||||
url: str,
|
||||
state: str,
|
||||
detail: str = "",
|
||||
retry_s: float | None = None,
|
||||
room: str = "",
|
||||
) -> list[str]:
|
||||
"""Lines for the KMS status overlay. Empty list means show the mosaic."""
|
||||
url = (url or "").strip()
|
||||
st = (state or "").strip().lower()
|
||||
room_line = f"room {room.strip()}" if (room or "").strip() else ""
|
||||
if st in ("connected", "ok", ""):
|
||||
return []
|
||||
if st == "connecting":
|
||||
lines = ["CONNECTING TO LIVEKIT"]
|
||||
if url:
|
||||
lines.append(url)
|
||||
if room_line:
|
||||
lines.append(room_line)
|
||||
return lines
|
||||
lines = ["LIVEKIT DISCONNECTED"]
|
||||
if url:
|
||||
lines.append(url)
|
||||
if room_line:
|
||||
lines.append(room_line)
|
||||
if detail:
|
||||
lines.append(str(detail)[:96])
|
||||
if retry_s is not None:
|
||||
lines.append(f"retrying in {max(0, int(round(float(retry_s))))}s")
|
||||
return lines
|
||||
|
||||
|
||||
# Status-screen palettes are BGRA. Amber = offline, cyan = connecting.
|
||||
_STATUS_AMBER = (36, 158, 255, 255)
|
||||
_STATUS_CYAN = (255, 186, 64, 255)
|
||||
_STATUS_STEEL = (196, 148, 92, 255)
|
||||
_STATUS_INK = (18, 14, 12, 255)
|
||||
_STATUS_PAPER = (245, 242, 238, 255)
|
||||
_STATUS_MUTED = (168, 158, 150, 255)
|
||||
|
||||
|
||||
def _status_kind(title: str) -> str:
|
||||
t = (title or "").upper()
|
||||
if "CONNECTING" in t:
|
||||
return "connecting"
|
||||
if "NO LIVE" in t or "NO CAMERA" in t:
|
||||
return "empty"
|
||||
return "disconnected"
|
||||
|
||||
|
||||
def _round_rect(img: np.ndarray, x: int, y: int, w: int, h: int, r: int,
|
||||
color, thickness: int = -1) -> None:
|
||||
r = max(0, min(int(r), w // 2, h // 2))
|
||||
x, y, w, h = int(x), int(y), int(w), int(h)
|
||||
if r <= 0:
|
||||
cv2.rectangle(img, (x, y), (x + w, y + h), color, thickness)
|
||||
return
|
||||
if thickness < 0:
|
||||
cv2.rectangle(img, (x + r, y), (x + w - r, y + h), color, -1)
|
||||
cv2.rectangle(img, (x, y + r), (x + w, y + h - r), color, -1)
|
||||
cv2.circle(img, (x + r, y + r), r, color, -1)
|
||||
cv2.circle(img, (x + w - r, y + r), r, color, -1)
|
||||
cv2.circle(img, (x + r, y + h - r), r, color, -1)
|
||||
cv2.circle(img, (x + w - r, y + h - r), r, color, -1)
|
||||
return
|
||||
cv2.ellipse(img, (x + r, y + r), (r, r), 180, 0, 90, color, thickness, cv2.LINE_AA)
|
||||
cv2.ellipse(img, (x + w - r, y + r), (r, r), 270, 0, 90, color, thickness, cv2.LINE_AA)
|
||||
cv2.ellipse(img, (x + r, y + h - r), (r, r), 90, 0, 90, color, thickness, cv2.LINE_AA)
|
||||
cv2.ellipse(img, (x + w - r, y + h - r), (r, r), 0, 0, 90, color, thickness, cv2.LINE_AA)
|
||||
cv2.line(img, (x + r, y), (x + w - r, y), color, thickness, cv2.LINE_AA)
|
||||
cv2.line(img, (x + r, y + h), (x + w - r, y + h), color, thickness, cv2.LINE_AA)
|
||||
cv2.line(img, (x, y + r), (x, y + h - r), color, thickness, cv2.LINE_AA)
|
||||
cv2.line(img, (x + w, y + r), (x + w, y + h - r), color, thickness, cv2.LINE_AA)
|
||||
|
||||
|
||||
def _text_size(text: str, scale: float, thick: int) -> tuple[int, int]:
|
||||
(tw, th), _ = cv2.getTextSize(text, cv2.FONT_HERSHEY_SIMPLEX, scale, thick)
|
||||
return tw, th
|
||||
|
||||
|
||||
def _draw_text(img: np.ndarray, text: str, x: int, y: int, scale: float,
|
||||
color, thick: int = 1) -> None:
|
||||
cv2.putText(img, text, (int(x), int(y)), cv2.FONT_HERSHEY_SIMPLEX,
|
||||
scale, color, thick, cv2.LINE_AA)
|
||||
|
||||
|
||||
def _draw_signal_mark(img: np.ndarray, cx: int, cy: int, accent, *, ok: bool) -> None:
|
||||
"""Wifi-style arcs; a slash when the SFU is down."""
|
||||
for rad in (18, 30, 42):
|
||||
cv2.ellipse(img, (cx, cy + 10), (rad, rad), 0, 225, 315, accent, 2, cv2.LINE_AA)
|
||||
cv2.circle(img, (cx, cy + 18), 4, accent, -1, cv2.LINE_AA)
|
||||
if not ok:
|
||||
cv2.line(img, (cx - 28, cy - 22), (cx + 28, cy + 28), (40, 40, 220, 255), 3, cv2.LINE_AA)
|
||||
|
||||
|
||||
def status_screen(width: int, height: int, lines: list[str] | None) -> np.ndarray:
|
||||
"""Full-frame KMS status: gradient, ghost grid, accent rail, info card."""
|
||||
w, h = max(64, int(width)), max(64, int(height))
|
||||
raw = [str(x).strip() for x in (lines or []) if str(x).strip()]
|
||||
title = raw[0] if raw else "NO LIVE CAMERAS"
|
||||
url = next((ln for ln in raw[1:] if ln.startswith(("ws://", "wss://", "http://"))), "")
|
||||
retry = next((ln for ln in raw[1:] if "retry" in ln.lower()), "")
|
||||
room = next((ln for ln in raw[1:] if ln.lower().startswith("room ")), "")
|
||||
detail = next(
|
||||
(ln for ln in raw[1:] if ln not in (url, retry, room) and ln != title),
|
||||
"",
|
||||
)
|
||||
kind = _status_kind(title)
|
||||
accent = {"connecting": _STATUS_CYAN, "empty": _STATUS_STEEL}.get(kind, _STATUS_AMBER)
|
||||
badge = {"connecting": "CONNECTING", "empty": "STANDBY"}.get(kind, "OFFLINE")
|
||||
if kind == "connecting" and not retry:
|
||||
retry = "waiting for signal"
|
||||
|
||||
yy = np.linspace(0.0, 1.0, h, dtype=np.float32)[:, None]
|
||||
if kind == "connecting":
|
||||
top = np.array([28, 22, 16], dtype=np.float32)
|
||||
bot = np.array([42, 28, 14], dtype=np.float32)
|
||||
elif kind == "empty":
|
||||
top = np.array([24, 20, 18], dtype=np.float32)
|
||||
bot = np.array([32, 26, 22], dtype=np.float32)
|
||||
else:
|
||||
top = np.array([22, 16, 12], dtype=np.float32)
|
||||
bot = np.array([16, 14, 20], dtype=np.float32)
|
||||
img = np.empty((h, w, 4), dtype=np.uint8)
|
||||
rgb = top * (1.0 - yy) + bot * yy
|
||||
img[:, :, :3] = rgb.astype(np.uint8)[:, None, :]
|
||||
img[:, :, 3] = 255
|
||||
|
||||
# Ghost 5x4 mosaic — the wall that should be here.
|
||||
cols, rows, gap = 5, 4, max(4, w // 240)
|
||||
gx0, gy0 = int(w * 0.08), int(h * 0.12)
|
||||
gx1, gy1 = int(w * 0.92), int(h * 0.88)
|
||||
tw = max(8, (gx1 - gx0 - gap * (cols + 1)) // cols)
|
||||
th = max(8, (gy1 - gy0 - gap * (rows + 1)) // rows)
|
||||
ghost = (58, 48, 42, 255) if kind != "connecting" else (48, 50, 42, 255)
|
||||
for r in range(rows):
|
||||
for c in range(cols):
|
||||
x = gx0 + gap + c * (tw + gap)
|
||||
y = gy0 + gap + r * (th + gap)
|
||||
cv2.rectangle(img, (x, y), (x + tw, y + th), ghost, 1, cv2.LINE_AA)
|
||||
|
||||
rail = max(6, w // 240)
|
||||
img[:, :rail] = accent
|
||||
|
||||
s = max(0.55, min(w / 1920.0, h / 1080.0))
|
||||
pad = int(48 * s)
|
||||
foot_h = int(58 * s) if retry else 0
|
||||
body = int(188 * s)
|
||||
if url:
|
||||
body += int(72 * s)
|
||||
if room:
|
||||
body += int(72 * s)
|
||||
if detail:
|
||||
body += int(80 * s)
|
||||
body += foot_h + int(28 * s)
|
||||
cw = min(int(w * 0.62), 1080)
|
||||
ch = int(min(max(body, int(h * 0.34)), int(h * 0.56)))
|
||||
cx0 = (w - cw) // 2
|
||||
cy0 = (h - ch) // 2
|
||||
card = (32, 26, 24, 255) if kind != "connecting" else (36, 30, 22, 255)
|
||||
_round_rect(img, cx0, cy0, cw, ch, 18, card, -1)
|
||||
_round_rect(img, cx0, cy0, cw, ch, 18, accent, 2)
|
||||
img[cy0:cy0 + 6, cx0 + 18:cx0 + cw - 18] = accent
|
||||
|
||||
x = cx0 + pad
|
||||
y = cy0 + int(56 * s)
|
||||
pill_w = int(188 * s)
|
||||
_round_rect(img, x, y - int(28 * s), pill_w, int(36 * s), 8, accent, -1)
|
||||
bw, bh = _text_size(badge, 0.52 * s, 1)
|
||||
_draw_text(img, badge, x + (pill_w - bw) // 2, y - int(28 * s) + bh + int(8 * s),
|
||||
0.52 * s, _STATUS_INK, 1)
|
||||
|
||||
_draw_signal_mark(img, cx0 + cw - pad - 20, y - 4, accent, ok=(kind == "connecting"))
|
||||
|
||||
y += int(70 * s)
|
||||
tscale = 1.35 * s
|
||||
tw, th = _text_size(title, tscale, 2)
|
||||
while tw > cw - 2 * pad and tscale > 0.6:
|
||||
tscale *= 0.92
|
||||
tw, th = _text_size(title, tscale, 2)
|
||||
_draw_text(img, title, x, y + th, tscale, _STATUS_PAPER, 2)
|
||||
y += th + int(22 * s)
|
||||
cv2.line(img, (x, y), (cx0 + cw - pad, y), (64, 54, 48, 255), 1, cv2.LINE_AA)
|
||||
y += int(36 * s)
|
||||
|
||||
if url:
|
||||
_draw_text(img, "ENDPOINT", x, y, 0.48 * s, _STATUS_MUTED, 1)
|
||||
y += int(28 * s)
|
||||
uscale = 0.78 * s
|
||||
uw, uh = _text_size(url, uscale, 1)
|
||||
while uw > cw - 2 * pad and uscale > 0.4:
|
||||
uscale *= 0.92
|
||||
uw, uh = _text_size(url, uscale, 1)
|
||||
_draw_text(img, url, x, y + uh, uscale, _STATUS_PAPER, 1)
|
||||
y += uh + int(28 * s)
|
||||
if room:
|
||||
_draw_text(img, "ROOM", x, y, 0.48 * s, _STATUS_MUTED, 1)
|
||||
y += int(28 * s)
|
||||
rscale = 0.9 * s
|
||||
rw, rh = _text_size(room, rscale, 2)
|
||||
while rw > cw - 2 * pad and rscale > 0.45:
|
||||
rscale *= 0.92
|
||||
rw, rh = _text_size(room, rscale, 2)
|
||||
_draw_text(img, room, x, y + rh, rscale, _STATUS_PAPER, 2)
|
||||
y += rh + int(28 * s)
|
||||
if detail:
|
||||
_draw_text(img, "REASON", x, y, 0.48 * s, _STATUS_MUTED, 1)
|
||||
y += int(28 * s)
|
||||
_draw_text(img, detail[:72], x, y + int(22 * s), 0.78 * s, accent, 1)
|
||||
|
||||
if retry:
|
||||
fy = cy0 + ch - foot_h
|
||||
band = (26, 20, 16, 255) if kind != "connecting" else (28, 24, 16, 255)
|
||||
img[fy:cy0 + ch - 8, cx0 + 10:cx0 + cw - 10] = band
|
||||
_draw_text(img, retry.upper(), x, fy + int(38 * s), 0.62 * s, accent, 1)
|
||||
|
||||
corner = room.upper() if room else "CAMERA WALL"
|
||||
_draw_text(img, corner, int(w * 0.08), int(h * 0.07), 0.48 * s, (140, 132, 126, 255), 1)
|
||||
return img
|
||||
|
||||
|
||||
ROLES = ("grid", "speaker", "screenshare")
|
||||
|
||||
# BGRA
|
||||
@@ -14,6 +305,8 @@ CELL_BG = (18, 18, 18, 255)
|
||||
EMPTY_FILL = (36, 36, 36, 255)
|
||||
FRAME_COLOR = (196, 196, 196, 255)
|
||||
FRAME_LIVE = (168, 168, 168, 255)
|
||||
FRAME_RAISE = (0, 220, 255, 255) # BGRA yellow — hand raised
|
||||
FRAME_SPEAK = (0, 210, 40, 255) # BGRA green — LiveKit active speaker
|
||||
LABEL_BG = (12, 12, 12, 255)
|
||||
LABEL_FG = (245, 245, 245, 255)
|
||||
NAME_FG = (220, 220, 220, 255)
|
||||
@@ -29,6 +322,193 @@ def is_camera_identity(identity: str, prefix: str = "cam") -> bool:
|
||||
return ident.startswith(p + "-") or ident.startswith(p)
|
||||
|
||||
|
||||
def active_speaker_ids(speakers, *, want=None) -> list[str]:
|
||||
"""LiveKit active_speakers_changed payload -> identity list."""
|
||||
out: list[str] = []
|
||||
for p in speakers:
|
||||
ident = (getattr(p, "identity", None) or "").strip()
|
||||
if not ident:
|
||||
continue
|
||||
if want is not None and not want(p):
|
||||
continue
|
||||
out.append(ident)
|
||||
return out
|
||||
|
||||
|
||||
def primary_grid_speaker(ids: Iterable[str], identities: Iterable[str]) -> list[str]:
|
||||
"""Loudest LiveKit speaker that occupies a grid tile (skip rally)."""
|
||||
tiles = set(identities)
|
||||
for ident in ids:
|
||||
if ident in tiles:
|
||||
return [ident]
|
||||
return []
|
||||
|
||||
|
||||
def select_grid_speakers(
|
||||
ranked: Iterable[tuple[str, float]],
|
||||
*,
|
||||
min_db: float = -40.0,
|
||||
margin_db: float = 6.0,
|
||||
floors: dict[str, float] | None = None,
|
||||
rise_db: float = 6.0,
|
||||
) -> list[str]:
|
||||
"""Keep real talkers; drop idle floor and quieter bleed of the same source.
|
||||
|
||||
Speech is ``rise_db`` above that mic's noise floor (not a global dBFS
|
||||
cut). Two adjacent people stay within ``margin_db`` of each other.
|
||||
"""
|
||||
rise = max(0.0, float(rise_db))
|
||||
live: list[tuple[str, float]] = []
|
||||
for ident, raw in ranked:
|
||||
if not ident:
|
||||
continue
|
||||
db = float(raw)
|
||||
if floors is not None:
|
||||
fl = floors.get(ident)
|
||||
if fl is None or db < float(fl) + rise:
|
||||
continue
|
||||
elif db < float(min_db):
|
||||
continue
|
||||
live.append((ident, db))
|
||||
if not live:
|
||||
return []
|
||||
peak = max(db for _, db in live)
|
||||
margin = max(0.0, float(margin_db))
|
||||
return [i for i, db in live if (peak - db) <= margin]
|
||||
|
||||
|
||||
def update_noise_floor(
|
||||
floor: float,
|
||||
db: float,
|
||||
*,
|
||||
speaking: bool = False,
|
||||
rise_db: float = 4.0,
|
||||
alpha: float = 0.2,
|
||||
) -> float:
|
||||
"""Follow typical idle; freeze while this hop looks like speech."""
|
||||
del speaking
|
||||
fl = float(floor)
|
||||
x = float(db)
|
||||
a = min(1.0, max(0.0, float(alpha)))
|
||||
rise = max(0.0, float(rise_db))
|
||||
if x >= fl + rise:
|
||||
return fl
|
||||
return (1.0 - a) * fl + a * x
|
||||
|
||||
|
||||
def confirm_speakers(
|
||||
counts: dict[str, int],
|
||||
picked: Iterable[str],
|
||||
*,
|
||||
need: int = 2,
|
||||
already: Iterable[str] | None = None,
|
||||
) -> tuple[list[str], dict[str, int]]:
|
||||
"""Drop one-hop idle spikes; already-on seats stay without re-arming."""
|
||||
picked_l = [i for i in picked if i]
|
||||
already_s = set(already or ())
|
||||
n_need = max(1, int(need))
|
||||
new_counts: dict[str, int] = {}
|
||||
out: list[str] = []
|
||||
for ident in picked_l:
|
||||
n = int(counts.get(ident, 0)) + 1
|
||||
new_counts[ident] = n
|
||||
if n >= n_need or ident in already_s:
|
||||
out.append(ident)
|
||||
return out, new_counts
|
||||
|
||||
|
||||
def waveform_similarity(
|
||||
a: np.ndarray,
|
||||
b: np.ndarray,
|
||||
max_lag: int = 320,
|
||||
) -> float:
|
||||
"""Peak |normalized xcorr| over ±max_lag samples (bleed vs two talkers)."""
|
||||
x = np.asarray(a, dtype=np.float64).reshape(-1)
|
||||
y = np.asarray(b, dtype=np.float64).reshape(-1)
|
||||
n = min(x.size, y.size)
|
||||
if n < 32:
|
||||
return 0.0
|
||||
x = x[:n] - x[:n].mean()
|
||||
y = y[:n] - y[:n].mean()
|
||||
nx = float(np.linalg.norm(x))
|
||||
ny = float(np.linalg.norm(y))
|
||||
if nx < 1e-9 or ny < 1e-9:
|
||||
return 0.0
|
||||
lag = max(0, min(int(max_lag), n - 1))
|
||||
corr = np.correlate(x, y, mode="full")
|
||||
mid = n - 1
|
||||
sl = corr[mid - lag: mid + lag + 1]
|
||||
return float(np.max(np.abs(sl)) / (nx * ny))
|
||||
|
||||
|
||||
def collapse_bleed(
|
||||
ids: Iterable[str],
|
||||
waves: dict[str, np.ndarray],
|
||||
levels: dict[str, float],
|
||||
*,
|
||||
corr_min: float = 0.4,
|
||||
max_lag: int = 320,
|
||||
) -> list[str]:
|
||||
"""If two hops are the same waveform, keep the louder mic only."""
|
||||
ordered = [i for i in ids if i]
|
||||
if len(ordered) < 2:
|
||||
return ordered
|
||||
keep = set(ordered)
|
||||
thresh = float(corr_min)
|
||||
for i, a in enumerate(ordered):
|
||||
if a not in keep:
|
||||
continue
|
||||
wa = waves.get(a)
|
||||
if wa is None:
|
||||
continue
|
||||
for b in ordered[i + 1:]:
|
||||
if b not in keep:
|
||||
continue
|
||||
wb = waves.get(b)
|
||||
if wb is None:
|
||||
continue
|
||||
if waveform_similarity(wa, wb, max_lag=max_lag) < thresh:
|
||||
continue
|
||||
if float(levels.get(a, -120.0)) >= float(levels.get(b, -120.0)):
|
||||
keep.discard(b)
|
||||
else:
|
||||
keep.discard(a)
|
||||
break
|
||||
return [i for i in ordered if i in keep]
|
||||
|
||||
|
||||
def hold_speakers(
|
||||
prev: list[str],
|
||||
new: list[str],
|
||||
*,
|
||||
prev_peak: float,
|
||||
new_peak: float,
|
||||
now: float,
|
||||
hold_until: float,
|
||||
hold_s: float = 0.6,
|
||||
switch_db: float = 3.0,
|
||||
) -> tuple[list[str], float]:
|
||||
"""Stick chrome across SFU jitter; never block a second real talker."""
|
||||
prev_l = list(prev)
|
||||
new_l = list(new)
|
||||
hold = max(0.0, float(hold_s))
|
||||
if new_l == prev_l:
|
||||
if new_l:
|
||||
return new_l, max(hold_until, now + hold)
|
||||
return [], hold_until
|
||||
if not prev_l:
|
||||
return new_l, now + hold
|
||||
if set(prev_l) <= set(new_l):
|
||||
return new_l, now + hold
|
||||
if not new_l:
|
||||
if now < hold_until:
|
||||
return prev_l, hold_until
|
||||
return [], hold_until
|
||||
if new_peak >= prev_peak + float(switch_db) or now >= hold_until:
|
||||
return new_l, now + hold
|
||||
return prev_l, hold_until
|
||||
|
||||
|
||||
def display_name(identity: str) -> str:
|
||||
"""Human label for a camera identity (cam-07 -> CAM-07)."""
|
||||
return (identity or "").strip().upper() or "CAMERA"
|
||||
@@ -56,6 +536,75 @@ def letterbox_bgra(src: np.ndarray, tw: int, th: int) -> np.ndarray:
|
||||
return out
|
||||
|
||||
|
||||
def i420_nbytes(width: int, height: int) -> int:
|
||||
return int(width) * int(height) * 3 // 2
|
||||
|
||||
|
||||
def _even(n: int) -> int:
|
||||
n = int(n)
|
||||
if n <= 0:
|
||||
return 0
|
||||
return n if n % 2 == 0 else n - 1
|
||||
|
||||
|
||||
def letterbox_i420(src: bytes | memoryview, sw: int, sh: int, tw: int, th: int) -> bytes:
|
||||
"""Scale packed I420 into tw x th, even UV, studio-black bars (Y=16).
|
||||
|
||||
Y=0 is below-black and sparkles on HDMI (kmssink limited range).
|
||||
"""
|
||||
tw, th = max(2, _even(tw)), max(2, _even(th))
|
||||
sw, sh = int(sw), int(sh)
|
||||
need = i420_nbytes(sw, sh)
|
||||
raw = bytes(src) if not isinstance(src, (bytes, bytearray)) else src
|
||||
if len(raw) < need or sw < 2 or sh < 2:
|
||||
raise ValueError("bad I420 source")
|
||||
arr = np.frombuffer(raw, dtype=np.uint8, count=need)
|
||||
ysz = sw * sh
|
||||
uvw, uvh = sw // 2, sh // 2
|
||||
uvsz = uvw * uvh
|
||||
y = arr[:ysz].reshape(sh, sw)
|
||||
u = arr[ysz:ysz + uvsz].reshape(uvh, uvw)
|
||||
v = arr[ysz + uvsz:ysz + 2 * uvsz].reshape(uvh, uvw)
|
||||
scale = min(tw / sw, th / sh)
|
||||
nw, nh = _even(round(sw * scale)), _even(round(sh * scale))
|
||||
y2 = cv2.resize(y, (nw, nh), interpolation=cv2.INTER_AREA)
|
||||
u2 = cv2.resize(u, (nw // 2, nh // 2), interpolation=cv2.INTER_AREA)
|
||||
v2 = cv2.resize(v, (nw // 2, nh // 2), interpolation=cv2.INTER_AREA)
|
||||
out = np.empty(i420_nbytes(tw, th), dtype=np.uint8)
|
||||
oy = tw * th
|
||||
ouv = (tw // 2) * (th // 2)
|
||||
out[:oy] = 16
|
||||
out[oy:] = 128
|
||||
yo = out[:oy].reshape(th, tw)
|
||||
uo = out[oy:oy + ouv].reshape(th // 2, tw // 2)
|
||||
vo = out[oy + ouv:].reshape(th // 2, tw // 2)
|
||||
x = _even((tw - nw) // 2)
|
||||
y0 = _even((th - nh) // 2)
|
||||
yo[y0:y0 + nh, x:x + nw] = y2
|
||||
uo[y0 // 2:y0 // 2 + nh // 2, x // 2:x // 2 + nw // 2] = u2
|
||||
vo[y0 // 2:y0 // 2 + nh // 2, x // 2:x // 2 + nw // 2] = v2
|
||||
return out.tobytes()
|
||||
|
||||
|
||||
def bgra_to_i420(bgra: np.ndarray) -> bytes:
|
||||
"""Packed I420 bytes (h*w*3//2) from HxWx4 BGRA."""
|
||||
if bgra.ndim != 3 or bgra.shape[2] < 3:
|
||||
raise ValueError(f"expected HxWx3/4, got {bgra.shape}")
|
||||
bgr = bgra[:, :, :3]
|
||||
yuv = cv2.cvtColor(bgr, cv2.COLOR_BGR2YUV_I420)
|
||||
return np.ascontiguousarray(yuv).tobytes()
|
||||
|
||||
|
||||
def bgra_to_yuv_pixel(bgra: tuple[int, int, int, int] | tuple[int, int, int]) -> tuple[int, int, int]:
|
||||
"""Sample Y,U,V of a solid BGRA color via OpenCV I420."""
|
||||
img = np.empty((2, 2, 4), dtype=np.uint8)
|
||||
b, g, r = int(bgra[0]), int(bgra[1]), int(bgra[2])
|
||||
a = int(bgra[3]) if len(bgra) > 3 else 255
|
||||
img[:] = (b, g, r, a)
|
||||
arr = np.frombuffer(bgra_to_i420(img), dtype=np.uint8)
|
||||
return int(arr[0]), int(arr[4]), int(arr[5])
|
||||
|
||||
|
||||
def bgra_from_livekit(data: memoryview | bytes, width: int, height: int) -> np.ndarray:
|
||||
arr = np.frombuffer(data, dtype=np.uint8)
|
||||
return arr.reshape((height, width, 4)).copy()
|
||||
@@ -123,33 +672,138 @@ class CameraGrid:
|
||||
gap: int = 8,
|
||||
border: int = 3,
|
||||
label_h: int = 28,
|
||||
chrome_out: int = 4,
|
||||
) -> None:
|
||||
if cols < 1 or rows < 1:
|
||||
raise ValueError("cols/rows must be >= 1")
|
||||
self.width = int(width)
|
||||
self.height = int(height)
|
||||
self.max_cols = int(cols)
|
||||
self.max_rows = int(rows)
|
||||
self.cols = int(cols)
|
||||
self.rows = int(rows)
|
||||
self.gap = max(0, int(gap))
|
||||
self.border = max(1, int(border))
|
||||
# Extra speak/raise thickness in the gutter only (even, for I420 UV).
|
||||
out = min(self.gap, max(0, int(chrome_out)))
|
||||
self.chrome_out = out if out % 2 == 0 else out - 1
|
||||
self.label_h = max(16, int(label_h))
|
||||
self.identities = list(identities) if identities is not None else camera_ids(cols * rows)
|
||||
self._status_lines: list[str] | None = None
|
||||
self._live: set[str] = set()
|
||||
self._raised: set[str] = set()
|
||||
self._speaking: set[str] = set()
|
||||
self._tiles: dict[str, np.ndarray] = {}
|
||||
self._lock = threading.Lock()
|
||||
self._yuv_raise = bgra_to_yuv_pixel(FRAME_RAISE)
|
||||
self._yuv_speak = bgra_to_yuv_pixel(FRAME_SPEAK)
|
||||
self._canvas = np.zeros((self.height, self.width, 4), dtype=np.uint8)
|
||||
self._apply_layout(self.identities, cols=self.cols, rows=self.rows)
|
||||
|
||||
def _layout_metrics(self, cols: int, rows: int) -> None:
|
||||
cols = max(1, int(cols))
|
||||
rows = max(1, int(rows))
|
||||
self.cols = cols
|
||||
self.rows = rows
|
||||
self.tile_w = max(32, (self.width - self.gap * (self.cols + 1)) // self.cols)
|
||||
self.tile_h = max(32, (self.height - self.gap * (self.rows + 1)) // self.rows)
|
||||
self.inner_w = max(1, self.tile_w - 2 * self.border)
|
||||
self.inner_h = max(1, self.tile_h - 2 * self.border - self.label_h)
|
||||
self._live: set[str] = set()
|
||||
self._tiles: dict[str, np.ndarray] = {}
|
||||
for ident in self.identities:
|
||||
self._tiles[ident] = self._compose_cell(ident, None)
|
||||
self._canvas = np.zeros((self.height, self.width, 4), dtype=np.uint8)
|
||||
|
||||
def _rebuild_chrome(self) -> None:
|
||||
chrome = np.frombuffer(self._render_mosaic(), dtype=np.uint8).reshape(
|
||||
self.height, self.width, 4)
|
||||
self._i420 = np.frombuffer(bgra_to_i420(chrome), dtype=np.uint8).copy()
|
||||
self._i420_chrome = self._i420.copy()
|
||||
|
||||
def _apply_layout(
|
||||
self,
|
||||
identities: Iterable[str],
|
||||
*,
|
||||
cols: int | None = None,
|
||||
rows: int | None = None,
|
||||
) -> None:
|
||||
ids = [i for i in identities if i]
|
||||
if cols is None or rows is None:
|
||||
cols, rows = grid_shape(len(ids), self.max_cols, self.max_rows)
|
||||
self.identities = ids
|
||||
keep = set(ids)
|
||||
self._live &= keep
|
||||
self._raised &= keep
|
||||
self._speaking &= keep
|
||||
self._layout_metrics(cols, rows)
|
||||
self._tiles = {ident: self._compose_cell(ident, None) for ident in ids}
|
||||
if self._canvas.shape != (self.height, self.width, 4):
|
||||
self._canvas = np.zeros((self.height, self.width, 4), dtype=np.uint8)
|
||||
self._rebuild_chrome()
|
||||
|
||||
def set_identities(self, identities: Iterable[str]) -> None:
|
||||
"""Replace the tile set and shrink/grow cols x rows to match."""
|
||||
ids = [i for i in identities if i]
|
||||
cols, rows = grid_shape(len(ids), self.max_cols, self.max_rows)
|
||||
with self._lock:
|
||||
if ids == self.identities and cols == self.cols and rows == self.rows:
|
||||
return
|
||||
self._apply_layout(ids, cols=cols, rows=rows)
|
||||
|
||||
def add_identity(self, identity: str) -> bool:
|
||||
ident = (identity or "").strip()
|
||||
if not ident:
|
||||
return False
|
||||
if ident in self.identities:
|
||||
return True
|
||||
cap = self.max_cols * self.max_rows
|
||||
if len(self.identities) >= cap:
|
||||
return False
|
||||
ids = sorted(self.identities + [ident])
|
||||
self.set_identities(ids)
|
||||
return ident in self.identities
|
||||
|
||||
def remove_identity(self, identity: str) -> None:
|
||||
ident = (identity or "").strip()
|
||||
if ident not in self.identities:
|
||||
return
|
||||
self.set_identities([i for i in self.identities if i != ident])
|
||||
|
||||
def set_status(self, lines: list[str] | None) -> None:
|
||||
"""Full-frame overlay (LiveKit errors). None restores the mosaic."""
|
||||
cleaned = [str(x) for x in lines if str(x).strip()] if lines else None
|
||||
with self._lock:
|
||||
self._status_lines = cleaned or None
|
||||
|
||||
def _status_frame(self) -> np.ndarray:
|
||||
lines = self._status_lines or ["NO LIVE CAMERAS"]
|
||||
return status_screen(self.width, self.height, lines)
|
||||
|
||||
def tile_origin(self, index: int) -> tuple[int, int]:
|
||||
r, c = divmod(int(index), self.cols)
|
||||
x = self.gap + c * (self.tile_w + self.gap)
|
||||
y = self.gap + r * (self.tile_h + self.gap)
|
||||
return x, y
|
||||
|
||||
def inner_rect(self, index: int) -> tuple[int, int, int, int]:
|
||||
"""Inner video rect (x, y, w, h) for tile index, inside the frame."""
|
||||
x, y = self.tile_origin(index)
|
||||
return x + self.border, y + self.border, self.inner_w, self.inner_h
|
||||
|
||||
def ensure_slot(self, identity: str) -> bool:
|
||||
"""Bind `identity` to an empty tile so remotes can occupy the grid."""
|
||||
if identity in self._tiles:
|
||||
return True
|
||||
for i, ident in enumerate(self.identities):
|
||||
if ident not in self._live:
|
||||
del self._tiles[ident]
|
||||
self.identities[i] = identity
|
||||
self._tiles[identity] = self._compose_cell(identity, None)
|
||||
return True
|
||||
return False
|
||||
|
||||
def _compose_cell(self, identity: str, inner: Optional[np.ndarray]) -> np.ndarray:
|
||||
tw, th = self.tile_w, self.tile_h
|
||||
b = self.border
|
||||
cell = np.empty((th, tw, 4), dtype=np.uint8)
|
||||
cell[:] = CELL_BG
|
||||
frame = FRAME_LIVE if inner is not None else FRAME_COLOR
|
||||
frame = self._frame_color(identity, inner is not None)
|
||||
cv2.rectangle(cell, (0, 0), (tw - 1, th - 1), frame, b)
|
||||
# video / placeholder region
|
||||
iy0, ix0 = b, b
|
||||
@@ -170,19 +824,218 @@ class CameraGrid:
|
||||
)
|
||||
return cell
|
||||
|
||||
def _frame_color(self, identity: str, live: bool) -> tuple[int, int, int, int]:
|
||||
if identity in self._speaking:
|
||||
return FRAME_SPEAK
|
||||
if identity in self._raised:
|
||||
return FRAME_RAISE
|
||||
return FRAME_LIVE if live else FRAME_COLOR
|
||||
|
||||
def highlighted(self, identity: str) -> bool:
|
||||
return identity in self._raised
|
||||
|
||||
def speaking(self, identity: str) -> bool:
|
||||
return identity in self._speaking
|
||||
|
||||
def _paint_border(self, identity: str) -> None:
|
||||
"""Stamp I420 chrome for the current speak/raise/live winner."""
|
||||
try:
|
||||
idx = self.identities.index(identity)
|
||||
except ValueError:
|
||||
return
|
||||
ox, oy = self.tile_origin(idx)
|
||||
color = self._frame_color(identity, identity in self._live)
|
||||
if color == FRAME_SPEAK:
|
||||
yv, uv, vv = self._yuv_speak
|
||||
self._fill_i420_border(self._i420, ox, oy, yv, uv, vv)
|
||||
elif color == FRAME_RAISE:
|
||||
yv, uv, vv = self._yuv_raise
|
||||
self._fill_i420_border(self._i420, ox, oy, yv, uv, vv)
|
||||
else:
|
||||
self._copy_i420_border(self._i420_chrome, self._i420, ox, oy)
|
||||
cell = self._tiles.get(identity)
|
||||
if cell is not None:
|
||||
cv2.rectangle(
|
||||
cell, (0, 0), (self.tile_w - 1, self.tile_h - 1),
|
||||
color, self.border)
|
||||
|
||||
def set_highlight(self, identity: str, raised: bool) -> None:
|
||||
"""Yellow tile chrome when `raised`. Live I420 path only blits the inner rect."""
|
||||
if identity not in self._tiles:
|
||||
return
|
||||
want = bool(raised)
|
||||
with self._lock:
|
||||
had = identity in self._raised
|
||||
if want:
|
||||
self._raised.add(identity)
|
||||
else:
|
||||
self._raised.discard(identity)
|
||||
if had == want:
|
||||
return
|
||||
self._paint_border(identity)
|
||||
|
||||
def set_speakers(self, identities: Iterable[str]) -> None:
|
||||
"""Green tile chrome for LiveKit active speakers (grid identities only)."""
|
||||
want = {i for i in identities if i in self._tiles}
|
||||
with self._lock:
|
||||
previous = set(self._speaking)
|
||||
if want == previous:
|
||||
return
|
||||
self._speaking = want
|
||||
for ident in previous | want:
|
||||
self._paint_border(ident)
|
||||
|
||||
def set_frame(self, identity: str, bgra: np.ndarray) -> None:
|
||||
if identity not in self._tiles:
|
||||
return
|
||||
self._live.add(identity)
|
||||
self._tiles[identity] = self._compose_cell(identity, bgra)
|
||||
|
||||
def _chrome_geom(self, x: int, y: int) -> tuple[int, int, int, int, int]:
|
||||
"""Outer speak/raise ring: original border plus gutter pad."""
|
||||
out = self.chrome_out
|
||||
return x - out, y - out, self.tile_w + 2 * out, self.tile_h + 2 * out, self.border + out
|
||||
|
||||
def _fill_i420_border(
|
||||
self, mosa: np.ndarray, x: int, y: int, yv: int, uv: int, vv: int,
|
||||
) -> None:
|
||||
W, H = self.width, self.height
|
||||
x, y, tw, th, b = self._chrome_geom(x, y)
|
||||
self._stamp_i420_border(mosa, W, H, x, y, tw, th, b, yv, uv, vv)
|
||||
|
||||
def _copy_i420_border(self, src: np.ndarray, dst: np.ndarray, x: int, y: int) -> None:
|
||||
W, H = self.width, self.height
|
||||
x, y, tw, th, b = self._chrome_geom(x, y)
|
||||
x0, y0 = max(0, x), max(0, y)
|
||||
x1, y1 = min(W, x + tw), min(H, y + th)
|
||||
if x1 <= x0 or y1 <= y0 or b <= 0:
|
||||
return
|
||||
oy = W * H
|
||||
ouv = (W // 2) * (H // 2)
|
||||
sy = src[:oy].reshape(H, W)
|
||||
su = src[oy:oy + ouv].reshape(H // 2, W // 2)
|
||||
sv = src[oy + ouv:].reshape(H // 2, W // 2)
|
||||
dy = dst[:oy].reshape(H, W)
|
||||
du = dst[oy:oy + ouv].reshape(H // 2, W // 2)
|
||||
dv = dst[oy + ouv:].reshape(H // 2, W // 2)
|
||||
yt1 = min(y0 + b, y1)
|
||||
yb0 = max(y1 - b, y0)
|
||||
xl1 = min(x0 + b, x1)
|
||||
xr0 = max(x1 - b, x0)
|
||||
dy[y0:yt1, x0:x1] = sy[y0:yt1, x0:x1]
|
||||
dy[yb0:y1, x0:x1] = sy[yb0:y1, x0:x1]
|
||||
dy[y0:y1, x0:xl1] = sy[y0:y1, x0:xl1]
|
||||
dy[y0:y1, xr0:x1] = sy[y0:y1, xr0:x1]
|
||||
ux0, uy0 = x0 // 2, y0 // 2
|
||||
ux1, uy1 = (x1 + 1) // 2, (y1 + 1) // 2
|
||||
ub = max(1, (b + 1) // 2)
|
||||
du[uy0:uy0 + ub, ux0:ux1] = su[uy0:uy0 + ub, ux0:ux1]
|
||||
du[uy1 - ub:uy1, ux0:ux1] = su[uy1 - ub:uy1, ux0:ux1]
|
||||
du[uy0:uy1, ux0:ux0 + ub] = su[uy0:uy1, ux0:ux0 + ub]
|
||||
du[uy0:uy1, ux1 - ub:ux1] = su[uy0:uy1, ux1 - ub:ux1]
|
||||
dv[uy0:uy0 + ub, ux0:ux1] = sv[uy0:uy0 + ub, ux0:ux1]
|
||||
dv[uy1 - ub:uy1, ux0:ux1] = sv[uy1 - ub:uy1, ux0:ux1]
|
||||
dv[uy0:uy1, ux0:ux0 + ub] = sv[uy0:uy1, ux0:ux0 + ub]
|
||||
dv[uy0:uy1, ux1 - ub:ux1] = sv[uy0:uy1, ux1 - ub:ux1]
|
||||
|
||||
@staticmethod
|
||||
def _stamp_i420_border(
|
||||
mosa: np.ndarray, W: int, H: int, x: int, y: int, tw: int, th: int,
|
||||
b: int, yv: int, uv: int, vv: int,
|
||||
) -> None:
|
||||
x0, y0 = max(0, x), max(0, y)
|
||||
x1, y1 = min(W, x + tw), min(H, y + th)
|
||||
if x1 <= x0 or y1 <= y0 or b <= 0:
|
||||
return
|
||||
oy = W * H
|
||||
ouv = (W // 2) * (H // 2)
|
||||
yp = mosa[:oy].reshape(H, W)
|
||||
up = mosa[oy:oy + ouv].reshape(H // 2, W // 2)
|
||||
vp = mosa[oy + ouv:].reshape(H // 2, W // 2)
|
||||
yt1 = min(y0 + b, y1)
|
||||
yb0 = max(y1 - b, y0)
|
||||
xl1 = min(x0 + b, x1)
|
||||
xr0 = max(x1 - b, x0)
|
||||
yp[y0:yt1, x0:x1] = yv
|
||||
yp[yb0:y1, x0:x1] = yv
|
||||
yp[y0:y1, x0:xl1] = yv
|
||||
yp[y0:y1, xr0:x1] = yv
|
||||
ux0, uy0 = x0 // 2, y0 // 2
|
||||
ux1, uy1 = (x1 + 1) // 2, (y1 + 1) // 2
|
||||
ub = max(1, (b + 1) // 2)
|
||||
up[uy0:uy0 + ub, ux0:ux1] = uv
|
||||
up[uy1 - ub:uy1, ux0:ux1] = uv
|
||||
up[uy0:uy1, ux0:ux0 + ub] = uv
|
||||
up[uy0:uy1, ux1 - ub:ux1] = uv
|
||||
vp[uy0:uy0 + ub, ux0:ux1] = vv
|
||||
vp[uy1 - ub:uy1, ux0:ux1] = vv
|
||||
vp[uy0:uy1, ux0:ux0 + ub] = vv
|
||||
vp[uy0:uy1, ux1 - ub:ux1] = vv
|
||||
|
||||
def set_frame_i420(self, identity: str, data: bytes | memoryview,
|
||||
width: int, height: int) -> None:
|
||||
try:
|
||||
idx = self.identities.index(identity)
|
||||
except ValueError:
|
||||
return
|
||||
iw, ih = self.inner_w, self.inner_h
|
||||
fitted = letterbox_i420(data, width, height, iw, ih)
|
||||
with self._lock:
|
||||
self._live.add(identity)
|
||||
self._blit_i420_inner(idx, fitted)
|
||||
|
||||
def _blit_i420_inner(self, index: int, tile: bytes) -> None:
|
||||
x, y, w, h = self.inner_rect(index)
|
||||
x, y, w, h = _even(x), _even(y), _even(w), _even(h)
|
||||
W, H = self.width, self.height
|
||||
need = i420_nbytes(w, h)
|
||||
if len(tile) < need:
|
||||
return
|
||||
src = np.frombuffer(tile, dtype=np.uint8, count=need)
|
||||
ysz = w * h
|
||||
uvsz = (w // 2) * (h // 2)
|
||||
sy = src[:ysz].reshape(h, w)
|
||||
su = src[ysz:ysz + uvsz].reshape(h // 2, w // 2)
|
||||
sv = src[ysz + uvsz:ysz + 2 * uvsz].reshape(h // 2, w // 2)
|
||||
mosa = self._i420
|
||||
oy = W * H
|
||||
ouv = (W // 2) * (H // 2)
|
||||
mosa[:oy].reshape(H, W)[y:y + h, x:x + w] = sy
|
||||
mosa[oy:oy + ouv].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = su
|
||||
mosa[oy + ouv:].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = sv
|
||||
|
||||
def clear(self, identity: str) -> None:
|
||||
if identity not in self._tiles:
|
||||
return
|
||||
self._live.discard(identity)
|
||||
self._raised.discard(identity)
|
||||
self._speaking.discard(identity)
|
||||
self._tiles[identity] = self._compose_cell(identity, None)
|
||||
try:
|
||||
idx = self.identities.index(identity)
|
||||
except ValueError:
|
||||
return
|
||||
x, y = self.tile_origin(idx)
|
||||
w, h = self.tile_w, self.tile_h
|
||||
x, y, w, h = _even(x), _even(y), _even(w), _even(h)
|
||||
W, H = self.width, self.height
|
||||
with self._lock:
|
||||
oy = W * H
|
||||
ouv = (W // 2) * (H // 2)
|
||||
chrome, mosa = self._i420_chrome, self._i420
|
||||
mosa[:oy].reshape(H, W)[y:y + h, x:x + w] = chrome[:oy].reshape(H, W)[y:y + h, x:x + w]
|
||||
mosa[oy:oy + ouv].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = (
|
||||
chrome[oy:oy + ouv].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2])
|
||||
mosa[oy + ouv:].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2] = (
|
||||
chrome[oy + ouv:].reshape(H // 2, W // 2)[y // 2:y // 2 + h // 2, x // 2:x // 2 + w // 2])
|
||||
|
||||
def render(self) -> bytes:
|
||||
def render_i420(self) -> bytes:
|
||||
with self._lock:
|
||||
if self._status_lines:
|
||||
return bgra_to_i420(self._status_frame())
|
||||
return self._i420.tobytes()
|
||||
|
||||
def _render_mosaic(self) -> bytes:
|
||||
canvas = self._canvas
|
||||
canvas[:] = CANVAS_BG
|
||||
g = self.gap
|
||||
@@ -195,3 +1048,9 @@ class CameraGrid:
|
||||
tile = self._tiles[ident]
|
||||
canvas[y : y + self.tile_h, x : x + self.tile_w] = tile
|
||||
return canvas.tobytes()
|
||||
|
||||
def render(self) -> bytes:
|
||||
with self._lock:
|
||||
if self._status_lines:
|
||||
return self._status_frame().tobytes()
|
||||
return self._render_mosaic()
|
||||
|
||||
+56
-22
@@ -189,33 +189,67 @@ def list_outputs() -> list[DrmOutput]:
|
||||
return outputs
|
||||
|
||||
|
||||
def _rank_wall(o: DrmOutput) -> tuple:
|
||||
"""Arc HDMI first, then any connected HDMI. Names shuffle across boots."""
|
||||
return (
|
||||
0 if o.connected else 1,
|
||||
0 if o.driver == "xe" else 1,
|
||||
0 if o.name.startswith("HDMI-A") else 1,
|
||||
o.name,
|
||||
)
|
||||
|
||||
|
||||
def pick_outputs(
|
||||
all_outs: list[DrmOutput],
|
||||
n: int = 3,
|
||||
names: Optional[list[str]] = None,
|
||||
connected_only: bool = False,
|
||||
) -> list[DrmOutput]:
|
||||
"""Choose DRM heads. Stale DISPLAY_CONNECTORS names must not bind a dead port
|
||||
when a connected Arc HDMI is available — the wall has to stay lit."""
|
||||
if names:
|
||||
by_name: dict[str, list[DrmOutput]] = {}
|
||||
for o in all_outs:
|
||||
by_name.setdefault(o.name, []).append(o)
|
||||
missing = [name for name in names if name not in by_name]
|
||||
if missing:
|
||||
have = sorted({o.name for o in all_outs})
|
||||
raise KeyError(f"DRM connector {missing[0]!r} not found. Have: {have}")
|
||||
picked: list[DrmOutput | None] = []
|
||||
used: set[str] = set()
|
||||
for name in names:
|
||||
cands = sorted(by_name[name], key=_rank_wall)
|
||||
chosen = next((o for o in cands if o.connected and o.sysfs not in used), None)
|
||||
if chosen is None:
|
||||
picked.append(None)
|
||||
else:
|
||||
used.add(chosen.sysfs)
|
||||
picked.append(chosen)
|
||||
fallback = sorted(
|
||||
(o for o in all_outs if o.connected and o.sysfs not in used),
|
||||
key=_rank_wall,
|
||||
)
|
||||
fi = 0
|
||||
for i, o in enumerate(picked):
|
||||
if o is None and fi < len(fallback):
|
||||
picked[i] = fallback[fi]
|
||||
used.add(fallback[fi].sysfs)
|
||||
fi += 1
|
||||
return [o for o in picked if o is not None]
|
||||
connected = [o for o in all_outs if o.connected]
|
||||
rest = [o for o in all_outs if not o.connected]
|
||||
connected.sort(key=_rank_wall)
|
||||
rest.sort(key=_rank_wall)
|
||||
ordered = connected + ([] if connected_only else rest)
|
||||
return ordered[:n]
|
||||
|
||||
|
||||
def select_outputs(
|
||||
n: int = 3,
|
||||
names: Optional[list[str]] = None,
|
||||
connected_only: bool = False,
|
||||
) -> list[DrmOutput]:
|
||||
all_outs = list_outputs()
|
||||
if names:
|
||||
by_name = {o.name: o for o in all_outs}
|
||||
picked = []
|
||||
for name in names:
|
||||
if name not in by_name:
|
||||
raise KeyError(f"DRM connector {name!r} not found. Have: {sorted(by_name)}")
|
||||
picked.append(by_name[name])
|
||||
return picked
|
||||
connected = [o for o in all_outs if o.connected]
|
||||
rest = [o for o in all_outs if not o.connected]
|
||||
# Prefer the discrete GPU (xe / Arc) HDMI heads for a 3-display wall.
|
||||
def rank(o: DrmOutput) -> tuple:
|
||||
return (
|
||||
0 if o.driver == "xe" else 1,
|
||||
0 if o.name.startswith("HDMI-A") else 1,
|
||||
o.name,
|
||||
)
|
||||
connected.sort(key=rank)
|
||||
rest.sort(key=rank)
|
||||
ordered = connected + ([] if connected_only else rest)
|
||||
return ordered[:n]
|
||||
return pick_outputs(list_outputs(), n=n, names=names, connected_only=connected_only)
|
||||
|
||||
|
||||
def format_outputs(outs: list[DrmOutput]) -> str:
|
||||
|
||||
@@ -0,0 +1,324 @@
|
||||
"""I420 SHM -> iGPU H.264 (or Arc AV1) -> latest-AU mmap. Does not open /dev/camNN."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import select
|
||||
import shutil
|
||||
import signal
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
from annexb import AnnexBSplitter, nal_unit_type
|
||||
from av1_obu import Av1TuSplitter, is_keyframe as av1_is_keyframe
|
||||
from frame_shm import LatestFrameReader, frame_bytes, wait_for_ready
|
||||
from i420_nv12 import coded_dim, i420_to_nv12_into, nv12_bytes
|
||||
from nal_shm import AU_MAX, NalWriter
|
||||
from portrait import i420_source_path
|
||||
|
||||
log = logging.getLogger("cameras.encode-daemon")
|
||||
GST = shutil.which("gst-launch-1.0") or "gst-launch-1.0"
|
||||
READY_NAME = "ready"
|
||||
GROUP = 5
|
||||
|
||||
|
||||
def _pipes(n: int) -> tuple[list[int], list[int]]:
|
||||
rs, ws = [], []
|
||||
for _ in range(n):
|
||||
r, w = os.pipe()
|
||||
try:
|
||||
import fcntl
|
||||
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 1048576)
|
||||
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 1048576)
|
||||
except OSError:
|
||||
pass
|
||||
rs.append(r)
|
||||
ws.append(w)
|
||||
return rs, ws
|
||||
|
||||
|
||||
def _one_enc(width: int, height: int, fps: int, bitrate_kbps: int,
|
||||
fd_in: int, fd_out: int, codec: str = "h264",
|
||||
render: str = "/dev/dri/renderD129", hi_kbps: int = 20000) -> list[str]:
|
||||
need = frame_bytes(width, height)
|
||||
kbps = max(100, int(bitrate_kbps))
|
||||
hi = int(height) >= 720
|
||||
if hi:
|
||||
kbps = max(kbps, max(100, int(hi_kbps)))
|
||||
# GOP=1: a 1s GOP left motion trails until the next IDR. Rally is PTZ.
|
||||
key_int = 1
|
||||
usage = 2 if hi else 7
|
||||
cpb = max(80, kbps // 8)
|
||||
if codec == "av1":
|
||||
# HW AV1 is NV12-only (no I420 entrypoint). Convert is CPU I420→NV12
|
||||
# in this process; gst is parse → vaav1enc only (no vapostproc).
|
||||
enc = [
|
||||
"!", "vaav1enc", f"bitrate={kbps}", f"key-int-max={int(fps)}",
|
||||
"target-usage=7", "hierarchical-level=1", "ref-frames=1",
|
||||
"gf-group-size=1", "rate-control=cbr",
|
||||
f"cpb-size={cpb}",
|
||||
]
|
||||
fmt = "nv12"
|
||||
else:
|
||||
# iGPU H.264. NV12-only. GOP=1 so fdsrc preroll does not hang.
|
||||
# vah264enc.device-path is read-only; use the per-node factory.
|
||||
enc_name = f"va{Path(render).name}h264enc"
|
||||
enc = [
|
||||
"!", enc_name, f"bitrate={kbps}",
|
||||
f"key-int-max={key_int}", "b-frames=0", "ref-frames=1",
|
||||
f"target-usage={usage}", "rate-control=cbr", f"cpb-size={cpb}",
|
||||
"!", "h264parse", "config-interval=-1",
|
||||
"!", "video/x-h264,stream-format=byte-stream,alignment=au",
|
||||
]
|
||||
fmt = "nv12"
|
||||
need = nv12_bytes(width, height)
|
||||
head = [
|
||||
"fdsrc", f"fd={int(fd_in)}", f"blocksize={need}",
|
||||
"do-timestamp=true",
|
||||
"!", "rawvideoparse", f"width={int(width)}", f"height={int(height)}",
|
||||
f"format={fmt}", f"framerate={int(fps)}/1",
|
||||
"!", "queue", "max-size-buffers=1", "max-size-bytes=0", "max-size-time=0",
|
||||
"leaky=downstream",
|
||||
]
|
||||
tail = [
|
||||
"!", "queue", "max-size-buffers=1", "max-size-bytes=0", "max-size-time=0",
|
||||
"leaky=downstream",
|
||||
"!", "fdsink", f"fd={int(fd_out)}", "sync=false",
|
||||
]
|
||||
return head + enc + tail
|
||||
|
||||
|
||||
def grouped(cams: list[dict], size: int) -> list[list[dict]]:
|
||||
by: dict[tuple[int, int], list[dict]] = {}
|
||||
for c in cams:
|
||||
key = (int(c["width"]), int(c["height"]))
|
||||
by.setdefault(key, []).append(c)
|
||||
out: list[list[dict]] = []
|
||||
for group in by.values():
|
||||
for i in range(0, len(group), size):
|
||||
out.append(group[i:i + size])
|
||||
return out
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--cameras-json", required=True)
|
||||
ap.add_argument("--i420-dir", default="/run/livekit-cameras/raw")
|
||||
ap.add_argument("--h264-dir", default="/run/livekit-cameras/h264")
|
||||
ap.add_argument("--fps", type=int, default=30)
|
||||
ap.add_argument("--bitrate", type=int, default=2_000_000)
|
||||
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
|
||||
ap.add_argument("--group-size", type=int, default=GROUP)
|
||||
ap.add_argument("--codec", default="h264")
|
||||
ap.add_argument("--hi-bitrate", type=int, default=20_000_000,
|
||||
help="bps for height>=720 (Rally). Rollback: 12000000")
|
||||
ap.add_argument("--portrait-dir", default="",
|
||||
help="C920 I420 from portrait daemon; rally stays in --i420-dir")
|
||||
args = ap.parse_args(argv)
|
||||
codec = (args.codec or "h264").lower()
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [encode-daemon] %(levelname)s %(message)s")
|
||||
cams = json.loads(args.cameras_json)
|
||||
if not cams:
|
||||
log.error("no cameras")
|
||||
return 2
|
||||
|
||||
from vaapi_pin import igpu_render
|
||||
|
||||
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
|
||||
os.environ["GST_VA_ALL_DRIVERS"] = "1"
|
||||
os.environ.pop("LIBVA_DRM_DEVICE", None)
|
||||
encode_render = igpu_render()
|
||||
if codec == "av1":
|
||||
log.info("encoder=vaav1enc card0=/dev/dri/renderD128 codec=%s", codec)
|
||||
else:
|
||||
log.info("encoder=%s codec=%s", f"va{Path(encode_render).name}h264enc", codec)
|
||||
|
||||
os.makedirs(args.h264_dir, exist_ok=True)
|
||||
kbps = max(100, int(args.bitrate) // 1000)
|
||||
stop = False
|
||||
|
||||
def _stop(*_a):
|
||||
nonlocal stop
|
||||
stop = True
|
||||
|
||||
signal.signal(signal.SIGTERM, _stop)
|
||||
signal.signal(signal.SIGINT, _stop)
|
||||
|
||||
gst_procs: list[subprocess.Popen] = []
|
||||
states = []
|
||||
extra_fds: list[int] = []
|
||||
|
||||
for group in grouped(cams, 1 if codec == "av1" else max(1, args.group_size)):
|
||||
n = len(group)
|
||||
in_r, in_w = _pipes(n)
|
||||
out_r, out_w = _pipes(n)
|
||||
for fd in in_r + out_w:
|
||||
os.set_inheritable(fd, True)
|
||||
for fd in in_w + out_r:
|
||||
os.set_inheritable(fd, False)
|
||||
for cam, w_in in zip(group, in_w):
|
||||
w, h = int(cam["width"]), int(cam["height"])
|
||||
if codec in ("av1", "h264"):
|
||||
need = nv12_bytes(coded_dim(w), coded_dim(h))
|
||||
else:
|
||||
need = frame_bytes(w, h)
|
||||
try:
|
||||
import fcntl
|
||||
fcntl.fcntl(w_in, fcntl.F_SETPIPE_SZ, max(1048576, need + 4096))
|
||||
except OSError:
|
||||
pass
|
||||
os.write(w_in, bytes(need))
|
||||
for fd in in_w + out_r:
|
||||
os.set_blocking(fd, False)
|
||||
cmd = [GST, "-q"]
|
||||
for cam, r, wfd in zip(group, in_r, out_w):
|
||||
w, h = int(cam["width"]), int(cam["height"])
|
||||
if codec in ("av1", "h264"):
|
||||
w, h = coded_dim(w), coded_dim(h)
|
||||
cmd += _one_enc(w, h, args.fps, kbps, r, wfd, codec,
|
||||
render=encode_render,
|
||||
hi_kbps=max(100, int(args.hi_bitrate) // 1000))
|
||||
env = dict(os.environ)
|
||||
env["GST_REGISTRY_UPDATE"] = "no"
|
||||
proc = subprocess.Popen(
|
||||
cmd, pass_fds=tuple(in_r + out_w), env=env,
|
||||
stderr=subprocess.STDOUT)
|
||||
gst_procs.append(proc)
|
||||
for fd in in_r + out_w:
|
||||
extra_fds.append(fd)
|
||||
for cam, w_in, r_out in zip(group, in_w, out_r):
|
||||
ident = cam["identity"]
|
||||
w, h = int(cam["width"]), int(cam["height"])
|
||||
i420 = i420_source_path(ident, args.i420_dir, args.portrait_dir)
|
||||
h264 = os.path.join(args.h264_dir, f"{ident}.h264")
|
||||
au_max = 512 * 1024 if codec == "av1" else AU_MAX
|
||||
i420_n = frame_bytes(w, h)
|
||||
if codec in ("av1", "h264"):
|
||||
cw, ch = coded_dim(w), coded_dim(h)
|
||||
pipe_n = nv12_bytes(cw, ch)
|
||||
nv12 = bytearray(pipe_n)
|
||||
else:
|
||||
pipe_n = i420_n
|
||||
nv12 = None
|
||||
states.append({
|
||||
"ident": ident, "w": w, "h": h,
|
||||
"need": pipe_n,
|
||||
"reader": LatestFrameReader(i420),
|
||||
"writer": NalWriter(h264, max_payload=au_max),
|
||||
"in_w": w_in, "out_r": r_out,
|
||||
"scratch": bytearray(i420_n),
|
||||
"nv12": nv12,
|
||||
"last": -1,
|
||||
"split": Av1TuSplitter() if codec == "av1" else AnnexBSplitter(),
|
||||
"sps": b"", "pps": b"", "woff": 0, "wseq": -1,
|
||||
"codec": codec,
|
||||
})
|
||||
extra_fds.extend([w_in, r_out])
|
||||
log.info("gst pid=%s cams=%s", proc.pid,
|
||||
",".join(c["identity"] for c in group))
|
||||
|
||||
states.sort(key=lambda s: 0 if s["ident"] == "rally" else 1)
|
||||
|
||||
ready = os.path.join(args.h264_dir, READY_NAME)
|
||||
Path(ready).write_text("ok\n")
|
||||
|
||||
out_map = {s["out_r"]: s for s in states}
|
||||
t0 = time.monotonic()
|
||||
seqs = {s["ident"]: 0 for s in states}
|
||||
try:
|
||||
while not stop:
|
||||
for s in states:
|
||||
pipe = s["nv12"] if s.get("nv12") is not None else s["scratch"]
|
||||
if s["woff"]:
|
||||
view = memoryview(pipe)
|
||||
try:
|
||||
n = os.write(s["in_w"], view[s["woff"]:])
|
||||
s["woff"] += n
|
||||
if s["woff"] >= s["need"]:
|
||||
s["woff"] = 0
|
||||
s["last"] = s["wseq"]
|
||||
except BlockingIOError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
seq = s["reader"].copy_into(s["scratch"])
|
||||
if seq is None or seq == s["last"]:
|
||||
continue
|
||||
if s.get("nv12") is not None:
|
||||
i420_to_nv12_into(s["scratch"], s["w"], s["h"], s["nv12"])
|
||||
pipe = s["nv12"]
|
||||
view = memoryview(pipe)
|
||||
try:
|
||||
n = os.write(s["in_w"], view)
|
||||
if n < s["need"]:
|
||||
s["woff"] = n
|
||||
s["wseq"] = seq
|
||||
else:
|
||||
s["last"] = seq
|
||||
except BlockingIOError:
|
||||
s["woff"] = 0
|
||||
except OSError:
|
||||
pass
|
||||
rds, _, _ = select.select(list(out_map), [], [], 0.01)
|
||||
for fd in rds:
|
||||
s = out_map[fd]
|
||||
try:
|
||||
chunk = os.read(fd, 65536)
|
||||
except OSError:
|
||||
continue
|
||||
if not chunk:
|
||||
continue
|
||||
for nal in s["split"].push(chunk):
|
||||
if s.get("codec") == "av1":
|
||||
if av1_is_keyframe(nal) or nal:
|
||||
seqs[s["ident"]] = s["writer"].write(nal)
|
||||
continue
|
||||
ntype = nal_unit_type(nal)
|
||||
if ntype == 7:
|
||||
s["sps"] = nal
|
||||
continue
|
||||
if ntype == 8:
|
||||
s["pps"] = nal
|
||||
continue
|
||||
if ntype == 5:
|
||||
payload = s["sps"] + s["pps"] + nal
|
||||
elif ntype in (1, 2, 3, 4):
|
||||
payload = nal
|
||||
else:
|
||||
continue
|
||||
seqs[s["ident"]] = s["writer"].write(payload)
|
||||
if time.monotonic() - t0 >= 10:
|
||||
log.info("seqs %s", " ".join(
|
||||
f"{k}:{v}" for k, v in sorted(seqs.items())))
|
||||
t0 = time.monotonic()
|
||||
for p in gst_procs:
|
||||
if p.poll() is not None:
|
||||
log.error("gst exited rc=%s", p.returncode)
|
||||
stop = True
|
||||
break
|
||||
finally:
|
||||
for p in gst_procs:
|
||||
if p.poll() is None:
|
||||
p.terminate()
|
||||
for fd in extra_fds:
|
||||
try:
|
||||
os.close(fd)
|
||||
except OSError:
|
||||
pass
|
||||
try:
|
||||
os.unlink(ready)
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,48 @@
|
||||
"""ctypes: push pre-encoded AUs into LiveKit VideoSource (pass-through)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
|
||||
_CODEC = {"h264": 0, "h265": 1, "vp8": 2, "vp9": 3, "av1": 4}
|
||||
|
||||
|
||||
def capture_encoded(
|
||||
video_source,
|
||||
payload: bytes,
|
||||
width: int,
|
||||
height: int,
|
||||
keyframe: bool,
|
||||
timestamp_us: int = 0,
|
||||
codec: str = "h264",
|
||||
) -> int:
|
||||
from livekit.rtc._ffi_client import FfiClient
|
||||
|
||||
if not payload:
|
||||
return -1
|
||||
lib = FfiClient.instance._ffi_lib
|
||||
fn = lib.livekit_ffi_capture_encoded
|
||||
fn.argtypes = [
|
||||
ctypes.c_uint64,
|
||||
ctypes.c_void_p,
|
||||
ctypes.c_size_t,
|
||||
ctypes.c_int32,
|
||||
ctypes.c_int32,
|
||||
ctypes.c_int32,
|
||||
ctypes.c_int64,
|
||||
ctypes.c_int32,
|
||||
]
|
||||
fn.restype = ctypes.c_int32
|
||||
handle = int(video_source._ffi_handle.handle)
|
||||
buf = (ctypes.c_uint8 * len(payload)).from_buffer_copy(payload)
|
||||
return int(
|
||||
fn(
|
||||
handle,
|
||||
ctypes.cast(buf, ctypes.c_void_p),
|
||||
len(payload),
|
||||
int(width),
|
||||
int(height),
|
||||
1 if keyframe else 0,
|
||||
int(timestamp_us),
|
||||
int(_CODEC.get(codec, 0)),
|
||||
)
|
||||
)
|
||||
@@ -0,0 +1,166 @@
|
||||
"""Shared MetricGAN daemon: one model, many camera clients."""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from typing import Callable, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
from enhance_ipc import read_frame, write_frame
|
||||
|
||||
log = logging.getLogger("audio.enhance_daemon")
|
||||
|
||||
Handler = Callable[[str, np.ndarray], np.ndarray]
|
||||
|
||||
|
||||
class SharedEnhanceEngine:
|
||||
"""One AudioCleaner model; per-identity context and TDOA."""
|
||||
|
||||
def __init__(self, make_cleaner: Callable):
|
||||
self._template = make_cleaner()
|
||||
self._sessions: dict[str, object] = {}
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def identities(self) -> list[str]:
|
||||
return list(self._sessions)
|
||||
|
||||
def _session(self, ident: str):
|
||||
sess = self._sessions.get(ident)
|
||||
if sess is not None:
|
||||
return sess
|
||||
sess = self._clone()
|
||||
self._sessions[ident] = sess
|
||||
return sess
|
||||
|
||||
def _clone(self):
|
||||
import copy
|
||||
sess = copy.copy(self._template)
|
||||
sess._ctx = np.zeros(0, dtype=np.float32)
|
||||
sess._last_tdoas = None
|
||||
sess._tdoa_age = 0
|
||||
sess._client = None
|
||||
sess.last_dt_ms = -1.0
|
||||
compiled = getattr(sess, "_ov_compiled", None)
|
||||
if compiled is not None:
|
||||
sess._ov_request = compiled.create_infer_request()
|
||||
return sess
|
||||
|
||||
def process(self, ident: str, pcm: np.ndarray) -> np.ndarray:
|
||||
with self._lock:
|
||||
sess = self._session(ident)
|
||||
return sess.process(pcm)
|
||||
|
||||
|
||||
def serve_unix(path: str, handler: Handler, stop: Optional[threading.Event] = None) -> None:
|
||||
"""Accept unix-socket clients; handler(ident, pcm) -> int16 mono."""
|
||||
if os.path.exists(path):
|
||||
os.unlink(path)
|
||||
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
||||
srv = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
srv.bind(path)
|
||||
srv.listen(64)
|
||||
srv.settimeout(0.2)
|
||||
|
||||
def _client(conn: socket.socket) -> None:
|
||||
try:
|
||||
while stop is None or not stop.is_set():
|
||||
ident, pcm = read_frame(conn)
|
||||
out = handler(ident, pcm)
|
||||
out = np.asarray(out, dtype=np.int16)
|
||||
write_frame(conn, ident, out)
|
||||
except (EOFError, ConnectionError, BrokenPipeError, TimeoutError, OSError):
|
||||
pass
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
threads: list[threading.Thread] = []
|
||||
try:
|
||||
while stop is None or not stop.is_set():
|
||||
try:
|
||||
conn, _ = srv.accept()
|
||||
except TimeoutError:
|
||||
continue
|
||||
conn.settimeout(15.0)
|
||||
t = threading.Thread(target=_client, args=(conn,), daemon=True)
|
||||
t.start()
|
||||
threads.append(t)
|
||||
finally:
|
||||
srv.close()
|
||||
if os.path.exists(path):
|
||||
os.unlink(path)
|
||||
|
||||
|
||||
def wait_for_socket(path: str, timeout: float = 90.0) -> bool:
|
||||
"""Return True once a unix socket at path accepts a connection."""
|
||||
deadline = time.time() + timeout
|
||||
while time.time() < deadline:
|
||||
if os.path.exists(path):
|
||||
try:
|
||||
s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
s.settimeout(0.3)
|
||||
s.connect(path)
|
||||
s.close()
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
|
||||
|
||||
def make_production_cleaner(**kw):
|
||||
from audio_cleanup import AudioCleaner, apply_torch_thread_limits
|
||||
|
||||
threads = int(kw.pop("torch_num_threads", 2))
|
||||
apply_torch_thread_limits(threads)
|
||||
return AudioCleaner(torch_num_threads=threads, **kw)
|
||||
|
||||
|
||||
def main(argv: Optional[list[str]] = None) -> int:
|
||||
import argparse
|
||||
|
||||
ap = argparse.ArgumentParser(description="Shared MetricGAN enhance daemon")
|
||||
ap.add_argument("--socket", default="/tmp/livekit-enhance.sock")
|
||||
ap.add_argument("--enhance-mode", default="force")
|
||||
ap.add_argument("--enhance-model",
|
||||
default="speechbrain/metricgan-plus-voicebank")
|
||||
ap.add_argument("--enhance-chunk-s", type=float, default=1.0)
|
||||
ap.add_argument("--enhance-hop-s", type=float, default=0.1)
|
||||
ap.add_argument("--enhance-infer", default="auto")
|
||||
ap.add_argument("--beamform-mode", default="auto")
|
||||
ap.add_argument("--torch-num-threads", type=int, default=2)
|
||||
ap.add_argument("--tdoa-every", type=int, default=5)
|
||||
ap.add_argument("--audio-rate", type=int, default=16000)
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [enhance-daemon] %(levelname)s %(message)s")
|
||||
log.info("loading shared MetricGAN (one copy for all cameras)")
|
||||
|
||||
def make():
|
||||
return make_production_cleaner(
|
||||
mode=args.enhance_mode,
|
||||
model_source=args.enhance_model,
|
||||
sample_rate=args.audio_rate,
|
||||
chunk_s=args.enhance_chunk_s,
|
||||
hop_s=args.enhance_hop_s,
|
||||
beamform=args.beamform_mode,
|
||||
torch_num_threads=args.torch_num_threads,
|
||||
tdoa_every=args.tdoa_every,
|
||||
enhance_infer=args.enhance_infer,
|
||||
)
|
||||
|
||||
engine = SharedEnhanceEngine(make)
|
||||
log.info("model ready backend=%s socket=%s",
|
||||
getattr(engine._template, "backend", "?"), args.socket)
|
||||
serve_unix(args.socket, engine.process)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,87 @@
|
||||
"""Length-prefixed int16 audio frames for the shared enhance daemon."""
|
||||
from __future__ import annotations
|
||||
|
||||
import socket
|
||||
import struct
|
||||
|
||||
import numpy as np
|
||||
|
||||
MAGIC = b"ENH1"
|
||||
# magic(4) + ident_len(H) + channels(H) + n_samples(I)
|
||||
_HEADER = struct.Struct("!4sHHI")
|
||||
|
||||
|
||||
def dump_frame(identity: str, pcm: np.ndarray) -> bytes:
|
||||
"""Serialize identity + int16 pcm ([N] or [N, C]) to one frame."""
|
||||
pcm = np.asarray(pcm, dtype=np.int16)
|
||||
if pcm.ndim == 1:
|
||||
pcm = pcm.reshape(-1, 1)
|
||||
elif pcm.ndim != 2:
|
||||
raise ValueError(f"pcm must be 1-D or 2-D, got {pcm.shape}")
|
||||
n_samples, channels = pcm.shape
|
||||
ident = identity.encode("utf-8")
|
||||
header = _HEADER.pack(MAGIC, len(ident), channels, n_samples)
|
||||
return header + ident + np.ascontiguousarray(pcm).tobytes()
|
||||
|
||||
|
||||
def load_frame(blob: bytes) -> tuple[str, np.ndarray]:
|
||||
"""Parse a frame into (identity, int16 array)."""
|
||||
if len(blob) < _HEADER.size:
|
||||
raise ValueError("truncated enhance frame")
|
||||
magic, ident_len, channels, n_samples = _HEADER.unpack_from(blob, 0)
|
||||
if magic != MAGIC:
|
||||
raise ValueError(f"bad enhance magic {magic!r}")
|
||||
ident_end = _HEADER.size + ident_len
|
||||
payload = blob[ident_end:]
|
||||
need = n_samples * channels * 2
|
||||
if len(payload) != need:
|
||||
raise ValueError(f"pcm size {len(payload)} != {need}")
|
||||
ident = blob[_HEADER.size:ident_end].decode("utf-8")
|
||||
pcm = np.frombuffer(payload, dtype=np.int16).reshape(n_samples, channels)
|
||||
if channels == 1:
|
||||
pcm = pcm[:, 0]
|
||||
return ident, pcm
|
||||
|
||||
|
||||
def recvall(sock: socket.socket, n: int) -> bytes:
|
||||
buf = bytearray()
|
||||
while len(buf) < n:
|
||||
chunk = sock.recv(n - len(buf))
|
||||
if not chunk:
|
||||
raise EOFError("enhance socket closed")
|
||||
buf.extend(chunk)
|
||||
return bytes(buf)
|
||||
|
||||
|
||||
def read_frame(sock: socket.socket) -> tuple[str, np.ndarray]:
|
||||
header = recvall(sock, _HEADER.size)
|
||||
magic, ident_len, channels, n_samples = _HEADER.unpack(header)
|
||||
if magic != MAGIC:
|
||||
raise ValueError(f"bad enhance magic {magic!r}")
|
||||
rest = recvall(sock, ident_len + n_samples * channels * 2)
|
||||
return load_frame(header + rest)
|
||||
|
||||
|
||||
def write_frame(sock: socket.socket, identity: str, pcm: np.ndarray) -> None:
|
||||
sock.sendall(dump_frame(identity, pcm))
|
||||
|
||||
|
||||
class EnhanceClient:
|
||||
"""One persistent unix-socket connection to the enhance daemon."""
|
||||
|
||||
def __init__(self, path: str):
|
||||
self.path = path
|
||||
self._sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
self._sock.settimeout(15.0)
|
||||
self._sock.connect(path)
|
||||
|
||||
def process(self, identity: str, pcm: np.ndarray) -> np.ndarray:
|
||||
write_frame(self._sock, identity, pcm)
|
||||
_, out = read_frame(self._sock)
|
||||
return np.asarray(out, dtype=np.int16)
|
||||
|
||||
def close(self) -> None:
|
||||
try:
|
||||
self._sock.close()
|
||||
except OSError:
|
||||
pass
|
||||
+136
@@ -0,0 +1,136 @@
|
||||
"""Latest-frame I420 slots in mmap files (single writer, many readers).
|
||||
|
||||
Double-buffered: the writer fills the inactive slot, then publishes seq.
|
||||
Readers copy the active slot; a torn read (seq changed) is discarded.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import mmap
|
||||
import os
|
||||
import struct
|
||||
import time
|
||||
from typing import Optional
|
||||
|
||||
MAGIC = b"LKFR"
|
||||
VER = 2
|
||||
HEADER = 64
|
||||
_HDR = struct.Struct("<4sIIIIIQ") # magic, ver, w, h, seq, nbytes, ts_ns
|
||||
SLOTS = 2
|
||||
|
||||
|
||||
def frame_bytes(width: int, height: int) -> int:
|
||||
return int(width) * int(height) * 3 // 2
|
||||
|
||||
|
||||
def slot_size(width: int, height: int) -> int:
|
||||
return HEADER + SLOTS * frame_bytes(width, height)
|
||||
|
||||
|
||||
def _mmap(fd: int, size: int, writable: bool) -> mmap.mmap:
|
||||
prot = mmap.PROT_READ | (mmap.PROT_WRITE if writable else 0)
|
||||
return mmap.mmap(fd, size, flags=mmap.MAP_SHARED, prot=prot)
|
||||
|
||||
|
||||
class LatestFrameWriter:
|
||||
def __init__(self, path: str, width: int, height: int):
|
||||
self.path = path
|
||||
self.width = int(width)
|
||||
self.height = int(height)
|
||||
self.nbytes = frame_bytes(self.width, self.height)
|
||||
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
||||
size = slot_size(self.width, self.height)
|
||||
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
|
||||
try:
|
||||
os.ftruncate(fd, size)
|
||||
self._map = _mmap(fd, size, True)
|
||||
finally:
|
||||
os.close(fd)
|
||||
self._seq = 0
|
||||
self._pack_header(0)
|
||||
|
||||
def _pack_header(self, seq: int) -> None:
|
||||
_HDR.pack_into(
|
||||
self._map, 0, MAGIC, VER, self.width, self.height, seq, self.nbytes, 0)
|
||||
|
||||
def write(self, payload: bytes, ts_ns: int = 0) -> int:
|
||||
if len(payload) < self.nbytes:
|
||||
return self._seq
|
||||
nxt = self._seq + 1
|
||||
slot = nxt % SLOTS
|
||||
off = HEADER + slot * self.nbytes
|
||||
self._map[off:off + self.nbytes] = payload[:self.nbytes]
|
||||
self._seq = nxt
|
||||
_HDR.pack_into(
|
||||
self._map, 0, MAGIC, VER, self.width, self.height,
|
||||
self._seq, self.nbytes, int(ts_ns))
|
||||
return self._seq
|
||||
|
||||
def close(self) -> None:
|
||||
self._map.close()
|
||||
|
||||
|
||||
class LatestFrameReader:
|
||||
def __init__(self, path: str):
|
||||
self.path = path
|
||||
self._map: Optional[mmap.mmap] = None
|
||||
self.width = 0
|
||||
self.height = 0
|
||||
self.nbytes = 0
|
||||
|
||||
def _open(self) -> bool:
|
||||
if self._map is not None:
|
||||
return True
|
||||
try:
|
||||
fd = os.open(self.path, os.O_RDONLY)
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
try:
|
||||
st = os.fstat(fd)
|
||||
if st.st_size < HEADER:
|
||||
return False
|
||||
buf = _mmap(fd, st.st_size, False)
|
||||
finally:
|
||||
os.close(fd)
|
||||
magic, ver, w, h, _seq, nbytes, _ts = _HDR.unpack_from(buf, 0)
|
||||
if magic != MAGIC or ver != VER or nbytes <= 0:
|
||||
buf.close()
|
||||
return False
|
||||
if st.st_size < HEADER + SLOTS * nbytes:
|
||||
buf.close()
|
||||
return False
|
||||
self._map = buf
|
||||
self.width, self.height, self.nbytes = w, h, nbytes
|
||||
return True
|
||||
|
||||
def latest(self) -> Optional[tuple[int, bytes]]:
|
||||
buf = bytearray(self.nbytes or frame_bytes(16, 16))
|
||||
seq = self.copy_into(buf)
|
||||
if seq is None:
|
||||
return None
|
||||
return seq, bytes(buf[:self.nbytes])
|
||||
|
||||
def copy_into(self, dest: bytearray) -> Optional[int]:
|
||||
"""Copy the active slot into dest (must be >= nbytes). None if torn/empty."""
|
||||
if not self._open() or self._map is None:
|
||||
return None
|
||||
if len(dest) < self.nbytes:
|
||||
raise ValueError("dest too small")
|
||||
s1 = struct.unpack_from("<I", self._map, 16)[0]
|
||||
if s1 == 0:
|
||||
return None
|
||||
slot = s1 % SLOTS
|
||||
off = HEADER + slot * self.nbytes
|
||||
dest[:self.nbytes] = self._map[off:off + self.nbytes]
|
||||
s2 = struct.unpack_from("<I", self._map, 16)[0]
|
||||
if s1 != s2:
|
||||
return None
|
||||
return s1
|
||||
|
||||
|
||||
def wait_for_ready(path: str, timeout: float = 90.0) -> bool:
|
||||
deadline = time.time() + timeout
|
||||
while time.time() < deadline:
|
||||
if os.path.exists(path):
|
||||
return True
|
||||
time.sleep(0.05)
|
||||
return False
|
||||
@@ -13,6 +13,8 @@ import re
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
from usb_ids import is_fleet_video, is_rally_pid, usb_vid_pid_from_sysfs
|
||||
|
||||
HEADER = """\
|
||||
# Stable slots for 20x Logitech HD Pro Webcam C920 (046d:08e5).
|
||||
# Slot = physical camera, identified by USB serial. Survives reboots,
|
||||
@@ -67,6 +69,11 @@ def current_state():
|
||||
if idx_attr != "0":
|
||||
continue
|
||||
real = sh(["readlink", "-f", str(node / "device")]).strip()
|
||||
ids = usb_vid_pid_from_sysfs(real)
|
||||
if ids is not None and not is_fleet_video(ids[0], ids[1]):
|
||||
continue
|
||||
if ids is not None and is_rally_pid(ids[1]):
|
||||
continue
|
||||
s = _serial_of(real)
|
||||
primary_by_serial[s] = f"/dev/{node.name}"
|
||||
return primary_by_serial, cards
|
||||
|
||||
+299
-10
@@ -3,6 +3,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import fcntl
|
||||
import shutil
|
||||
import subprocess
|
||||
from typing import Optional
|
||||
@@ -39,21 +40,131 @@ def test_pattern_cmd(out: DrmOutput, pattern: str = "smpte", width: int = 1280,
|
||||
]
|
||||
|
||||
|
||||
def appsrc_cmd(out: DrmOutput, width: int, height: int, fps: int = 30) -> list[str]:
|
||||
"""Raw BGRA frames on stdin -> kmssink."""
|
||||
block = width * height * 4
|
||||
def appsrc_cmd(
|
||||
out: DrmOutput,
|
||||
width: int,
|
||||
height: int,
|
||||
fps: int = 30,
|
||||
queue_buffers: int = 2,
|
||||
pixel_format: str = "bgra",
|
||||
) -> list[str]:
|
||||
"""Raw BGRA or I420 frames on stdin -> kmssink."""
|
||||
fmt = (pixel_format or "bgra").strip().lower()
|
||||
if fmt in ("i420", "yuv420p"):
|
||||
fmt = "i420"
|
||||
block = width * height * 3 // 2
|
||||
else:
|
||||
fmt = "bgra"
|
||||
block = width * height * 4
|
||||
q = max(1, int(queue_buffers))
|
||||
return [
|
||||
GST_LAUNCH, "-e",
|
||||
"fdsrc", "fd=0", f"blocksize={block}",
|
||||
"!", "videoparse",
|
||||
f"width={width}", f"height={height}", "format=bgra", f"framerate={fps}/1",
|
||||
"!", "queue", "max-size-buffers=2", "leaky=downstream",
|
||||
f"width={width}", f"height={height}", f"format={fmt}", f"framerate={fps}/1",
|
||||
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
|
||||
"!", "videoconvert",
|
||||
"!", "videoscale", "method=0",
|
||||
"!", *_kms_props(out),
|
||||
]
|
||||
|
||||
|
||||
def va_grid_cmd(
|
||||
out: DrmOutput,
|
||||
grid,
|
||||
ingest_w: int,
|
||||
ingest_h: int,
|
||||
fps: int,
|
||||
tile_fds: list[int] | None = None,
|
||||
chrome_fd: int | None = None,
|
||||
chrome_path: str | None = None,
|
||||
queue_buffers: int = 1,
|
||||
tile_h264_paths: list[str] | None = None,
|
||||
) -> list[str]:
|
||||
"""Chrome + N tiles -> Arc vacompositor -> kmssink.
|
||||
|
||||
Tiles are I420 fdsrcs (default) or Annex-B H.264 files/fifos decoded
|
||||
with varenderD129h264dec. Chrome is still a frozen BGRA overlay.
|
||||
"""
|
||||
q = max(1, int(queue_buffers))
|
||||
fps = max(1, int(fps))
|
||||
if tile_h264_paths:
|
||||
n = len(tile_h264_paths)
|
||||
elif tile_fds:
|
||||
n = len(tile_fds)
|
||||
else:
|
||||
raise ValueError("tile_fds or tile_h264_paths required")
|
||||
cmd: list[str] = [GST_LAUNCH, "-e", "varenderD129compositor", "name=c"]
|
||||
cmd += [
|
||||
"sink_0::zorder=0", "sink_0::xpos=0", "sink_0::ypos=0",
|
||||
f"sink_0::width={grid.width}", f"sink_0::height={grid.height}",
|
||||
]
|
||||
for i in range(n):
|
||||
x, y, w, h = grid.inner_rect(i)
|
||||
s = i + 1
|
||||
cmd += [
|
||||
f"sink_{s}::zorder=1",
|
||||
f"sink_{s}::xpos={x}",
|
||||
f"sink_{s}::ypos={y}",
|
||||
f"sink_{s}::width={w}",
|
||||
f"sink_{s}::height={h}",
|
||||
]
|
||||
cmd += [
|
||||
"!", f"video/x-raw,width={grid.width},height={grid.height},framerate={fps}/1",
|
||||
"!", "varenderD129postproc",
|
||||
"!", f"video/x-raw,format=NV12,width={grid.width},height={grid.height}",
|
||||
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
|
||||
"!", *_kms_props(out),
|
||||
]
|
||||
chrome_block = grid.width * grid.height * 4
|
||||
if chrome_path:
|
||||
cmd += [
|
||||
"filesrc", f"location={chrome_path}", f"blocksize={chrome_block}",
|
||||
"!", "videoparse",
|
||||
f"width={grid.width}", f"height={grid.height}",
|
||||
"format=bgra", f"framerate={fps}/1",
|
||||
"!", "imagefreeze", "is-live=true",
|
||||
"!", "varenderD129postproc",
|
||||
"!", "c.sink_0",
|
||||
]
|
||||
else:
|
||||
if chrome_fd is None:
|
||||
raise ValueError("chrome_fd or chrome_path required")
|
||||
cmd += [
|
||||
"fdsrc", f"fd={chrome_fd}", f"blocksize={chrome_block}",
|
||||
"!", "videoparse",
|
||||
f"width={grid.width}", f"height={grid.height}",
|
||||
"format=bgra", f"framerate={fps}/1",
|
||||
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
|
||||
"!", "varenderD129postproc",
|
||||
"!", "c.sink_0",
|
||||
]
|
||||
if tile_h264_paths:
|
||||
for i, path in enumerate(tile_h264_paths):
|
||||
s = i + 1
|
||||
cmd += [
|
||||
"filesrc", f"location={path}",
|
||||
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
|
||||
"!", "h264parse",
|
||||
"!", "varenderD129h264dec",
|
||||
"!", f"c.sink_{s}",
|
||||
]
|
||||
return cmd
|
||||
tile_block = ingest_w * ingest_h * 3 // 2
|
||||
assert tile_fds is not None
|
||||
for i, fd in enumerate(tile_fds):
|
||||
s = i + 1
|
||||
cmd += [
|
||||
"fdsrc", f"fd={fd}", f"blocksize={tile_block}",
|
||||
"!", "videoparse",
|
||||
f"width={ingest_w}", f"height={ingest_h}",
|
||||
"format=i420", f"framerate={fps}/1",
|
||||
"!", "queue", f"max-size-buffers={q}", "leaky=downstream",
|
||||
"!", "varenderD129postproc", "add-borders=true",
|
||||
"!", f"c.sink_{s}",
|
||||
]
|
||||
return cmd
|
||||
|
||||
class KmsPipeline:
|
||||
"""One gst-launch kmssink process, fed raw BGRA on stdin (or self-running testsrc)."""
|
||||
|
||||
@@ -64,6 +175,8 @@ class KmsPipeline:
|
||||
self.width = 0
|
||||
self.height = 0
|
||||
self.mode = "idle"
|
||||
self.queue_buffers = 2
|
||||
self.pixel_format = "bgra"
|
||||
|
||||
@property
|
||||
def alive(self) -> bool:
|
||||
@@ -103,17 +216,27 @@ class KmsPipeline:
|
||||
)
|
||||
self.mode = "pattern"
|
||||
|
||||
def start_appsrc(self, width: int, height: int, fps: int = 30) -> None:
|
||||
if self.alive and self.mode == "appsrc" and self.width == width and self.height == height:
|
||||
def start_appsrc(self, width: int, height: int, fps: int = 30,
|
||||
queue_buffers: int = 2, pixel_format: str = "bgra") -> None:
|
||||
queue_buffers = max(1, int(queue_buffers))
|
||||
fmt = (pixel_format or "bgra").strip().lower()
|
||||
if fmt in ("i420", "yuv420p"):
|
||||
fmt = "i420"
|
||||
else:
|
||||
fmt = "bgra"
|
||||
if (self.alive and self.mode == "appsrc" and self.width == width
|
||||
and self.height == height and self.queue_buffers == queue_buffers
|
||||
and self.pixel_format == fmt):
|
||||
return
|
||||
self.stop()
|
||||
if not self.out.connected:
|
||||
log.warning("%s: %s not connected, cannot start kmssink", self.identity, self.out.name)
|
||||
return
|
||||
cmd = appsrc_cmd(self.out, width, height, fps)
|
||||
cmd = appsrc_cmd(self.out, width, height, fps, queue_buffers=queue_buffers,
|
||||
pixel_format=fmt)
|
||||
log.info(
|
||||
"%s: gst appsrc %dx%d -> %s (connector %s %s)",
|
||||
self.identity, width, height, self.out.name, self.out.connector_id, self.out.driver,
|
||||
"%s: gst appsrc %dx%d %s -> %s (connector %s %s)",
|
||||
self.identity, width, height, fmt, self.out.name, self.out.connector_id, self.out.driver,
|
||||
)
|
||||
self.proc = subprocess.Popen(
|
||||
cmd,
|
||||
@@ -126,6 +249,8 @@ class KmsPipeline:
|
||||
self.width = width
|
||||
self.height = height
|
||||
self.mode = "appsrc"
|
||||
self.queue_buffers = queue_buffers
|
||||
self.pixel_format = fmt
|
||||
|
||||
def push(self, frame: bytes) -> bool:
|
||||
if not self.alive or self.proc is None or self.proc.stdin is None:
|
||||
@@ -152,10 +277,174 @@ class KmsPipeline:
|
||||
return chunk.decode("utf-8", "replace")
|
||||
|
||||
|
||||
class VaGridPipeline:
|
||||
"""Arc vacompositor mosaic: chrome + one BGRA pipe per tile."""
|
||||
|
||||
def __init__(self, out: DrmOutput):
|
||||
self.out = out
|
||||
self.proc: Optional[subprocess.Popen] = None
|
||||
self._writes: list[int] = []
|
||||
self._chrome_path: Optional[str] = None
|
||||
self._chrome_w: Optional[int] = None
|
||||
self.ingest_w = 0
|
||||
self.ingest_h = 0
|
||||
self.n_tiles = 0
|
||||
self.codec = "i420"
|
||||
self._fifo_holds: list[int] = []
|
||||
|
||||
@property
|
||||
def alive(self) -> bool:
|
||||
return self.proc is not None and self.proc.poll() is None
|
||||
|
||||
def start(self, grid, ingest_w: int, ingest_h: int, fps: int = 15,
|
||||
queue_buffers: int = 1, h264_paths: list[str] | None = None) -> None:
|
||||
self.stop()
|
||||
if not self.out.connected:
|
||||
log.warning("va-grid: %s not connected", self.out.name)
|
||||
return
|
||||
n = len(grid.identities)
|
||||
import tempfile
|
||||
from display_grid import placeholder, display_name, bgra_to_i420
|
||||
chrome_path = tempfile.NamedTemporaryFile(
|
||||
prefix="livekit-grid-chrome-", suffix=".bgra", delete=False).name
|
||||
with open(chrome_path, "wb") as f:
|
||||
f.write(grid.render())
|
||||
self._chrome_path = chrome_path
|
||||
self.codec = "h264" if h264_paths else "i420"
|
||||
reads: list[int] = []
|
||||
writes: list[int] = []
|
||||
if h264_paths:
|
||||
from h264_fifo import hold_fifo
|
||||
self._fifo_holds = [hold_fifo(p) for p in h264_paths]
|
||||
cmd = va_grid_cmd(
|
||||
self.out, grid, ingest_w, ingest_h, fps,
|
||||
tile_h264_paths=h264_paths, chrome_path=chrome_path,
|
||||
queue_buffers=queue_buffers)
|
||||
pass_fds: tuple[int, ...] = ()
|
||||
else:
|
||||
pipe_sz = 1048576
|
||||
for _ in range(n):
|
||||
r, w = os.pipe()
|
||||
os.set_inheritable(r, True)
|
||||
try:
|
||||
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, pipe_sz)
|
||||
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, pipe_sz)
|
||||
except OSError:
|
||||
pass
|
||||
reads.append(r)
|
||||
writes.append(w)
|
||||
cmd = va_grid_cmd(
|
||||
self.out, grid, ingest_w, ingest_h, fps,
|
||||
tile_fds=reads, chrome_path=chrome_path,
|
||||
queue_buffers=queue_buffers)
|
||||
pass_fds = tuple(reads)
|
||||
log.info(
|
||||
"va-grid: Arc compositor %dx%d ingest %dx%d %d tiles codec=%s -> %s id=%s",
|
||||
grid.width, grid.height, ingest_w, ingest_h, n, self.codec,
|
||||
self.out.name, self.out.connector_id)
|
||||
self.proc = subprocess.Popen(
|
||||
cmd,
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=None,
|
||||
pass_fds=pass_fds,
|
||||
env=_gst_env(),
|
||||
close_fds=True,
|
||||
)
|
||||
for r in reads:
|
||||
try:
|
||||
os.close(r)
|
||||
except OSError:
|
||||
pass
|
||||
self._writes = writes
|
||||
self.ingest_w = ingest_w
|
||||
self.ingest_h = ingest_h
|
||||
self.n_tiles = n
|
||||
if self.codec == "i420":
|
||||
for i, ident in enumerate(grid.identities):
|
||||
ph = placeholder(ingest_w, ingest_h, [display_name(ident)])
|
||||
os.set_blocking(writes[i], True)
|
||||
self._write_fd(writes[i], bgra_to_i420(ph))
|
||||
os.set_blocking(writes[i], False)
|
||||
|
||||
def push_tile(self, index: int, frame: bytes | memoryview) -> bool:
|
||||
if not self.alive or index < 0 or index >= self.n_tiles:
|
||||
return False
|
||||
return self._write_fd(self._writes[index], frame)
|
||||
|
||||
def _write_fd(self, fd: Optional[int], data: bytes | memoryview) -> bool:
|
||||
if fd is None:
|
||||
return False
|
||||
view = memoryview(data)
|
||||
off = 0
|
||||
try:
|
||||
while off < len(view):
|
||||
try:
|
||||
n = os.write(fd, view[off:])
|
||||
if n == 0:
|
||||
return False
|
||||
off += n
|
||||
except BlockingIOError:
|
||||
if off == 0:
|
||||
return True
|
||||
continue
|
||||
return True
|
||||
except BrokenPipeError:
|
||||
log.warning("va-grid: pipe closed (rc=%s)",
|
||||
None if self.proc is None else self.proc.poll())
|
||||
return False
|
||||
|
||||
def gst_errors(self) -> str:
|
||||
if self.proc is None or self.proc.stderr is None:
|
||||
return ""
|
||||
try:
|
||||
os.set_blocking(self.proc.stderr.fileno(), False)
|
||||
except OSError:
|
||||
pass
|
||||
try:
|
||||
chunk = self.proc.stderr.read() or b""
|
||||
except OSError:
|
||||
return ""
|
||||
return chunk.decode("utf-8", "replace")
|
||||
|
||||
def stop(self) -> None:
|
||||
for w in self._writes:
|
||||
try:
|
||||
os.close(w)
|
||||
except OSError:
|
||||
pass
|
||||
self._writes = []
|
||||
for fd in getattr(self, "_fifo_holds", []):
|
||||
try:
|
||||
os.close(fd)
|
||||
except OSError:
|
||||
pass
|
||||
self._fifo_holds = []
|
||||
self._chrome_w = None
|
||||
if self._chrome_path:
|
||||
try:
|
||||
os.unlink(self._chrome_path)
|
||||
except OSError:
|
||||
pass
|
||||
self._chrome_path = None
|
||||
if self.proc is None:
|
||||
return
|
||||
if self.proc.poll() is None:
|
||||
self.proc.terminate()
|
||||
try:
|
||||
self.proc.wait(timeout=3)
|
||||
except subprocess.TimeoutExpired:
|
||||
self.proc.kill()
|
||||
self.proc.wait(timeout=2)
|
||||
self.proc = None
|
||||
|
||||
|
||||
def _gst_env() -> dict:
|
||||
env = os.environ.copy()
|
||||
# Never try X11/Wayland; kmssink is the point.
|
||||
env.pop("DISPLAY", None)
|
||||
env.pop("WAYLAND_DISPLAY", None)
|
||||
env.setdefault("GST_GL_PLATFORM", "egl")
|
||||
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
|
||||
env.setdefault("LIBVA_DRM_DEVICE", "/dev/dri/renderD129")
|
||||
return env
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
"""Named-pipe helpers for the capture <-> display H.264 sidecar.
|
||||
|
||||
Linux fifos deadlock if one side opens O_RDONLY and the other O_WRONLY
|
||||
before either has a partner. Opening O_RDWR in the parent first makes
|
||||
filesink/filesrc open() return immediately.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import stat
|
||||
|
||||
|
||||
def hold_fifo(path: str, mode: int = 0o644) -> int:
|
||||
"""Create `path` as a fifo if needed and open it O_RDWR | O_NONBLOCK.
|
||||
|
||||
Do not unlink an existing fifo: that splits the inode from any process
|
||||
already blocked in open() and recreates the deadlock.
|
||||
"""
|
||||
directory = os.path.dirname(path)
|
||||
if directory:
|
||||
os.makedirs(directory, exist_ok=True)
|
||||
try:
|
||||
st = os.stat(path)
|
||||
if not stat.S_ISFIFO(st.st_mode):
|
||||
os.unlink(path)
|
||||
os.mkfifo(path, mode)
|
||||
except FileNotFoundError:
|
||||
os.mkfifo(path, mode)
|
||||
return os.open(path, os.O_RDWR | os.O_NONBLOCK)
|
||||
+287
@@ -0,0 +1,287 @@
|
||||
"""Cheap MoveNet Lightning hand-raise detector for the KMS wall.
|
||||
|
||||
OpenVINO CPU, FP32 single-pose, ~4 Hz per tile. Does not touch capture/encode.
|
||||
Arc/iGPU stay unused (kmssink + H.264).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from collections import deque
|
||||
from pathlib import Path
|
||||
from typing import Callable, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
log = logging.getLogger("cameras.display")
|
||||
|
||||
MODEL_PATH = Path(__file__).resolve().parent / "models" / "movenet_singlepose_lightning.onnx"
|
||||
MODELS_DIR = Path(__file__).resolve().parent / "models"
|
||||
INPUT_SIZE = 192
|
||||
MOVENET = {
|
||||
"lightning": ("movenet_singlepose_lightning.onnx", 192),
|
||||
"thunder": ("movenet_singlepose_thunder.onnx", 256),
|
||||
}
|
||||
|
||||
|
||||
def resolve_movenet(
|
||||
name: str = "thunder",
|
||||
models_dir: Path | None = None,
|
||||
) -> tuple[Path, int, str]:
|
||||
"""lightning=192, thunder=256. Same Xenova FP32 ONNX family as production Lightning."""
|
||||
key = (name or "thunder").strip().lower()
|
||||
if key not in MOVENET:
|
||||
raise ValueError(f"DISPLAY_HAND_RAISE_MODEL must be lightning|thunder, got {name!r}")
|
||||
fname, size = MOVENET[key]
|
||||
root = Path(models_dir) if models_dir is not None else MODELS_DIR
|
||||
return root / fname, int(size), key
|
||||
|
||||
# MoveNet COCO-17, output is (y, x, score) in [0, 1], y grows downward.
|
||||
NOSE = 0
|
||||
L_SHOULDER, R_SHOULDER = 5, 6
|
||||
L_ELBOW, R_ELBOW = 7, 8
|
||||
L_WRIST, R_WRIST = 9, 10
|
||||
MIN_SCORE = 0.20
|
||||
# Wrist must sit this far above the shoulder (normalized image y).
|
||||
LIFT = 0.05
|
||||
FACE_RADIUS2 = 0.12 * 0.12
|
||||
|
||||
|
||||
def letterbox_rgb(rgb: np.ndarray, size: int = INPUT_SIZE) -> np.ndarray:
|
||||
"""Pad to square so MoveNet is not stretched (16:9 C920 tiles)."""
|
||||
import cv2
|
||||
|
||||
size = int(size)
|
||||
h, w = rgb.shape[:2]
|
||||
if h < 1 or w < 1:
|
||||
return np.zeros((size, size, 3), dtype=np.uint8)
|
||||
scale = size / float(max(h, w))
|
||||
nw = max(1, int(round(w * scale)))
|
||||
nh = max(1, int(round(h * scale)))
|
||||
resized = cv2.resize(rgb, (nw, nh), interpolation=cv2.INTER_AREA)
|
||||
out = np.zeros((size, size, 3), dtype=np.uint8)
|
||||
y0 = (size - nh) // 2
|
||||
x0 = (size - nw) // 2
|
||||
out[y0:y0 + nh, x0:x0 + nw] = resized
|
||||
return out
|
||||
|
||||
|
||||
def wrist_is_raised(kpts: np.ndarray) -> bool:
|
||||
"""True if either wrist is clearly up — above the shoulder, or above the head."""
|
||||
if kpts is None or kpts.shape != (17, 3):
|
||||
return False
|
||||
k = kpts
|
||||
if k[NOSE, 2] < MIN_SCORE and k[L_SHOULDER, 2] < MIN_SCORE and k[R_SHOULDER, 2] < MIN_SCORE:
|
||||
return False
|
||||
return (
|
||||
_arm_up(k, L_WRIST, L_ELBOW, L_SHOULDER)
|
||||
or _arm_up(k, R_WRIST, R_ELBOW, R_SHOULDER)
|
||||
or _wrist_above_head(k)
|
||||
)
|
||||
|
||||
|
||||
def _near_face(k: np.ndarray, wr: np.ndarray) -> bool:
|
||||
if k[NOSE, 2] < MIN_SCORE:
|
||||
return False
|
||||
dy = wr[0] - k[NOSE, 0]
|
||||
dx = wr[1] - k[NOSE, 1]
|
||||
return dx * dx + dy * dy < FACE_RADIUS2
|
||||
|
||||
|
||||
def _wrist_above_head(k: np.ndarray) -> bool:
|
||||
if k[NOSE, 2] < MIN_SCORE:
|
||||
return False
|
||||
for wi in (L_WRIST, R_WRIST):
|
||||
wr = k[wi]
|
||||
if wr[2] < MIN_SCORE:
|
||||
continue
|
||||
if wr[0] >= k[NOSE, 0] - LIFT:
|
||||
continue
|
||||
if _near_face(k, wr):
|
||||
continue
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _arm_up(k: np.ndarray, wi: int, ei: int, si: int) -> bool:
|
||||
wr, el, sh = k[wi], k[ei], k[si]
|
||||
if wr[2] < MIN_SCORE or sh[2] < MIN_SCORE:
|
||||
return False
|
||||
if wr[0] >= sh[0] - LIFT:
|
||||
return False
|
||||
if el[2] >= MIN_SCORE and wr[0] >= el[0]:
|
||||
return False
|
||||
if _near_face(k, wr):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
class RaiseLatch:
|
||||
"""Vote `hits` in a sliding `window`; stay lit until `hold_s` without a hit."""
|
||||
|
||||
def __init__(self, hits: int = 3, hold_s: float = 1.0, window: int = 5) -> None:
|
||||
self.hits = max(1, int(hits))
|
||||
self.hold_s = max(0.0, float(hold_s))
|
||||
self.window = max(self.hits, int(window))
|
||||
self._buf: deque[bool] = deque(maxlen=self.window)
|
||||
self.raised = False
|
||||
self.last_true = 0.0
|
||||
|
||||
def update(self, pred: bool, now: float) -> bool:
|
||||
self._buf.append(bool(pred))
|
||||
if pred:
|
||||
self.last_true = now
|
||||
if sum(self._buf) >= self.hits:
|
||||
self.raised = True
|
||||
elif self.raised and (now - self.last_true) >= self.hold_s:
|
||||
self.raised = False
|
||||
return self.raised
|
||||
|
||||
|
||||
class HandRaiseMonitor:
|
||||
"""Background 2 Hz pose on the latest I420 tile per identity."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
enabled: bool = True,
|
||||
hz: float = 4.0,
|
||||
hold_s: float = 1.0,
|
||||
on_change: Optional[Callable[[str, bool], None]] = None,
|
||||
model_path: Optional[Path] = None,
|
||||
model: str = "thunder",
|
||||
input_size: Optional[int] = None,
|
||||
) -> None:
|
||||
self.on_change = on_change
|
||||
self.hz = min(10.0, max(0.5, float(hz)))
|
||||
self.hold_s = float(hold_s)
|
||||
if model_path is not None:
|
||||
self._path = Path(model_path)
|
||||
self._variant = (model or "custom").strip().lower()
|
||||
self._input_size = int(input_size or (256 if "thunder" in self._path.name else 192))
|
||||
else:
|
||||
self._path, self._input_size, self._variant = resolve_movenet(model)
|
||||
if input_size is not None:
|
||||
self._input_size = int(input_size)
|
||||
self._lock = threading.Lock()
|
||||
self._latest: dict[str, tuple[bytes, int, int]] = {}
|
||||
self._latch: dict[str, RaiseLatch] = {}
|
||||
self._raised: set[str] = set()
|
||||
self._stop = threading.Event()
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
self._request = None
|
||||
self._ov_input = None
|
||||
self.enabled = bool(enabled) and self._load()
|
||||
|
||||
def _load(self) -> bool:
|
||||
if not self._path.is_file():
|
||||
log.warning("hand-raise model missing: %s", self._path)
|
||||
return False
|
||||
try:
|
||||
import openvino as ov
|
||||
|
||||
core = ov.Core()
|
||||
model = core.read_model(str(self._path))
|
||||
compiled = core.compile_model(
|
||||
model,
|
||||
"CPU",
|
||||
{"INFERENCE_NUM_THREADS": 1, "NUM_STREAMS": 1},
|
||||
)
|
||||
self._request = compiled.create_infer_request()
|
||||
self._ov_input = compiled.input(0)
|
||||
log.info(
|
||||
"hand-raise MoveNet %s FP32 OpenVINO CPU hz=%.1f hold=%.1fs input=%d model=%s",
|
||||
self._variant, self.hz, self.hold_s, self._input_size, self._path.name,
|
||||
)
|
||||
return True
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.warning("hand-raise model load failed: %s", exc)
|
||||
self._request = None
|
||||
return False
|
||||
|
||||
def start(self) -> None:
|
||||
if not self.enabled or self._thread is not None:
|
||||
return
|
||||
self._thread = threading.Thread(
|
||||
target=self._run, name="hand-raise", daemon=True)
|
||||
self._thread.start()
|
||||
|
||||
def stop(self) -> None:
|
||||
self._stop.set()
|
||||
t = self._thread
|
||||
if t is not None and t.is_alive():
|
||||
t.join(timeout=2.0)
|
||||
self._thread = None
|
||||
|
||||
def offer(self, identity: str, i420: bytes | memoryview, width: int, height: int) -> None:
|
||||
if not self.enabled or width < 16 or height < 16:
|
||||
return
|
||||
blob = i420 if isinstance(i420, (bytes, bytearray)) else bytes(i420)
|
||||
with self._lock:
|
||||
self._latest[identity] = (blob, int(width), int(height))
|
||||
|
||||
def forget(self, identity: str) -> None:
|
||||
with self._lock:
|
||||
self._latest.pop(identity, None)
|
||||
self._latch.pop(identity, None)
|
||||
was = identity in self._raised
|
||||
self._raised.discard(identity)
|
||||
if was and self.on_change is not None:
|
||||
try:
|
||||
self.on_change(identity, False)
|
||||
except Exception: # noqa: BLE001
|
||||
log.exception("hand-raise on_change(%s, False)", identity)
|
||||
|
||||
def raised(self) -> set[str]:
|
||||
with self._lock:
|
||||
return set(self._raised)
|
||||
|
||||
def _run(self) -> None:
|
||||
import cv2
|
||||
|
||||
period = 1.0 / self.hz
|
||||
while not self._stop.wait(period):
|
||||
with self._lock:
|
||||
snapshot = list(self._latest.items())
|
||||
now = time.monotonic()
|
||||
for ident, (blob, w, h) in snapshot:
|
||||
try:
|
||||
pred = self._infer(cv2, blob, w, h)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.warning("hand-raise infer %s: %s", ident, exc)
|
||||
pred = False
|
||||
with self._lock:
|
||||
latch = self._latch.get(ident)
|
||||
if latch is None:
|
||||
latch = RaiseLatch(hits=3, hold_s=self.hold_s, window=5)
|
||||
self._latch[ident] = latch
|
||||
was = latch.raised
|
||||
now_raised = latch.update(pred, now)
|
||||
if now_raised:
|
||||
self._raised.add(ident)
|
||||
else:
|
||||
self._raised.discard(ident)
|
||||
if now_raised != was:
|
||||
log.info("hand-raise %s %s", ident, "on" if now_raised else "off")
|
||||
if self.on_change is not None:
|
||||
try:
|
||||
self.on_change(ident, now_raised)
|
||||
except Exception: # noqa: BLE001
|
||||
log.exception("hand-raise on_change(%s, %s)", ident, now_raised)
|
||||
|
||||
def infer_rgb(self, rgb: np.ndarray) -> np.ndarray:
|
||||
"""MoveNet (17, 3) y,x,score from an HxWx3 RGB uint8 image."""
|
||||
if self._request is None:
|
||||
return np.zeros((17, 3), dtype=np.float32)
|
||||
small = letterbox_rgb(rgb, self._input_size)
|
||||
inp = np.expand_dims(np.ascontiguousarray(small, dtype=np.int32), 0)
|
||||
result = self._request.infer({self._ov_input: inp})
|
||||
return np.asarray(next(iter(result.values())), dtype=np.float32).reshape(17, 3)
|
||||
|
||||
def _infer(self, cv2, blob: bytes, w: int, h: int) -> bool:
|
||||
need = w * h * 3 // 2
|
||||
if self._request is None or len(blob) < need:
|
||||
return False
|
||||
yuv = np.frombuffer(blob, dtype=np.uint8, count=need).reshape((h * 3 // 2, w))
|
||||
rgb = cv2.cvtColor(yuv, cv2.COLOR_YUV2RGB_I420)
|
||||
return wrist_is_raised(self.infer_rgb(rgb))
|
||||
+121
@@ -0,0 +1,121 @@
|
||||
"""Int16 PCM hop rings in mmap (single writer, one reader).
|
||||
|
||||
Unlike video latest-frame slots, audio must not skip hops. The ring keeps
|
||||
a few hops; a slow reader drops the oldest.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import mmap
|
||||
import os
|
||||
import struct
|
||||
from typing import Optional
|
||||
|
||||
MAGIC = b"LKHP"
|
||||
VER = 1
|
||||
HEADER = 64
|
||||
_HDR = struct.Struct("<4sIIIIQ") # magic, ver, hop_bytes, nslots, seq, _pad
|
||||
DEFAULT_SLOTS = 8
|
||||
|
||||
|
||||
def hop_bytes(rate: int, channels: int, hop_s: float) -> int:
|
||||
n = int(round(float(hop_s) * int(rate)))
|
||||
return n * int(channels) * 2
|
||||
|
||||
|
||||
def _mmap(fd: int, size: int, writable: bool) -> mmap.mmap:
|
||||
prot = mmap.PROT_READ | (mmap.PROT_WRITE if writable else 0)
|
||||
return mmap.mmap(fd, size, flags=mmap.MAP_SHARED, prot=prot)
|
||||
|
||||
|
||||
class HopWriter:
|
||||
def __init__(self, path: str, hop_nbytes: int, nslots: int = DEFAULT_SLOTS):
|
||||
self.path = path
|
||||
self.nbytes = int(hop_nbytes)
|
||||
self.nslots = int(nslots)
|
||||
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
||||
size = HEADER + self.nslots * self.nbytes
|
||||
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
|
||||
try:
|
||||
os.ftruncate(fd, size)
|
||||
self._map = _mmap(fd, size, True)
|
||||
finally:
|
||||
os.close(fd)
|
||||
self._seq = 0
|
||||
_HDR.pack_into(self._map, 0, MAGIC, VER, self.nbytes, self.nslots, 0, 0)
|
||||
|
||||
def write(self, payload: bytes) -> int:
|
||||
if len(payload) < self.nbytes:
|
||||
return self._seq
|
||||
nxt = self._seq + 1
|
||||
slot = nxt % self.nslots
|
||||
off = HEADER + slot * self.nbytes
|
||||
self._map[off:off + self.nbytes] = payload[:self.nbytes]
|
||||
self._seq = nxt
|
||||
_HDR.pack_into(
|
||||
self._map, 0, MAGIC, VER, self.nbytes, self.nslots, self._seq, 0)
|
||||
return self._seq
|
||||
|
||||
def close(self) -> None:
|
||||
self._map.close()
|
||||
|
||||
|
||||
class HopReader:
|
||||
def __init__(self, path: str):
|
||||
self.path = path
|
||||
self._map: Optional[mmap.mmap] = None
|
||||
self.nbytes = 0
|
||||
self.nslots = 0
|
||||
self._last = 0
|
||||
|
||||
def _open(self) -> bool:
|
||||
if self._map is not None:
|
||||
return True
|
||||
try:
|
||||
fd = os.open(self.path, os.O_RDONLY)
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
try:
|
||||
st = os.fstat(fd)
|
||||
if st.st_size < HEADER:
|
||||
return False
|
||||
buf = _mmap(fd, st.st_size, False)
|
||||
finally:
|
||||
os.close(fd)
|
||||
magic, ver, nbytes, nslots, seq, _p = _HDR.unpack_from(buf, 0)
|
||||
if magic != MAGIC or ver != VER or nbytes <= 0 or nslots < 2:
|
||||
buf.close()
|
||||
return False
|
||||
self._map = buf
|
||||
self.nbytes, self.nslots = nbytes, nslots
|
||||
self._last = 0
|
||||
return True
|
||||
|
||||
def _close(self) -> None:
|
||||
if self._map is None:
|
||||
return
|
||||
try:
|
||||
self._map.close()
|
||||
finally:
|
||||
self._map = None
|
||||
self.nbytes = 0
|
||||
self.nslots = 0
|
||||
|
||||
def read(self) -> Optional[bytes]:
|
||||
if not self._open() or self._map is None:
|
||||
return None
|
||||
seq = struct.unpack_from("<I", self._map, 16)[0]
|
||||
if seq < self._last:
|
||||
# Writer restarted (cameras-only): seq went back to 0.
|
||||
self._close()
|
||||
self._last = 0
|
||||
if not self._open() or self._map is None:
|
||||
return None
|
||||
seq = struct.unpack_from("<I", self._map, 16)[0]
|
||||
if seq <= self._last:
|
||||
return None
|
||||
if seq - self._last > self.nslots:
|
||||
self._last = seq - 1
|
||||
self._last += 1
|
||||
slot = self._last % self.nslots
|
||||
off = HEADER + slot * self.nbytes
|
||||
return bytes(self._map[off:off + self.nbytes])
|
||||
@@ -0,0 +1,76 @@
|
||||
"""16-align I420→NV12 for Intel VA (NV12 only). Pad up, never crop."""
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
def coded_dim(n: int) -> int:
|
||||
n = int(n)
|
||||
if n <= 0:
|
||||
return 0
|
||||
return (n + 15) // 16 * 16
|
||||
|
||||
|
||||
def nv12_bytes(width: int, height: int) -> int:
|
||||
return int(width) * int(height) * 3 // 2
|
||||
|
||||
|
||||
def va_visible_height(width: int, height: int) -> int:
|
||||
"""Drop the 8-row VA pad when the remainder is 16:9 (360/1080)."""
|
||||
w, h = int(width), int(height)
|
||||
vis = h - 8
|
||||
if w > 0 and vis > 0 and vis * 16 == w * 9:
|
||||
return vis
|
||||
return h
|
||||
|
||||
|
||||
def drop_va_pad_i420(src, width: int, height: int):
|
||||
"""Crop coded pad off I420. Returns (payload, w, visible_h)."""
|
||||
w, h = int(width), int(height)
|
||||
vis = va_visible_height(w, h)
|
||||
if vis == h:
|
||||
return src, w, h
|
||||
ysz = w * h
|
||||
uvw, uvh = w // 2, h // 2
|
||||
uvsz = uvw * uvh
|
||||
raw = np.frombuffer(src, dtype=np.uint8, count=ysz + 2 * uvsz)
|
||||
y = raw[:ysz].reshape(h, w)[:vis, :]
|
||||
u = raw[ysz:ysz + uvsz].reshape(uvh, uvw)[: vis // 2, :]
|
||||
v = raw[ysz + uvsz:ysz + 2 * uvsz].reshape(uvh, uvw)[: vis // 2, :]
|
||||
out = np.empty(w * vis * 3 // 2, dtype=np.uint8)
|
||||
out[: w * vis] = y.reshape(-1)
|
||||
out[w * vis:w * vis + (w // 2) * (vis // 2)] = u.reshape(-1)
|
||||
out[w * vis + (w // 2) * (vis // 2):] = v.reshape(-1)
|
||||
return out.tobytes(), w, vis
|
||||
|
||||
|
||||
def i420_to_nv12_into(src, width: int, height: int, out: bytearray) -> int:
|
||||
"""Pad to 16-aligned WxH and write NV12 into out. Returns nbytes."""
|
||||
w, h = int(width), int(height)
|
||||
cw, ch = coded_dim(w), coded_dim(h)
|
||||
need = nv12_bytes(cw, ch)
|
||||
if len(out) < need:
|
||||
raise ValueError("NV12 dest too small")
|
||||
ysz = w * h
|
||||
uvw, uvh = w // 2, h // 2
|
||||
uvsz = uvw * uvh
|
||||
raw = np.frombuffer(src, dtype=np.uint8, count=ysz + 2 * uvsz)
|
||||
y = raw[:ysz].reshape(h, w)
|
||||
u = raw[ysz:ysz + uvsz].reshape(uvh, uvw)
|
||||
v = raw[ysz + uvsz:ysz + 2 * uvsz].reshape(uvh, uvw)
|
||||
nv = np.frombuffer(out, dtype=np.uint8, count=need).reshape(ch + ch // 2, cw)
|
||||
nv[:h, :w] = y
|
||||
if ch > h:
|
||||
nv[h:ch, :w] = y[-1]
|
||||
if cw > w:
|
||||
nv[:ch, w:cw] = nv[:ch, w - 1:w]
|
||||
uv = nv[ch:, :].reshape(ch // 2, cw // 2, 2)
|
||||
uh, uw = h // 2, w // 2
|
||||
uv[:uh, :uw, 0] = u
|
||||
uv[:uh, :uw, 1] = v
|
||||
if ch // 2 > uh:
|
||||
uv[uh:, :uw, 0] = u[-1]
|
||||
uv[uh:, :uw, 1] = v[-1]
|
||||
if cw // 2 > uw:
|
||||
uv[:, uw:, :] = uv[:, uw - 1:uw, :]
|
||||
return need
|
||||
+136
@@ -0,0 +1,136 @@
|
||||
"""Pure helpers for env-flagged capture/display latency knobs."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def audio_queue_size_ms(hop_s: float, audio_queue_ms: int | None = None) -> int:
|
||||
"""LiveKit AudioSource queue. Unset AUDIO_QUEUE_MS keeps the old 150 ms floor."""
|
||||
hop_ms = int(round(float(hop_s) * 1000.0))
|
||||
if audio_queue_ms is not None:
|
||||
return max(1, int(audio_queue_ms))
|
||||
return max(150, hop_ms + 50)
|
||||
|
||||
|
||||
def video_hold_seconds(
|
||||
hop_s: float,
|
||||
has_audio: bool,
|
||||
video_hold_s: float | None = None,
|
||||
) -> float:
|
||||
"""Delay published video to match audio hops. VIDEO_HOLD_S=0 disables."""
|
||||
if not has_audio:
|
||||
return 0.0
|
||||
if video_hold_s is not None:
|
||||
return max(0.0, float(video_hold_s))
|
||||
return float(hop_s)
|
||||
|
||||
|
||||
def effective_enhance_mode(
|
||||
mode: str,
|
||||
identity: str,
|
||||
speaker_only: bool,
|
||||
speaker_camera: str,
|
||||
) -> str:
|
||||
"""ENHANCE_SPEAKER_ONLY=1 keeps enhance on DISPLAY_SPEAKER_CAMERA only.
|
||||
|
||||
``speaker_camera=active`` has no single identity, so every camera keeps
|
||||
the configured mode.
|
||||
"""
|
||||
if not speaker_only:
|
||||
return mode
|
||||
target = (speaker_camera or "").strip()
|
||||
if not target or target.lower() == "active":
|
||||
return mode
|
||||
if identity != target:
|
||||
return "off"
|
||||
return mode
|
||||
|
||||
|
||||
def next_video_capture(now: float, due: float, period: float) -> tuple[bool, float]:
|
||||
"""Pace VideoSource.capture_frame.
|
||||
|
||||
When the loop is late, capture once and jump the next due time to
|
||||
``now + period`` so we never catch up by pushing extra frames into
|
||||
libwebrtc's EncoderQueue (that path leaked ~2.6 MiB/s native RSS).
|
||||
"""
|
||||
period = float(period)
|
||||
if period <= 0:
|
||||
period = 1.0 / 30.0
|
||||
if now < due:
|
||||
return False, due
|
||||
return True, now + period
|
||||
|
||||
|
||||
def should_drop_stale_vaapi_frame(
|
||||
now_us: int,
|
||||
capture_us: int,
|
||||
fps: int,
|
||||
max_frames_behind: int = 3,
|
||||
is_keyframe: bool = False,
|
||||
) -> bool:
|
||||
"""True when VAAPI Encode should drop before ToI420 (encoder behind).
|
||||
|
||||
LiveKit's VAAPIH264EncoderWrapper::Encode never dropped; EncoderQueue
|
||||
kept I420 copies (~0.37 extra frames/s/cam → OOM). Keyframes always
|
||||
pass. Missing timestamps are not treated as stale.
|
||||
"""
|
||||
if is_keyframe:
|
||||
return False
|
||||
if int(capture_us) <= 0:
|
||||
return False
|
||||
fps = max(1, int(fps))
|
||||
budget = max(1, int(max_frames_behind))
|
||||
max_delay_us = budget * 1_000_000 // fps
|
||||
return int(now_us) - int(capture_us) > max_delay_us
|
||||
|
||||
|
||||
def should_drop_encoder_queue_depth(queued: int, max_queued: int = 1) -> bool:
|
||||
"""True when VideoTrackSource must not push another frame (latest-wins)."""
|
||||
return int(queued) >= max(1, int(max_queued))
|
||||
|
||||
|
||||
def video_capture_gate_acquire(gate: dict) -> bool:
|
||||
"""Admit a capture_frame if in-flight slots remain (AudioSource-like).
|
||||
|
||||
``gate`` is ``{in_flight, dropped, max_in_flight}``. Caller must
|
||||
``video_capture_gate_release`` when the encoder slot is free (or after
|
||||
one frame period as a stand-in when FFI has no completion callback).
|
||||
"""
|
||||
max_in_flight = max(1, int(gate.get("max_in_flight") or 3))
|
||||
if int(gate.get("in_flight") or 0) >= max_in_flight:
|
||||
gate["dropped"] = int(gate.get("dropped") or 0) + 1
|
||||
return False
|
||||
gate["in_flight"] = int(gate.get("in_flight") or 0) + 1
|
||||
return True
|
||||
|
||||
|
||||
def video_capture_gate_release(gate: dict) -> None:
|
||||
"""Free one in-flight capture slot."""
|
||||
n = int(gate.get("in_flight") or 0)
|
||||
gate["in_flight"] = n - 1 if n > 0 else 0
|
||||
|
||||
|
||||
def should_recycle_encoders(
|
||||
rss_anon_kb: int,
|
||||
limit_kb: int,
|
||||
now: float,
|
||||
last_recycle_at: float,
|
||||
min_interval_s: float = 60.0,
|
||||
) -> bool:
|
||||
"""True when native RSS is over the cap and a recycle is allowed."""
|
||||
if int(limit_kb) <= 0:
|
||||
return False
|
||||
if int(rss_anon_kb) < int(limit_kb):
|
||||
return False
|
||||
if last_recycle_at > 0 and (now - last_recycle_at) < float(min_interval_s):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def video_stream_kwargs(capacity: int = 0, format_name: str = "") -> dict:
|
||||
"""LiveKit VideoStream options. capacity=0 is unbounded (SDK default)."""
|
||||
out: dict = {}
|
||||
if int(capacity) > 0:
|
||||
out["capacity"] = int(capacity)
|
||||
fmt = (format_name or "").strip().lower()
|
||||
if fmt:
|
||||
out["format"] = fmt
|
||||
return out
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+109
@@ -0,0 +1,109 @@
|
||||
"""Latest-access-unit H.264 slots in mmap (single writer, one reader).
|
||||
|
||||
Double-buffered like frame_shm: fill the inactive slot, then publish seq.
|
||||
Unread AUs are overwritten. Oversize payloads are dropped (seq unchanged).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import mmap
|
||||
import os
|
||||
import struct
|
||||
from typing import Optional
|
||||
|
||||
MAGIC = b"LKNA"
|
||||
VER = 1
|
||||
HEADER = 64
|
||||
_HDR = struct.Struct("<4sIIIIQ") # magic, ver, max_payload, seq, nbytes, ts_ns
|
||||
SLOTS = 2
|
||||
AU_MAX = 256 * 1024
|
||||
|
||||
|
||||
def _mmap(fd: int, size: int, writable: bool) -> mmap.mmap:
|
||||
prot = mmap.PROT_READ | (mmap.PROT_WRITE if writable else 0)
|
||||
return mmap.mmap(fd, size, flags=mmap.MAP_SHARED, prot=prot)
|
||||
|
||||
|
||||
class NalWriter:
|
||||
def __init__(self, path: str, max_payload: int = AU_MAX):
|
||||
self.path = path
|
||||
self.max_payload = int(max_payload)
|
||||
if self.max_payload < 1 or self.max_payload > 2 * 1024 * 1024:
|
||||
raise ValueError("max_payload out of range")
|
||||
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
||||
size = HEADER + SLOTS * self.max_payload
|
||||
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
|
||||
try:
|
||||
os.ftruncate(fd, size)
|
||||
self._map = _mmap(fd, size, True)
|
||||
finally:
|
||||
os.close(fd)
|
||||
self._seq = 0
|
||||
_HDR.pack_into(self._map, 0, MAGIC, VER, self.max_payload, 0, 0, 0)
|
||||
|
||||
def write(self, payload: bytes, ts_ns: int = 0) -> int:
|
||||
n = len(payload)
|
||||
if n < 1 or n > self.max_payload:
|
||||
return self._seq
|
||||
nxt = self._seq + 1
|
||||
slot = nxt % SLOTS
|
||||
off = HEADER + slot * self.max_payload
|
||||
self._map[off:off + n] = payload
|
||||
self._seq = nxt
|
||||
_HDR.pack_into(
|
||||
self._map, 0, MAGIC, VER, self.max_payload, self._seq, n, int(ts_ns))
|
||||
return self._seq
|
||||
|
||||
def close(self) -> None:
|
||||
self._map.close()
|
||||
|
||||
|
||||
class NalReader:
|
||||
def __init__(self, path: str):
|
||||
self.path = path
|
||||
self._map: Optional[mmap.mmap] = None
|
||||
self.max_payload = 0
|
||||
self._last = 0
|
||||
|
||||
def _open(self) -> bool:
|
||||
if self._map is not None:
|
||||
return True
|
||||
try:
|
||||
fd = os.open(self.path, os.O_RDONLY)
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
try:
|
||||
st = os.fstat(fd)
|
||||
if st.st_size < HEADER:
|
||||
return False
|
||||
self._map = _mmap(fd, st.st_size, False)
|
||||
finally:
|
||||
os.close(fd)
|
||||
magic, ver, max_payload, seq, nbytes, _ts = _HDR.unpack_from(self._map, 0)
|
||||
if magic != MAGIC or ver != VER or max_payload < 1:
|
||||
self._map.close()
|
||||
self._map = None
|
||||
return False
|
||||
self.max_payload = int(max_payload)
|
||||
return True
|
||||
|
||||
def read(self) -> tuple[int, bytes] | None:
|
||||
if not self._open() or self._map is None:
|
||||
return None
|
||||
magic, ver, max_payload, seq, nbytes, _ts = _HDR.unpack_from(self._map, 0)
|
||||
if magic != MAGIC or seq <= self._last or nbytes < 1:
|
||||
return None
|
||||
if nbytes > max_payload:
|
||||
return None
|
||||
slot = seq % SLOTS
|
||||
off = HEADER + slot * max_payload
|
||||
payload = bytes(self._map[off:off + nbytes])
|
||||
magic2, _v, _m, seq2, nbytes2, _t = _HDR.unpack_from(self._map, 0)
|
||||
if seq2 != seq or nbytes2 != nbytes:
|
||||
return None
|
||||
self._last = seq
|
||||
return seq, payload
|
||||
|
||||
def close(self) -> None:
|
||||
if self._map is not None:
|
||||
self._map.close()
|
||||
self._map = None
|
||||
Executable
BIN
Binary file not shown.
+2
-1
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import fcntl
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
@@ -85,7 +86,7 @@ def compile_cpu(onnx_path: Path, num_threads: int = 1):
|
||||
"CPU",
|
||||
{
|
||||
"INFERENCE_NUM_THREADS": num_threads,
|
||||
"NUM_STREAMS": 1,
|
||||
"NUM_STREAMS": max(1, int(os.environ.get("OV_NUM_STREAMS", "8"))),
|
||||
},
|
||||
)
|
||||
request = compiled.create_infer_request()
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
"""LiveKit participant site tags (attributes/metadata) and display hiding."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Iterable, Mapping, Optional
|
||||
|
||||
from display_grid import is_camera_identity
|
||||
|
||||
|
||||
def parse_tags(spec: str | None) -> list[str]:
|
||||
out: list[str] = []
|
||||
seen: set[str] = set()
|
||||
for part in (spec or "").split(","):
|
||||
tag = part.strip().lower()
|
||||
if tag and tag not in seen:
|
||||
seen.add(tag)
|
||||
out.append(tag)
|
||||
return out
|
||||
|
||||
|
||||
def token_attributes(tags: Iterable[str]) -> dict[str, str]:
|
||||
cleaned = parse_tags(",".join(t for t in tags if t))
|
||||
if not cleaned:
|
||||
return {}
|
||||
return {"tag": cleaned[0], "tags": ",".join(cleaned)}
|
||||
|
||||
|
||||
def tags_from_fields(
|
||||
attributes: Optional[Mapping[str, str]],
|
||||
metadata: Optional[str],
|
||||
) -> set[str]:
|
||||
tags: set[str] = set()
|
||||
attrs = attributes or {}
|
||||
if attrs.get("tag"):
|
||||
tags.update(parse_tags(str(attrs["tag"])))
|
||||
if attrs.get("tags"):
|
||||
tags.update(parse_tags(str(attrs["tags"])))
|
||||
meta = (metadata or "").strip()
|
||||
if not meta:
|
||||
return tags
|
||||
try:
|
||||
data = json.loads(meta)
|
||||
except json.JSONDecodeError:
|
||||
tags.update(parse_tags(meta))
|
||||
return tags
|
||||
if isinstance(data, dict):
|
||||
if data.get("tag"):
|
||||
tags.update(parse_tags(str(data["tag"])))
|
||||
if data.get("tags"):
|
||||
tags.update(parse_tags(str(data["tags"])))
|
||||
elif isinstance(data, str):
|
||||
tags.update(parse_tags(data))
|
||||
return tags
|
||||
|
||||
|
||||
def should_hide_participant(
|
||||
attributes: Optional[Mapping[str, str]],
|
||||
metadata: Optional[str],
|
||||
hide_tags: Iterable[str],
|
||||
) -> bool:
|
||||
hide = {t.strip().lower() for t in hide_tags if t and str(t).strip()}
|
||||
if not hide:
|
||||
return False
|
||||
return bool(tags_from_fields(attributes, metadata) & hide)
|
||||
|
||||
|
||||
def want_display_video(
|
||||
*,
|
||||
identity: str,
|
||||
attributes: Optional[Mapping[str, str]],
|
||||
metadata: Optional[str],
|
||||
hide_tags: Iterable[str],
|
||||
participant_prefix: str,
|
||||
display_identity: str,
|
||||
speaker_identity: str = "",
|
||||
) -> bool:
|
||||
ident = (identity or "").strip()
|
||||
if not ident or ident == display_identity:
|
||||
return False
|
||||
if should_hide_participant(attributes, metadata, hide_tags):
|
||||
return False
|
||||
hide = [t for t in hide_tags if t]
|
||||
if hide:
|
||||
return True
|
||||
if speaker_identity and ident == speaker_identity:
|
||||
return True
|
||||
return is_camera_identity(ident, participant_prefix)
|
||||
@@ -0,0 +1,15 @@
|
||||
# livekit-ffi-backpressure.patch
|
||||
|
||||
Apply on livekit/rust-sdks (webrtc-sys VAAPI + VideoTrackSource).
|
||||
|
||||
1. VideoSource: latest-wins pending slot — at most one OnFrame in flight;
|
||||
a new capture overwrites pending instead of pushing another I420 onto
|
||||
EncoderQueue.
|
||||
2. VAAPI H264 Encode: drop delta frames older than 1 frame period, or
|
||||
Encode calls closer than 8 ms (queue drain), *before* ToI420.
|
||||
Logs `VAAPI encode sample delay_ms=` every 90 frames so the clock
|
||||
domain is visible in the cameras journal.
|
||||
|
||||
Rebuild: `CC=clang CXX=clang++ cargo build --release -p livekit-ffi`
|
||||
Install: copy `target/release/liblivekit_ffi.so` to `native/` and set
|
||||
`LIVEKIT_LIB_PATH` (see systemd drop-in `ffi-backpressure.conf`).
|
||||
@@ -0,0 +1,210 @@
|
||||
diff --git a/livekit-ffi/src/server/video_source.rs b/livekit-ffi/src/server/video_source.rs
|
||||
index ee1410a..b2ff9ca 100644
|
||||
--- a/livekit-ffi/src/server/video_source.rs
|
||||
+++ b/livekit-ffi/src/server/video_source.rs
|
||||
@@ -88,7 +88,9 @@ impl FfiVideoSource {
|
||||
buffer,
|
||||
};
|
||||
|
||||
- source.capture_frame(&frame);
|
||||
+ // false = adapter / in-flight cap dropped the frame (do not
|
||||
+ // treat as an FFI error; the source is still valid).
|
||||
+ let _forwarded = source.capture_frame(&frame);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
diff --git a/webrtc-sys/include/livekit/video_track.h b/webrtc-sys/include/livekit/video_track.h
|
||||
index 4146c45..733d110 100644
|
||||
--- a/webrtc-sys/include/livekit/video_track.h
|
||||
+++ b/webrtc-sys/include/livekit/video_track.h
|
||||
@@ -18,6 +18,7 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <memory>
|
||||
+#include <optional>
|
||||
|
||||
#include "api/media_stream_interface.h"
|
||||
#include "api/video/video_frame.h"
|
||||
@@ -127,6 +128,10 @@ class VideoTrackSource {
|
||||
std::shared_ptr<livekit::EncodedRateControlState> rate_control_state_ =
|
||||
std::make_shared<livekit::EncodedRateControlState>();
|
||||
bool is_screencast_;
|
||||
+ int64_t last_forwarded_us_ = 0;
|
||||
+ // Latest-wins: at most one OnFrame in flight; new captures overwrite.
|
||||
+ bool forwarding_ = false;
|
||||
+ std::optional<webrtc::VideoFrame> pending_frame_;
|
||||
};
|
||||
|
||||
public:
|
||||
diff --git a/webrtc-sys/src/vaapi/h264_encoder_impl.cpp b/webrtc-sys/src/vaapi/h264_encoder_impl.cpp
|
||||
index 5a3b76e..8c7ce11 100644
|
||||
--- a/webrtc-sys/src/vaapi/h264_encoder_impl.cpp
|
||||
+++ b/webrtc-sys/src/vaapi/h264_encoder_impl.cpp
|
||||
@@ -189,18 +189,6 @@ int32_t VAAPIH264EncoderWrapper::Encode(
|
||||
return WEBRTC_VIDEO_CODEC_UNINITIALIZED;
|
||||
}
|
||||
|
||||
- webrtc::scoped_refptr<I420BufferInterface> frame_buffer =
|
||||
- input_frame.video_frame_buffer()->ToI420();
|
||||
- if (!frame_buffer) {
|
||||
- RTC_LOG(LS_ERROR) << "Failed to convert "
|
||||
- << VideoFrameBufferTypeToString(
|
||||
- input_frame.video_frame_buffer()->type())
|
||||
- << " image to I420. Can't encode frame.";
|
||||
- return WEBRTC_VIDEO_CODEC_ENCODER_FAILURE;
|
||||
- }
|
||||
- RTC_CHECK(frame_buffer->type() == VideoFrameBuffer::Type::kI420 ||
|
||||
- frame_buffer->type() == VideoFrameBuffer::Type::kI420A);
|
||||
-
|
||||
bool is_keyframe_needed = false;
|
||||
if (configuration_.key_frame_request && configuration_.sending) {
|
||||
is_keyframe_needed = true;
|
||||
@@ -214,20 +202,67 @@ int32_t VAAPIH264EncoderWrapper::Encode(
|
||||
configuration_.key_frame_request = false;
|
||||
}
|
||||
|
||||
- RTC_DCHECK_EQ(configuration_.width, frame_buffer->width());
|
||||
- RTC_DCHECK_EQ(configuration_.height, frame_buffer->height());
|
||||
-
|
||||
if (!configuration_.sending) {
|
||||
return WEBRTC_VIDEO_CODEC_NO_OUTPUT;
|
||||
}
|
||||
|
||||
if (frame_types != nullptr) {
|
||||
- // Skip frame?
|
||||
if ((*frame_types)[0] == VideoFrameType::kEmptyFrame) {
|
||||
return WEBRTC_VIDEO_CODEC_NO_OUTPUT;
|
||||
}
|
||||
}
|
||||
|
||||
+ // Drop stale/burst delta frames BEFORE ToI420.
|
||||
+ // 1) Age: capture older than 1 frame period (was 3; that never fired).
|
||||
+ // 2) Burst: Encode called closer than 8 ms = EncoderQueue draining.
|
||||
+ // Mirror: livekit-cameras latency.should_drop_stale_vaapi_frame /
|
||||
+ // should_drop_encoder_queue_depth.
|
||||
+ const int fps = std::max(
|
||||
+ 1, static_cast<int>(configuration_.max_frame_rate > 0.0f
|
||||
+ ? configuration_.max_frame_rate
|
||||
+ : codec_.maxFramerate));
|
||||
+ const int64_t now_us = webrtc::TimeMicros();
|
||||
+ const int64_t capture_us = input_frame.timestamp_us();
|
||||
+ const int64_t delay_us = (capture_us > 0) ? (now_us - capture_us) : 0;
|
||||
+ encode_count_++;
|
||||
+ if (encode_count_ % 90 == 1) {
|
||||
+ RTC_LOG(LS_WARNING) << "VAAPI encode sample delay_ms=" << delay_us / 1000
|
||||
+ << " now_us=" << now_us << " capture_us=" << capture_us
|
||||
+ << " fps=" << fps << " n=" << encode_count_;
|
||||
+ }
|
||||
+ if (!send_key_frame) {
|
||||
+ const int64_t max_delay_us = 1000000 / fps; // 1 frame
|
||||
+ const bool stale =
|
||||
+ capture_us > 0 && now_us - capture_us > max_delay_us;
|
||||
+ const bool burst =
|
||||
+ last_encode_us_ != 0 && now_us - last_encode_us_ < 8000;
|
||||
+ if (stale || burst) {
|
||||
+ RTC_LOG(LS_WARNING) << "VAAPI H264 drop "
|
||||
+ << (stale ? "stale" : "burst")
|
||||
+ << " frame delay_ms=" << delay_us / 1000
|
||||
+ << " dt_ms=" << (now_us - last_encode_us_) / 1000
|
||||
+ << " fps=" << fps;
|
||||
+ last_encode_us_ = now_us;
|
||||
+ return WEBRTC_VIDEO_CODEC_OK;
|
||||
+ }
|
||||
+ }
|
||||
+ last_encode_us_ = now_us;
|
||||
+
|
||||
+ webrtc::scoped_refptr<I420BufferInterface> frame_buffer =
|
||||
+ input_frame.video_frame_buffer()->ToI420();
|
||||
+ if (!frame_buffer) {
|
||||
+ RTC_LOG(LS_ERROR) << "Failed to convert "
|
||||
+ << VideoFrameBufferTypeToString(
|
||||
+ input_frame.video_frame_buffer()->type())
|
||||
+ << " image to I420. Can't encode frame.";
|
||||
+ return WEBRTC_VIDEO_CODEC_ENCODER_FAILURE;
|
||||
+ }
|
||||
+ RTC_CHECK(frame_buffer->type() == VideoFrameBuffer::Type::kI420 ||
|
||||
+ frame_buffer->type() == VideoFrameBuffer::Type::kI420A);
|
||||
+
|
||||
+ RTC_DCHECK_EQ(configuration_.width, frame_buffer->width());
|
||||
+ RTC_DCHECK_EQ(configuration_.height, frame_buffer->height());
|
||||
+
|
||||
std::vector<uint8_t> output;
|
||||
encoder_->Encode(VA_FOURCC_I420, frame_buffer->DataY(), frame_buffer->DataU(),
|
||||
frame_buffer->DataV(), send_key_frame, output);
|
||||
@@ -267,6 +302,7 @@ VideoEncoder::EncoderInfo VAAPIH264EncoderWrapper::GetEncoderInfo() const {
|
||||
info.implementation_name = "VAAPI H264 Encoder";
|
||||
info.scaling_settings = VideoEncoder::ScalingSettings::kOff;
|
||||
info.is_hardware_accelerated = true;
|
||||
+ info.has_trusted_rate_controller = false;
|
||||
info.supports_simulcast = false;
|
||||
info.preferred_pixel_formats = {VideoFrameBuffer::Type::kI420};
|
||||
return info;
|
||||
diff --git a/webrtc-sys/src/vaapi/h264_encoder_impl.h b/webrtc-sys/src/vaapi/h264_encoder_impl.h
|
||||
index dc26355..bce1681 100644
|
||||
--- a/webrtc-sys/src/vaapi/h264_encoder_impl.h
|
||||
+++ b/webrtc-sys/src/vaapi/h264_encoder_impl.h
|
||||
@@ -76,6 +76,8 @@ class VAAPIH264EncoderWrapper : public VideoEncoder {
|
||||
const SdpVideoFormat format_;
|
||||
H264Profile profile_ = H264Profile::kProfileConstrainedBaseline;
|
||||
H264Level level_ = H264Level::kLevel1_b;
|
||||
+ int64_t last_encode_us_ = 0;
|
||||
+ int encode_count_ = 0;
|
||||
};
|
||||
|
||||
} // namespace webrtc
|
||||
diff --git a/webrtc-sys/src/video_track.cpp b/webrtc-sys/src/video_track.cpp
|
||||
index 92410b9..89e6791 100644
|
||||
--- a/webrtc-sys/src/video_track.cpp
|
||||
+++ b/webrtc-sys/src/video_track.cpp
|
||||
@@ -165,6 +165,7 @@ VideoResolution VideoTrackSource::InternalSource::video_resolution() const {
|
||||
bool VideoTrackSource::InternalSource::on_captured_frame(
|
||||
const webrtc::VideoFrame& frame,
|
||||
const FrameMetadata& frame_metadata) {
|
||||
+ {
|
||||
webrtc::MutexLock lock(&mutex_);
|
||||
|
||||
int64_t aligned_timestamp_us = timestamp_aligner_.TranslateTimestamp(
|
||||
@@ -235,13 +236,37 @@ bool VideoTrackSource::InternalSource::on_captured_frame(
|
||||
frame_metadata.has_packet_trailer ? frame_metadata.frame_id : 0);
|
||||
}
|
||||
|
||||
- OnFrame(webrtc::VideoFrame::Builder()
|
||||
- .set_video_frame_buffer(buffer)
|
||||
- .set_rotation(rotation)
|
||||
- .set_timestamp_us(aligned_timestamp_us)
|
||||
- .build());
|
||||
+ webrtc::VideoFrame built = webrtc::VideoFrame::Builder()
|
||||
+ .set_video_frame_buffer(buffer)
|
||||
+ .set_rotation(rotation)
|
||||
+ .set_timestamp_us(aligned_timestamp_us)
|
||||
+ .build();
|
||||
+
|
||||
+ // Latest-wins depth cap (max_queued=1). A capture that arrives while
|
||||
+ // OnFrame is in flight overwrites pending_frame_ instead of pushing
|
||||
+ // another I420 onto EncoderQueue.
|
||||
+ // Mirror: livekit-cameras latency.should_drop_encoder_queue_depth.
|
||||
+ pending_frame_ = std::move(built);
|
||||
+ if (forwarding_) {
|
||||
+ return true;
|
||||
+ }
|
||||
+ forwarding_ = true;
|
||||
+ last_forwarded_us_ = aligned_timestamp_us;
|
||||
+ }
|
||||
|
||||
- return true;
|
||||
+ for (;;) {
|
||||
+ std::optional<webrtc::VideoFrame> to_send;
|
||||
+ {
|
||||
+ webrtc::MutexLock lock(&mutex_);
|
||||
+ if (!pending_frame_) {
|
||||
+ forwarding_ = false;
|
||||
+ return true;
|
||||
+ }
|
||||
+ to_send = std::move(pending_frame_);
|
||||
+ pending_frame_.reset();
|
||||
+ }
|
||||
+ OnFrame(*to_send);
|
||||
+ }
|
||||
}
|
||||
|
||||
void VideoTrackSource::InternalSource::set_packet_trailer_handler(
|
||||
+403
@@ -0,0 +1,403 @@
|
||||
"""Person-aware background blur for C920 I420 SHM.
|
||||
|
||||
OpenVINO CPU only. iGPU stays H.264; Arc stays kmssink.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
MODELS_DIR = Path(__file__).resolve().parent / "models"
|
||||
SELFIE_ONNX = MODELS_DIR / "selfie_segmentation.onnx"
|
||||
MOVENET_LIGHTNING = MODELS_DIR / "movenet_singlepose_lightning.onnx"
|
||||
POSE_SIZE = 192
|
||||
YUNET_CANDIDATES = (
|
||||
"face_detection_yunet_2026may.onnx",
|
||||
"face_detection_yunet_2023mar.onnx",
|
||||
)
|
||||
RALLY_ID = "rally"
|
||||
|
||||
|
||||
def i420_source_path(identity: str, raw_dir: str, portrait_dir: str = "") -> str:
|
||||
"""C920s read portrait SHM when enabled; rally always stays on capture SHM."""
|
||||
ident = (identity or "").strip()
|
||||
raw = (raw_dir or "").rstrip("/")
|
||||
por = (portrait_dir or "").rstrip("/")
|
||||
if por and ident and ident != RALLY_ID:
|
||||
return f"{por}/{ident}.i420"
|
||||
return f"{raw}/{ident}.i420"
|
||||
|
||||
|
||||
def i420_nbytes(width: int, height: int) -> int:
|
||||
return int(width) * int(height) * 3 // 2
|
||||
|
||||
|
||||
def i420_to_bgr(buf, width: int, height: int) -> np.ndarray:
|
||||
import cv2
|
||||
|
||||
w, h = int(width), int(height)
|
||||
need = i420_nbytes(w, h)
|
||||
raw = np.frombuffer(buf, dtype=np.uint8, count=need)
|
||||
yuv = raw.reshape((h * 3 // 2, w))
|
||||
return cv2.cvtColor(yuv, cv2.COLOR_YUV2BGR_I420)
|
||||
|
||||
|
||||
def bgr_to_i420(bgr: np.ndarray) -> bytes:
|
||||
import cv2
|
||||
|
||||
yuv = cv2.cvtColor(bgr, cv2.COLOR_BGR2YUV_I420)
|
||||
return np.ascontiguousarray(yuv).tobytes()
|
||||
|
||||
|
||||
def soft_blur_bgr(img: np.ndarray, radius: int = 7) -> np.ndarray:
|
||||
"""Cheap bokeh: half-res box blur, then upsample."""
|
||||
import cv2
|
||||
|
||||
k = max(3, int(radius) | 1)
|
||||
h, w = img.shape[:2]
|
||||
sw, sh = max(2, w // 2), max(2, h // 2)
|
||||
small = cv2.resize(img, (sw, sh), interpolation=cv2.INTER_AREA)
|
||||
small = cv2.blur(small, (k, k))
|
||||
return cv2.resize(small, (w, h), interpolation=cv2.INTER_LINEAR)
|
||||
|
||||
|
||||
def feather_mask(mask: np.ndarray, ksize: int = 7) -> np.ndarray:
|
||||
import cv2
|
||||
|
||||
k = max(3, int(ksize) | 1)
|
||||
m = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
|
||||
return cv2.GaussianBlur(m, (k, k), 0)
|
||||
|
||||
|
||||
def composite_bgr(sharp: np.ndarray, blurred: np.ndarray, mask: np.ndarray) -> np.ndarray:
|
||||
a = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
|
||||
if a.ndim == 2:
|
||||
a = a[:, :, None]
|
||||
out = sharp.astype(np.float32) * a + blurred.astype(np.float32) * (1.0 - a)
|
||||
return np.clip(out, 0, 255).astype(np.uint8)
|
||||
|
||||
|
||||
def person_present(
|
||||
mask: np.ndarray,
|
||||
min_peak: float = 0.35,
|
||||
min_area: float = 0.04,
|
||||
) -> bool:
|
||||
m = np.asarray(mask, dtype=np.float32)
|
||||
if m.size == 0:
|
||||
return False
|
||||
if float(m.max()) < float(min_peak):
|
||||
return False
|
||||
return float((m >= min_peak).mean()) >= float(min_area)
|
||||
|
||||
|
||||
def union_masks(*masks: Optional[np.ndarray]) -> np.ndarray:
|
||||
acc: Optional[np.ndarray] = None
|
||||
for m in masks:
|
||||
if m is None:
|
||||
continue
|
||||
a = np.clip(np.asarray(m, dtype=np.float32), 0.0, 1.0)
|
||||
acc = a if acc is None else np.maximum(acc, a)
|
||||
if acc is None:
|
||||
return np.zeros((1, 1), dtype=np.float32)
|
||||
return acc
|
||||
|
||||
|
||||
# MoveNet COCO-17: nose, eyes, ears, shoulders, elbows, wrists.
|
||||
_POSE_R = {
|
||||
0: 0.14, # nose / face
|
||||
1: 0.08, 2: 0.08, 3: 0.08, 4: 0.08, # eyes, ears
|
||||
5: 0.10, 6: 0.10, # shoulders
|
||||
7: 0.09, 8: 0.09, # elbows
|
||||
9: 0.11, 10: 0.11, # wrists / hands
|
||||
}
|
||||
_POSE_MIN_SCORE = 0.25
|
||||
|
||||
|
||||
def pose_protect_mask(
|
||||
height: int,
|
||||
width: int,
|
||||
kpts: np.ndarray,
|
||||
min_score: float = _POSE_MIN_SCORE,
|
||||
) -> np.ndarray:
|
||||
"""Soft disks on face + arms so raised hands stay sharp."""
|
||||
h, w = int(height), int(width)
|
||||
out = np.zeros((h, w), dtype=np.float32)
|
||||
if kpts is None or kpts.shape != (17, 3) or h < 2 or w < 2:
|
||||
return out
|
||||
scale = float(min(h, w))
|
||||
yy, xx = np.ogrid[:h, :w]
|
||||
for idx, frac in _POSE_R.items():
|
||||
y, x, s = float(kpts[idx, 0]), float(kpts[idx, 1]), float(kpts[idx, 2])
|
||||
if s < min_score:
|
||||
continue
|
||||
cy, cx = y * h, x * w
|
||||
r = max(6.0, frac * scale)
|
||||
dist = ((xx - cx) / r) ** 2 + ((yy - cy) / r) ** 2
|
||||
out = np.maximum(out, np.clip(1.0 - dist, 0.0, 1.0).astype(np.float32))
|
||||
return out
|
||||
|
||||
|
||||
def apply_portrait(
|
||||
bgr: np.ndarray,
|
||||
mask: np.ndarray,
|
||||
radius: int = 7,
|
||||
feather: int = 7,
|
||||
) -> np.ndarray:
|
||||
"""Passthrough never: empty seats are full-frame blur."""
|
||||
if bgr is None:
|
||||
return bgr
|
||||
if mask is None or not person_present(mask):
|
||||
z = np.zeros(bgr.shape[:2], dtype=np.float32)
|
||||
return composite_bgr(bgr, soft_blur_bgr(bgr, radius), z)
|
||||
soft = feather_mask(mask, feather)
|
||||
return composite_bgr(bgr, soft_blur_bgr(bgr, radius), soft)
|
||||
|
||||
|
||||
class MaskHold:
|
||||
"""EMA + hold so a missed infer does not shimmer or drop the person."""
|
||||
|
||||
def __init__(self, hold_s: float = 0.8, ema: float = 0.4) -> None:
|
||||
self.hold_s = max(0.0, float(hold_s))
|
||||
self.ema = min(1.0, max(0.05, float(ema)))
|
||||
self.mask: Optional[np.ndarray] = None
|
||||
self.last_true = 0.0
|
||||
|
||||
def update(
|
||||
self,
|
||||
mask: Optional[np.ndarray],
|
||||
present: bool,
|
||||
now: float,
|
||||
) -> Optional[np.ndarray]:
|
||||
if present and mask is not None:
|
||||
m = np.clip(np.asarray(mask, dtype=np.float32), 0.0, 1.0)
|
||||
if self.mask is None or self.ema >= 1.0:
|
||||
self.mask = m
|
||||
else:
|
||||
a = self.ema
|
||||
self.mask = a * m + (1.0 - a) * self.mask
|
||||
self.last_true = float(now)
|
||||
return self.mask
|
||||
if self.mask is not None and (float(now) - self.last_true) < self.hold_s:
|
||||
return self.mask
|
||||
self.mask = None
|
||||
return None
|
||||
|
||||
|
||||
def apply_i420(
|
||||
payload,
|
||||
width: int,
|
||||
height: int,
|
||||
mask: Optional[np.ndarray],
|
||||
radius: int = 7,
|
||||
feather: int = 7,
|
||||
) -> bytes:
|
||||
"""I420 in/out. Same WxH so encode MCU crop is unchanged."""
|
||||
n = i420_nbytes(width, height)
|
||||
raw = payload[:n]
|
||||
if mask is None or not person_present(mask):
|
||||
z = np.zeros((int(height), int(width)), dtype=np.float32)
|
||||
return _composite_i420(raw, width, height, z, radius, feather=0)
|
||||
return _composite_i420(raw, width, height, mask, radius, feather)
|
||||
|
||||
|
||||
def _composite_i420(
|
||||
payload,
|
||||
width: int,
|
||||
height: int,
|
||||
mask: np.ndarray,
|
||||
radius: int,
|
||||
feather: int,
|
||||
) -> bytes:
|
||||
"""Blur background luma in I420. UV stays; avoids BGR round-trip."""
|
||||
import cv2
|
||||
|
||||
w, h = int(width), int(height)
|
||||
n = i420_nbytes(w, h)
|
||||
arr = np.frombuffer(memoryview(payload)[:n], dtype=np.uint8).copy()
|
||||
y = arr[: w * h].reshape(h, w)
|
||||
a = np.asarray(mask, dtype=np.float32)
|
||||
if a.shape != (h, w):
|
||||
a = cv2.resize(a, (w, h), interpolation=cv2.INTER_LINEAR)
|
||||
if feather and feather > 1:
|
||||
a = feather_mask(a, feather)
|
||||
k = max(3, int(radius) | 1)
|
||||
sw, sh = max(2, w // 2), max(2, h // 2)
|
||||
y_s = cv2.resize(y, (sw, sh), interpolation=cv2.INTER_AREA)
|
||||
y_s = cv2.blur(y_s, (k, k))
|
||||
yb = cv2.resize(y_s, (w, h), interpolation=cv2.INTER_LINEAR)
|
||||
af = a
|
||||
y[:] = (y.astype(np.float32) * af + yb.astype(np.float32) * (1.0 - af)).clip(0, 255).astype(np.uint8)
|
||||
return arr.tobytes()
|
||||
|
||||
|
||||
def letterbox_bgr(bgr: np.ndarray, size: int = 256) -> tuple[np.ndarray, tuple[int, int, int, int]]:
|
||||
"""Pad 16:9 into a square. Returns canvas and (x0, y0, nw, nh) content box."""
|
||||
import cv2
|
||||
|
||||
size = int(size)
|
||||
h, w = bgr.shape[:2]
|
||||
if h < 1 or w < 1:
|
||||
return np.zeros((size, size, 3), dtype=np.uint8), (0, 0, size, size)
|
||||
scale = size / float(max(h, w))
|
||||
nw = max(1, int(round(w * scale)))
|
||||
nh = max(1, int(round(h * scale)))
|
||||
resized = cv2.resize(bgr, (nw, nh), interpolation=cv2.INTER_AREA)
|
||||
canvas = np.zeros((size, size, 3), dtype=np.uint8)
|
||||
y0 = (size - nh) // 2
|
||||
x0 = (size - nw) // 2
|
||||
canvas[y0:y0 + nh, x0:x0 + nw] = resized
|
||||
return canvas, (x0, y0, nw, nh)
|
||||
|
||||
|
||||
def unletterbox_mask(
|
||||
mask: np.ndarray,
|
||||
box: tuple[int, int, int, int],
|
||||
height: int,
|
||||
width: int,
|
||||
) -> np.ndarray:
|
||||
import cv2
|
||||
|
||||
x0, y0, nw, nh = (int(v) for v in box)
|
||||
crop = np.asarray(mask, dtype=np.float32)[y0:y0 + nh, x0:x0 + nw]
|
||||
if crop.size == 0:
|
||||
return np.zeros((int(height), int(width)), dtype=np.float32)
|
||||
return cv2.resize(crop, (int(width), int(height)), interpolation=cv2.INTER_LINEAR)
|
||||
|
||||
|
||||
def resolve_yunet(models_dir: Path | None = None) -> Path:
|
||||
root = Path(models_dir) if models_dir is not None else MODELS_DIR
|
||||
for name in YUNET_CANDIDATES:
|
||||
path = root / name
|
||||
if path.is_file():
|
||||
return path
|
||||
return root / YUNET_CANDIDATES[0]
|
||||
|
||||
|
||||
def _face_soft_mask(bgr: np.ndarray, faces) -> Optional[np.ndarray]:
|
||||
"""Soft ellipse per YuNet face so close-up heads stay unblurred."""
|
||||
if faces is None or len(faces) == 0:
|
||||
return None
|
||||
h, w = bgr.shape[:2]
|
||||
acc = None
|
||||
yy, xx = np.ogrid[:h, :w]
|
||||
for best in faces:
|
||||
x, y, fw, fh = (float(best[0]), float(best[1]), float(best[2]), float(best[3]))
|
||||
if fw < 8 or fh < 8:
|
||||
continue
|
||||
cx = x + fw * 0.5
|
||||
cy = y + fh * 0.42
|
||||
rx = max(fw * 1.15, w * 0.12)
|
||||
ry = max(fh * 1.35, h * 0.16)
|
||||
dist = ((xx - cx) / rx) ** 2 + ((yy - cy) / ry) ** 2
|
||||
m = np.clip(1.0 - dist, 0.0, 1.0).astype(np.float32)
|
||||
acc = m if acc is None else np.maximum(acc, m)
|
||||
return acc
|
||||
|
||||
|
||||
class PersonSeg:
|
||||
"""MediaPipe selfie ONNX on OpenVINO CPU. Never GPU.0/GPU.1."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model_path: Path | None = None,
|
||||
yunet_path: Path | None = None,
|
||||
num_threads: int = 2,
|
||||
) -> None:
|
||||
self.device = "CPU"
|
||||
self._request = None
|
||||
self._ov_input = None
|
||||
self._size = 256
|
||||
self._yunet = None
|
||||
self._cv2 = None
|
||||
self._pose_request = None
|
||||
self._pose_input = None
|
||||
self._pose_size = POSE_SIZE
|
||||
path = Path(model_path) if model_path is not None else SELFIE_ONNX
|
||||
import cv2
|
||||
import openvino as ov
|
||||
|
||||
self._cv2 = cv2
|
||||
try:
|
||||
cv2.setNumThreads(1)
|
||||
except Exception:
|
||||
pass
|
||||
if not path.is_file():
|
||||
raise FileNotFoundError(path)
|
||||
core = ov.Core()
|
||||
model = core.read_model(str(path))
|
||||
compiled = core.compile_model(
|
||||
model,
|
||||
"CPU",
|
||||
{"INFERENCE_NUM_THREADS": max(1, int(num_threads)), "NUM_STREAMS": 1},
|
||||
)
|
||||
self._request = compiled.create_infer_request()
|
||||
self._ov_input = compiled.input(0)
|
||||
ypath = Path(yunet_path) if yunet_path is not None else resolve_yunet()
|
||||
if ypath.is_file():
|
||||
self._yunet = cv2.FaceDetectorYN_create(
|
||||
str(ypath), "", (320, 180), 0.45, 0.3, 5000)
|
||||
if MOVENET_LIGHTNING.is_file():
|
||||
pose = core.read_model(str(MOVENET_LIGHTNING))
|
||||
pose_c = core.compile_model(
|
||||
pose,
|
||||
"CPU",
|
||||
{"INFERENCE_NUM_THREADS": 1, "NUM_STREAMS": 1},
|
||||
)
|
||||
self._pose_request = pose_c.create_infer_request()
|
||||
self._pose_input = pose_c.input(0)
|
||||
|
||||
def infer_bgr(self, bgr: np.ndarray) -> np.ndarray:
|
||||
h, w = bgr.shape[:2]
|
||||
canvas, box = letterbox_bgr(bgr, self._size)
|
||||
rgb = canvas[:, :, ::-1].astype(np.float32) * (1.0 / 255.0)
|
||||
inp = np.transpose(rgb, (2, 0, 1))[None]
|
||||
self._request.infer({self._ov_input: inp})
|
||||
raw = np.asarray(self._request.get_output_tensor(0).data, dtype=np.float32)
|
||||
raw = np.squeeze(raw)
|
||||
mask = unletterbox_mask(raw, box, h, w)
|
||||
face = self._yunet_mask(bgr)
|
||||
if face is not None:
|
||||
mask = union_masks(mask, face)
|
||||
if person_present(mask) or face is not None:
|
||||
pose = self._pose_mask(bgr)
|
||||
if pose is not None and float(pose.max()) > 0:
|
||||
mask = union_masks(mask, pose)
|
||||
return mask
|
||||
|
||||
def _pose_mask(self, bgr: np.ndarray) -> Optional[np.ndarray]:
|
||||
if self._pose_request is None or self._pose_input is None:
|
||||
return None
|
||||
h, w = bgr.shape[:2]
|
||||
canvas, box = letterbox_bgr(bgr, self._pose_size)
|
||||
rgb = np.ascontiguousarray(canvas[:, :, ::-1], dtype=np.int32)
|
||||
inp = rgb[None]
|
||||
self._pose_request.infer({self._pose_input: inp})
|
||||
kpts = np.asarray(
|
||||
self._pose_request.get_output_tensor(0).data, dtype=np.float32
|
||||
).reshape(17, 3)
|
||||
x0, y0, nw, nh = box
|
||||
size = float(self._pose_size)
|
||||
mapped = kpts.copy()
|
||||
mapped[:, 0] = (kpts[:, 0] * size - y0) / max(1.0, float(nh))
|
||||
mapped[:, 1] = (kpts[:, 1] * size - x0) / max(1.0, float(nw))
|
||||
return pose_protect_mask(h, w, mapped)
|
||||
|
||||
def _yunet_mask(self, bgr: np.ndarray) -> Optional[np.ndarray]:
|
||||
if self._yunet is None or self._cv2 is None:
|
||||
return None
|
||||
h, w = bgr.shape[:2]
|
||||
dw, dh = 320, 180
|
||||
small = self._cv2.resize(bgr, (dw, dh), interpolation=self._cv2.INTER_AREA)
|
||||
self._yunet.setInputSize((dw, dh))
|
||||
_ok, faces = self._yunet.detect(small)
|
||||
if faces is None or len(faces) == 0:
|
||||
return None
|
||||
sx = w / float(dw)
|
||||
sy = h / float(dh)
|
||||
scaled = []
|
||||
for f in faces:
|
||||
scaled.append([f[0] * sx, f[1] * sy, f[2] * sx, f[3] * sy])
|
||||
return _face_soft_mask(bgr, scaled)
|
||||
@@ -0,0 +1,233 @@
|
||||
"""C920 I420 SHM -> person mask -> blurred background SHM.
|
||||
|
||||
OpenVINO CPU selfie seg. Does not open DRM. Rally is not processed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
os.environ.setdefault("OMP_NUM_THREADS", "1")
|
||||
os.environ.setdefault("OPENBLAS_NUM_THREADS", "1")
|
||||
os.environ.setdefault("MKL_NUM_THREADS", "1")
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import signal
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
import numpy as np
|
||||
|
||||
from frame_shm import LatestFrameReader, LatestFrameWriter, wait_for_ready
|
||||
from portrait import (
|
||||
MaskHold,
|
||||
PersonSeg,
|
||||
apply_i420,
|
||||
i420_nbytes,
|
||||
i420_to_bgr,
|
||||
person_present,
|
||||
)
|
||||
|
||||
log = logging.getLogger("cameras.portrait-daemon")
|
||||
READY_NAME = "ready"
|
||||
InferFn = Callable[[np.ndarray], np.ndarray]
|
||||
|
||||
|
||||
@dataclass
|
||||
class CamSlot:
|
||||
identity: str
|
||||
width: int
|
||||
height: int
|
||||
reader: LatestFrameReader
|
||||
writer: LatestFrameWriter
|
||||
infer_hz: float = 8.0
|
||||
hold_s: float = 0.8
|
||||
blur_px: int = 7
|
||||
last_seq: int = -1
|
||||
last_infer: float = 0.0
|
||||
hold: MaskHold = field(default_factory=MaskHold)
|
||||
scratch: bytearray = field(default_factory=bytearray)
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
self.hold = MaskHold(hold_s=self.hold_s, ema=0.4)
|
||||
n = i420_nbytes(self.width, self.height)
|
||||
self.scratch = bytearray(n)
|
||||
self.infer_scratch = bytearray(n)
|
||||
self.infer_reader = LatestFrameReader(self.reader.path)
|
||||
|
||||
|
||||
def tick_once(
|
||||
slots: list[CamSlot],
|
||||
infer: InferFn,
|
||||
now: float,
|
||||
infer_budget_s: float = 0.008,
|
||||
) -> int:
|
||||
"""Infer a few due cameras, composite every new SHM frame. Returns writes."""
|
||||
wrote = 0
|
||||
deadline = time.monotonic() + max(0.001, float(infer_budget_s))
|
||||
pending: list[tuple[CamSlot, int]] = []
|
||||
for s in sorted(slots, key=lambda c: c.last_infer):
|
||||
seq = s.reader.copy_into(s.scratch)
|
||||
if seq is None or seq == s.last_seq:
|
||||
continue
|
||||
due = (now - s.last_infer) >= (1.0 / max(0.5, s.infer_hz))
|
||||
if due and time.monotonic() < deadline:
|
||||
bgr = i420_to_bgr(s.scratch, s.width, s.height)
|
||||
raw = infer(bgr)
|
||||
present = person_present(raw)
|
||||
s.hold.update(raw if present else None, present, now)
|
||||
s.last_infer = now
|
||||
pending.append((s, seq))
|
||||
for s, seq in pending:
|
||||
out = apply_i420(
|
||||
s.scratch, s.width, s.height, s.hold.mask, radius=s.blur_px)
|
||||
s.writer.write(out)
|
||||
s.last_seq = seq
|
||||
wrote += 1
|
||||
return wrote
|
||||
|
||||
|
||||
def composite_once(slots: list[CamSlot]) -> int:
|
||||
"""30 fps path: no OpenVINO."""
|
||||
wrote = 0
|
||||
for s in slots:
|
||||
seq = s.reader.copy_into(s.scratch)
|
||||
if seq is None or seq == s.last_seq:
|
||||
continue
|
||||
out = apply_i420(
|
||||
s.scratch, s.width, s.height, s.hold.mask, radius=s.blur_px)
|
||||
s.writer.write(out)
|
||||
s.last_seq = seq
|
||||
wrote += 1
|
||||
return wrote
|
||||
|
||||
|
||||
def infer_once(slots: list[CamSlot], infer: InferFn, now: float, budget_s: float = 0.012) -> int:
|
||||
n = 0
|
||||
deadline = time.monotonic() + max(0.001, float(budget_s))
|
||||
for s in sorted(slots, key=lambda c: c.last_infer):
|
||||
if time.monotonic() >= deadline:
|
||||
break
|
||||
if (now - s.last_infer) < (1.0 / max(0.5, s.infer_hz)):
|
||||
continue
|
||||
seq = s.infer_reader.copy_into(s.infer_scratch)
|
||||
if seq is None:
|
||||
continue
|
||||
bgr = i420_to_bgr(s.infer_scratch, s.width, s.height)
|
||||
raw = infer(bgr)
|
||||
present = person_present(raw)
|
||||
s.hold.update(raw if present else None, present, now)
|
||||
s.last_infer = now
|
||||
n += 1
|
||||
return n
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--cameras-json", required=True)
|
||||
ap.add_argument("--in-dir", default="/run/livekit-cameras/raw")
|
||||
ap.add_argument("--out-dir", default="/run/livekit-cameras/portrait")
|
||||
ap.add_argument("--hz", type=float, default=8.0)
|
||||
ap.add_argument("--blur-px", type=int, default=7)
|
||||
ap.add_argument("--hold-s", type=float, default=0.8)
|
||||
ap.add_argument("--model", default="")
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [portrait-daemon] %(levelname)s %(message)s")
|
||||
import cv2
|
||||
cv2.setNumThreads(1)
|
||||
os.environ.setdefault("OMP_NUM_THREADS", "1")
|
||||
os.environ.setdefault("OPENBLAS_NUM_THREADS", "1")
|
||||
cams = json.loads(args.cameras_json)
|
||||
cams = [c for c in cams if c.get("identity") != "rally"]
|
||||
if not cams:
|
||||
log.error("no C920 cameras")
|
||||
return 2
|
||||
|
||||
model_path = Path(args.model) if args.model else None
|
||||
seg = PersonSeg(model_path=model_path, num_threads=2)
|
||||
log.info(
|
||||
"portrait OpenVINO %s model=%s cams=%d hz=%.1f blur=%d hold=%.2fs",
|
||||
seg.device,
|
||||
(model_path or Path("models/selfie_segmentation.onnx")).name,
|
||||
len(cams), args.hz, args.blur_px, args.hold_s,
|
||||
)
|
||||
|
||||
os.makedirs(args.out_dir, exist_ok=True)
|
||||
slots: list[CamSlot] = []
|
||||
for c in cams:
|
||||
ident = c["identity"]
|
||||
w, h = int(c["width"]), int(c["height"])
|
||||
src = os.path.join(args.in_dir, f"{ident}.i420")
|
||||
dst = os.path.join(args.out_dir, f"{ident}.i420")
|
||||
if not wait_for_ready(src, timeout=20):
|
||||
log.warning("missing capture SHM %s", src)
|
||||
continue
|
||||
slots.append(CamSlot(
|
||||
identity=ident, width=w, height=h,
|
||||
reader=LatestFrameReader(src),
|
||||
writer=LatestFrameWriter(dst, w, h),
|
||||
infer_hz=float(args.hz),
|
||||
hold_s=float(args.hold_s),
|
||||
blur_px=int(args.blur_px),
|
||||
))
|
||||
if not slots:
|
||||
log.error("no portrait slots")
|
||||
return 2
|
||||
|
||||
ready = os.path.join(args.out_dir, READY_NAME)
|
||||
Path(ready).write_text("ok\n")
|
||||
stop = False
|
||||
t0 = time.monotonic()
|
||||
writes = 0
|
||||
infers = 0
|
||||
stop_ev = threading.Event()
|
||||
|
||||
def _stop(*_a):
|
||||
nonlocal stop
|
||||
stop = True
|
||||
stop_ev.set()
|
||||
|
||||
signal.signal(signal.SIGTERM, _stop)
|
||||
signal.signal(signal.SIGINT, _stop)
|
||||
|
||||
def _infer_loop() -> None:
|
||||
nonlocal infers
|
||||
while not stop_ev.is_set():
|
||||
infers += infer_once(slots, seg.infer_bgr, time.monotonic(), 0.012)
|
||||
time.sleep(0.004)
|
||||
|
||||
th = threading.Thread(target=_infer_loop, name="portrait-infer", daemon=True)
|
||||
th.start()
|
||||
try:
|
||||
while not stop:
|
||||
n = composite_once(slots)
|
||||
writes += n
|
||||
if time.monotonic() - t0 >= 10:
|
||||
log.info(
|
||||
"writes=%d infers=%d last_seq %s",
|
||||
writes, infers,
|
||||
" ".join(f"{s.identity}:{s.last_seq}" for s in slots),
|
||||
)
|
||||
writes = 0
|
||||
infers = 0
|
||||
t0 = time.monotonic()
|
||||
if n == 0:
|
||||
time.sleep(0.003)
|
||||
finally:
|
||||
stop_ev.set()
|
||||
try:
|
||||
os.unlink(ready)
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,99 @@
|
||||
"""Logitech Rally Camera V-R0010 + Dante room mic (speaker identity)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from discovery import Camera
|
||||
from usb_ids import (
|
||||
RALLY_IDENTITY,
|
||||
is_rally_pid,
|
||||
looks_like_rally_name,
|
||||
usb_vid_pid_of_video_node,
|
||||
)
|
||||
|
||||
log = logging.getLogger("cameras.rally")
|
||||
|
||||
_CARD_RE = re.compile(r"\s*(\d+)\s+\[([A-Za-z0-9_]+)\s*\]:\s*(.*)")
|
||||
|
||||
|
||||
def dante_capture_card() -> int | None:
|
||||
"""ALSA card index for the Audinate Dante USB I/O module, if present."""
|
||||
path = Path("/proc/asound/cards")
|
||||
if not path.is_file():
|
||||
return None
|
||||
for line in path.read_text().splitlines():
|
||||
m = _CARD_RE.match(line)
|
||||
if not m:
|
||||
continue
|
||||
rest = (m.group(3) or "").lower()
|
||||
name = (m.group(2) or "").lower()
|
||||
if "dante" in rest or "dante" in name:
|
||||
return int(m.group(1))
|
||||
return None
|
||||
|
||||
|
||||
def rally_video_node() -> str | None:
|
||||
pinned = Path("/dev/rally")
|
||||
if pinned.exists():
|
||||
return str(pinned)
|
||||
sysfs = Path("/sys/class/video4linux")
|
||||
if not sysfs.is_dir():
|
||||
return None
|
||||
for p in sorted(sysfs.glob("video*"), key=lambda x: int(x.name[5:] or 0)):
|
||||
name = (p / "name").read_text().strip() if (p / "name").is_file() else ""
|
||||
idx = (p / "index").read_text().strip() if (p / "index").is_file() else ""
|
||||
node = f"/dev/{p.name}"
|
||||
if idx != "0":
|
||||
continue
|
||||
if looks_like_rally_name(name):
|
||||
return node
|
||||
ids = usb_vid_pid_of_video_node(node)
|
||||
if ids is not None and is_rally_pid(ids[1]):
|
||||
return node
|
||||
return None
|
||||
|
||||
|
||||
def _serial_of_video(node: str) -> str:
|
||||
link = Path(f"/sys/class/video4linux/{Path(node).name}/device")
|
||||
try:
|
||||
p = link.resolve()
|
||||
except OSError:
|
||||
return ""
|
||||
for _ in range(12):
|
||||
cand = p / "serial"
|
||||
if cand.is_file():
|
||||
return cand.read_text().strip()
|
||||
if str(p).startswith("/sys/devices"):
|
||||
p = p.parent
|
||||
else:
|
||||
break
|
||||
return ""
|
||||
|
||||
|
||||
def discover_rally(*, width: int = 1920, height: int = 1080) -> Camera | None:
|
||||
"""Speaker camera: Rally UVC + Dante capture. None if the head is missing."""
|
||||
video = rally_video_node()
|
||||
if not video:
|
||||
return None
|
||||
card = dante_capture_card()
|
||||
serial = _serial_of_video(video)
|
||||
cam = Camera(
|
||||
index=20,
|
||||
identity=RALLY_IDENTITY,
|
||||
video_device=video,
|
||||
audio_card=card,
|
||||
serial=serial,
|
||||
width=width,
|
||||
height=height,
|
||||
)
|
||||
log.info(
|
||||
"rally video=%s audio=%s serial=%s %dx%d",
|
||||
video,
|
||||
f"hw:{card},0" if card is not None else "none",
|
||||
serial or "-",
|
||||
width,
|
||||
height,
|
||||
)
|
||||
return cam
|
||||
+580
@@ -0,0 +1,580 @@
|
||||
"""Rally speaker follow: capture /dev/rally to I420 SHM + YuNet PTZ.
|
||||
|
||||
No X11. Local HDMI preview is a LiveKit subscriber (run_displays speaker
|
||||
role on i915 HDMI-A-5). This process does not kmssink unless --connector
|
||||
is set. Person tracking is host software. V-R0010 has no native VISCA.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import os
|
||||
import select
|
||||
import signal
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from drm_outputs import DrmOutput, list_outputs
|
||||
from frame_shm import LatestFrameWriter, frame_bytes
|
||||
from gst_sink import GST_LAUNCH, _kms_props
|
||||
from hop_shm import HopWriter, hop_bytes
|
||||
from visca_xu_bridge import RallyV4L2
|
||||
|
||||
log = logging.getLogger("cameras.rally_follow")
|
||||
|
||||
|
||||
def pick_preview(name: str) -> DrmOutput:
|
||||
for o in list_outputs():
|
||||
if o.name == name:
|
||||
return o
|
||||
have = [f"{o.name}/{o.driver}" for o in list_outputs() if o.connected]
|
||||
raise SystemExit(f"DRM connector {name!r} not found. connected={have}")
|
||||
|
||||
|
||||
def video_cmd(device: str, width: int, height: int, fps: int,
|
||||
fd: int, preview: DrmOutput | None = None) -> list[str]:
|
||||
caps = f"image/jpeg,width={int(width)},height={int(height)},framerate={int(fps)}/1"
|
||||
head = [
|
||||
GST_LAUNCH, "-q",
|
||||
"v4l2src", f"device={device}", "do-timestamp=false",
|
||||
"!", caps,
|
||||
"!", "jpegparse",
|
||||
"!", "jpegdec", "qos=false",
|
||||
]
|
||||
i420 = [
|
||||
"!", "videoconvert", "qos=false", "n-threads=2",
|
||||
"!", f"video/x-raw,format=I420,width={int(width)},height={int(height)}",
|
||||
"!", "fdsink", f"fd={int(fd)}", "sync=false",
|
||||
]
|
||||
if preview is None:
|
||||
return head + i420
|
||||
return head + [
|
||||
"!", "tee", "name=t",
|
||||
"t.", "!", "queue", "max-size-buffers=1", "max-size-bytes=0",
|
||||
"max-size-time=0", "leaky=downstream",
|
||||
"!", "videoconvert", "qos=false",
|
||||
"!", "videoscale", "method=0",
|
||||
"!", *_kms_props(preview),
|
||||
"t.", "!", "queue", "max-size-buffers=1", "max-size-bytes=0",
|
||||
"max-size-time=0", "leaky=downstream",
|
||||
] + i420
|
||||
|
||||
|
||||
def audio_cmd(card: int, rate: int, fd: int) -> list[str]:
|
||||
return [
|
||||
GST_LAUNCH, "-q",
|
||||
"alsasrc", f"device=hw:{int(card)},0", "do-timestamp=false",
|
||||
"!", "audioconvert",
|
||||
"!", "audioresample",
|
||||
"!", f"audio/x-raw,format=S16LE,rate={int(rate)},channels=2",
|
||||
"!", "queue", "max-size-buffers=2", "max-size-bytes=0",
|
||||
"max-size-time=0", "leaky=downstream",
|
||||
"!", "fdsink", f"fd={int(fd)}", "sync=false",
|
||||
]
|
||||
|
||||
|
||||
_YUNET_2026 = Path(__file__).parent / "models" / "face_detection_yunet_2026may.onnx"
|
||||
_YUNET_2023 = Path(__file__).parent / "models" / "face_detection_yunet_2023mar.onnx"
|
||||
_YUNET = _YUNET_2026 if _YUNET_2026.is_file() else _YUNET_2023
|
||||
|
||||
# YuNet row: x,y,w,h, re, le, nose, rmouth, lmouth, score (15).
|
||||
_SCORE_MIN = 0.80
|
||||
_MIN_FRAC = 0.06
|
||||
_MAX_FRAC = 0.50
|
||||
_TOP_BAND = 0.10
|
||||
_BOT_BAND = 0.90
|
||||
_ASPECT_LO = 0.45
|
||||
_ASPECT_HI = 1.40
|
||||
_CONFIRM = 3
|
||||
_HOLD_S = 2.0
|
||||
_MOTION_MAD = 8.0
|
||||
|
||||
|
||||
def resolve_yunet(models_dir: Path | None = None) -> Path:
|
||||
root = Path(models_dir) if models_dir is not None else Path(__file__).parent / "models"
|
||||
newer = root / "face_detection_yunet_2026may.onnx"
|
||||
old = root / "face_detection_yunet_2023mar.onnx"
|
||||
return newer if newer.is_file() else old
|
||||
|
||||
|
||||
def _box_ok(x: float, y: float, w: float, h: float, score: float,
|
||||
frame_w: int, frame_h: int) -> bool:
|
||||
if score < _SCORE_MIN or w <= 1 or h <= 1:
|
||||
return False
|
||||
fw = float(frame_w)
|
||||
fh = float(frame_h)
|
||||
short = min(fw, fh)
|
||||
if min(w, h) < _MIN_FRAC * short:
|
||||
return False
|
||||
if max(w, h) > _MAX_FRAC * max(fw, fh):
|
||||
return False
|
||||
ar = w / h
|
||||
if ar < _ASPECT_LO or ar > _ASPECT_HI:
|
||||
return False
|
||||
cx = (x + w / 2.0) / fw
|
||||
cy = (y + h / 2.0) / fh
|
||||
if cy < _TOP_BAND or cy > _BOT_BAND:
|
||||
return False
|
||||
if cx < 0.02 or cx > 0.98:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _frontal_landmarks(row, x: float, y: float, w: float, h: float) -> bool:
|
||||
re_x, re_y = float(row[4]), float(row[5])
|
||||
le_x, le_y = float(row[6]), float(row[7])
|
||||
ns_x, ns_y = float(row[8]), float(row[9])
|
||||
rm_x, rm_y = float(row[10]), float(row[11])
|
||||
lm_x, lm_y = float(row[12]), float(row[13])
|
||||
eye_y = 0.5 * (re_y + le_y)
|
||||
mouth_y = 0.5 * (rm_y + lm_y)
|
||||
if not (eye_y + 0.04 * h < ns_y < mouth_y - 0.04 * h):
|
||||
return False
|
||||
eye_lo = min(re_x, le_x)
|
||||
eye_hi = max(re_x, le_x)
|
||||
if not (eye_lo + 0.08 * w <= ns_x <= eye_hi - 0.08 * w):
|
||||
return False
|
||||
eye_dist = ((le_x - re_x) ** 2 + (le_y - re_y) ** 2) ** 0.5
|
||||
if eye_dist < 0.22 * w or eye_dist > 0.85 * w:
|
||||
return False
|
||||
if abs(le_y - re_y) > 0.35 * h:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _profile_landmarks(row, x: float, y: float, w: float, h: float) -> bool:
|
||||
"""Side / turned head: landmarks clustered, nose not between the eyes."""
|
||||
pts = [(float(row[i]), float(row[i + 1])) for i in range(4, 14, 2)]
|
||||
pad_x, pad_y = 0.2 * w, 0.2 * h
|
||||
inside = 0
|
||||
for px, py in pts:
|
||||
if (x - pad_x) <= px <= (x + w + pad_x) and (y - pad_y) <= py <= (y + h + pad_y):
|
||||
inside += 1
|
||||
if inside < 3:
|
||||
return False
|
||||
re_x, re_y = float(row[4]), float(row[5])
|
||||
le_x, le_y = float(row[6]), float(row[7])
|
||||
eye_dist = ((le_x - re_x) ** 2 + (le_y - re_y) ** 2) ** 0.5
|
||||
if eye_dist >= 0.22 * w:
|
||||
return False
|
||||
ns_y = float(row[9])
|
||||
rm_y, lm_y = float(row[11]), float(row[13])
|
||||
eye_y = 0.5 * (re_y + le_y)
|
||||
mouth_y = 0.5 * (rm_y + lm_y)
|
||||
if mouth_y + 0.02 * h < eye_y:
|
||||
return False
|
||||
if ns_y + 0.12 * h < eye_y:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def accept_yunet_face(row, frame_w: int, frame_h: int) -> bool:
|
||||
"""True if this YuNet hit looks like a real face (frontal or turned), not ceiling/stand."""
|
||||
if row is None or len(row) < 15:
|
||||
return False
|
||||
x, y, w, h = (float(row[0]), float(row[1]), float(row[2]), float(row[3]))
|
||||
score = float(row[14])
|
||||
if not _box_ok(x, y, w, h, score, frame_w, frame_h):
|
||||
return False
|
||||
return _frontal_landmarks(row, x, y, w, h) or _profile_landmarks(row, x, y, w, h)
|
||||
|
||||
|
||||
def select_yunet_face(faces, frame_w: int, frame_h: int):
|
||||
"""Highest-score accepted face, or None."""
|
||||
if faces is None or len(faces) == 0:
|
||||
return None
|
||||
best = None
|
||||
best_score = -1.0
|
||||
for row in faces:
|
||||
if not accept_yunet_face(row, frame_w, frame_h):
|
||||
continue
|
||||
sc = float(row[14])
|
||||
if sc > best_score:
|
||||
best_score = sc
|
||||
best = row
|
||||
return best
|
||||
|
||||
|
||||
class FaceGate:
|
||||
"""Need `need` similar hits before the box is trusted for PTZ."""
|
||||
|
||||
def __init__(self, need: int = _CONFIRM, max_jump: float = 0.18) -> None:
|
||||
self.need = max(1, int(need))
|
||||
self.max_jump = float(max_jump)
|
||||
self.hits = 0
|
||||
self.cx = 0.0
|
||||
self.cy = 0.0
|
||||
|
||||
def update(self, box: tuple[int, int, int, int] | None,
|
||||
frame_w: int, frame_h: int) -> tuple[int, int, int, int] | None:
|
||||
if box is None:
|
||||
self.hits = 0
|
||||
return None
|
||||
x, y, w, h = box
|
||||
cx = (x + w / 2.0) / float(frame_w)
|
||||
cy = (y + h / 2.0) / float(frame_h)
|
||||
if self.hits > 0:
|
||||
jump = ((cx - self.cx) ** 2 + (cy - self.cy) ** 2) ** 0.5
|
||||
if jump > self.max_jump:
|
||||
self.hits = 1
|
||||
else:
|
||||
self.hits += 1
|
||||
else:
|
||||
self.hits = 1
|
||||
self.cx, self.cy = cx, cy
|
||||
if self.hits >= self.need:
|
||||
return box
|
||||
return None
|
||||
|
||||
|
||||
def roi_motion(
|
||||
prev: np.ndarray | None,
|
||||
gray: np.ndarray | None,
|
||||
box: tuple[int, int, int, int] | None,
|
||||
*,
|
||||
expand: float = 1.4,
|
||||
min_mad: float = _MOTION_MAD,
|
||||
) -> bool:
|
||||
"""True if the last head ROI moved; motion elsewhere is ignored."""
|
||||
if prev is None or gray is None or box is None:
|
||||
return False
|
||||
if prev.shape != gray.shape:
|
||||
return False
|
||||
h_img, w_img = gray.shape[:2]
|
||||
x, y, w, h = (int(box[0]), int(box[1]), int(box[2]), int(box[3]))
|
||||
extra_w = int((expand - 1.0) * w / 2.0)
|
||||
extra_h = int((expand - 1.0) * h / 2.0)
|
||||
x0 = max(0, x - extra_w)
|
||||
y0 = max(0, y - extra_h)
|
||||
x1 = min(w_img, x + w + extra_w)
|
||||
y1 = min(h_img, y + h + extra_h)
|
||||
if x1 - x0 < 4 or y1 - y0 < 4:
|
||||
return False
|
||||
a = prev[y0:y1, x0:x1].astype(np.float32)
|
||||
b = gray[y0:y1, x0:x1].astype(np.float32)
|
||||
return float(np.mean(np.abs(a - b))) >= float(min_mad)
|
||||
|
||||
|
||||
class HeadHold:
|
||||
"""Keep PTZ on the last confirmed face while the head ROI is still moving."""
|
||||
|
||||
def __init__(self, hold_s: float = _HOLD_S) -> None:
|
||||
self.hold_s = max(0.0, float(hold_s))
|
||||
self.box: tuple[int, int, int, int] | None = None
|
||||
self.until = 0.0
|
||||
|
||||
def update(
|
||||
self,
|
||||
box: tuple[int, int, int, int] | None,
|
||||
*,
|
||||
moving: bool,
|
||||
now: float,
|
||||
) -> tuple[int, int, int, int] | None:
|
||||
if box is not None:
|
||||
self.box = box
|
||||
self.until = now + self.hold_s
|
||||
return box
|
||||
if self.box is None:
|
||||
return None
|
||||
if moving:
|
||||
self.until = now + self.hold_s
|
||||
return self.box
|
||||
if now < self.until:
|
||||
return self.box
|
||||
self.box = None
|
||||
return None
|
||||
|
||||
|
||||
def steer_speeds(
|
||||
box: tuple[int, int, int, int],
|
||||
frame_w: int,
|
||||
frame_h: int,
|
||||
deadband: float = 0.14,
|
||||
) -> tuple[int, int]:
|
||||
x, y, w, h = box
|
||||
err_x = ((x + w / 2.0) - frame_w / 2.0) / (frame_w / 2.0)
|
||||
err_y = ((y + h / 2.0) - frame_h / 2.0) / (frame_h / 2.0)
|
||||
pan = 1 if err_x > deadband else (-1 if err_x < -deadband else 0)
|
||||
tilt = -1 if err_y > deadband else (1 if err_y < -deadband else 0)
|
||||
return pan, tilt
|
||||
|
||||
|
||||
def lock_action(
|
||||
confirmed: tuple[int, int, int, int] | None,
|
||||
held: tuple[int, int, int, int] | None,
|
||||
) -> str:
|
||||
"""Steer only on a live face. A stale hold freezes PTZ (no ghost pan)."""
|
||||
if confirmed is not None:
|
||||
return "steer"
|
||||
if held is not None:
|
||||
return "freeze"
|
||||
return "lost"
|
||||
|
||||
|
||||
class Follower:
|
||||
"""Keep a confirmed YuNet face near frame center. Hold on loss.
|
||||
|
||||
OpenCV 5 dropped Haar/HOG; YuNet FaceDetectorYN is the replacement.
|
||||
Tiny / ceiling / stand / landmark-invalid hits are dropped before PTZ.
|
||||
"""
|
||||
|
||||
def __init__(self, ptz: RallyV4L2, width: int, height: int,
|
||||
deadband: float = 0.14, detect_every: int = 5):
|
||||
self.ptz = ptz
|
||||
self.w = int(width)
|
||||
self.h = int(height)
|
||||
self.deadband = float(deadband)
|
||||
self.detect_every = max(1, int(detect_every))
|
||||
self._n = 0
|
||||
self._lost = 0
|
||||
self._det = None
|
||||
self._cv2 = None
|
||||
self._gate = FaceGate()
|
||||
self._hold = HeadHold()
|
||||
self._frozen = False
|
||||
self._gray: np.ndarray | None = None
|
||||
self._prev: np.ndarray | None = None
|
||||
self._dw = max(160, self.w // 2)
|
||||
self._dh = max(90, self.h // 2)
|
||||
model = resolve_yunet()
|
||||
try:
|
||||
import cv2
|
||||
self._cv2 = cv2
|
||||
if not model.is_file():
|
||||
raise FileNotFoundError(model)
|
||||
self._det = cv2.FaceDetectorYN_create(
|
||||
str(model), "", (self._dw, self._dh), _SCORE_MIN, 0.3, 5000)
|
||||
log.info("yunet face detector %s input=%dx%d score>=%.2f",
|
||||
model.name, self._dw, self._dh, _SCORE_MIN)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log.warning("opencv tracker unavailable: %s", exc)
|
||||
|
||||
def _boxes(self, payload: bytes) -> list[tuple[int, int, int, int]]:
|
||||
if self._cv2 is None or self._det is None:
|
||||
return []
|
||||
yuv = np.frombuffer(payload, dtype=np.uint8).reshape((self.h * 3 // 2, self.w))
|
||||
bgr = self._cv2.cvtColor(yuv, self._cv2.COLOR_YUV2BGR_I420)
|
||||
small = self._cv2.resize(bgr, (self._dw, self._dh))
|
||||
self._gray = self._cv2.cvtColor(small, self._cv2.COLOR_BGR2GRAY)
|
||||
self._det.setInputSize((self._dw, self._dh))
|
||||
_ok, faces = self._det.detect(small)
|
||||
picked = select_yunet_face(faces, self._dw, self._dh)
|
||||
if picked is None:
|
||||
return []
|
||||
sx = self.w / float(self._dw)
|
||||
sy = self.h / float(self._dh)
|
||||
x, yy, ww, hh = (float(picked[0]), float(picked[1]),
|
||||
float(picked[2]), float(picked[3]))
|
||||
return [(int(x * sx), int(yy * sy), int(ww * sx), int(hh * sy))]
|
||||
|
||||
def on_i420(self, payload: bytes) -> None:
|
||||
self._n += 1
|
||||
if self._n % self.detect_every != 0:
|
||||
return
|
||||
boxes = self._boxes(payload)
|
||||
now = time.monotonic()
|
||||
hold_box = self._hold.box
|
||||
if hold_box is not None and self._gray is not None:
|
||||
sx = self._dw / float(self.w)
|
||||
sy = self._dh / float(self.h)
|
||||
small_box = (
|
||||
int(hold_box[0] * sx), int(hold_box[1] * sy),
|
||||
int(hold_box[2] * sx), int(hold_box[3] * sy),
|
||||
)
|
||||
else:
|
||||
small_box = None
|
||||
moving = roi_motion(self._prev, self._gray, small_box)
|
||||
self._prev = None if self._gray is None else self._gray.copy()
|
||||
face = boxes[0] if boxes else None
|
||||
confirmed = self._gate.update(face, self.w, self.h)
|
||||
held = self._hold.update(confirmed, moving=moving, now=now)
|
||||
action = lock_action(confirmed, held)
|
||||
if action == "steer" and confirmed is not None:
|
||||
self._lost = 0
|
||||
self._frozen = False
|
||||
pan, tilt = steer_speeds(confirmed, self.w, self.h, self.deadband)
|
||||
self.ptz.set_speed(pan, tilt)
|
||||
return
|
||||
if action == "freeze":
|
||||
self._lost = 0
|
||||
if not self._frozen:
|
||||
self.ptz.stop()
|
||||
self._frozen = True
|
||||
return
|
||||
self._lost += 1
|
||||
if self._lost == 1 or self._lost == 6:
|
||||
self.ptz.stop()
|
||||
self._frozen = True
|
||||
if self._lost == 6:
|
||||
log.info("target lost; hold")
|
||||
|
||||
|
||||
def _pipe() -> tuple[int, int]:
|
||||
r, w = os.pipe()
|
||||
try:
|
||||
import fcntl
|
||||
fcntl.fcntl(r, fcntl.F_SETPIPE_SZ, 1048576)
|
||||
fcntl.fcntl(w, fcntl.F_SETPIPE_SZ, 1048576)
|
||||
except OSError:
|
||||
pass
|
||||
os.set_inheritable(w, True)
|
||||
os.set_inheritable(r, False)
|
||||
return r, w
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("--device", default="/dev/rally")
|
||||
ap.add_argument("--connector", default="",
|
||||
help="optional local kmssink; empty = LiveKit hairpin only")
|
||||
ap.add_argument("--width", type=int, default=1920)
|
||||
ap.add_argument("--height", type=int, default=1080)
|
||||
ap.add_argument("--fps", type=int, default=30)
|
||||
ap.add_argument("--shm-dir", default="/run/livekit-cameras/raw")
|
||||
ap.add_argument("--audio-shm-dir", default="/run/livekit-cameras/pcm")
|
||||
ap.add_argument("--audio-card", type=int, default=-1)
|
||||
ap.add_argument("--audio-rate", type=int, default=16000)
|
||||
ap.add_argument("--hop-s", type=float, default=0.1)
|
||||
ap.add_argument("--identity", default="rally")
|
||||
ap.add_argument("--no-follow", action="store_true")
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [rally-follow] %(levelname)s %(message)s")
|
||||
|
||||
preview = pick_preview(args.connector) if args.connector else None
|
||||
if preview is not None and preview.driver == "xe":
|
||||
log.error("refusing to kmssink on Arc (%s); wall stays there", preview.name)
|
||||
return 2
|
||||
if preview is None:
|
||||
log.info("preview via LiveKit hairpin (no local kmssink)")
|
||||
else:
|
||||
log.info("preview %s id=%s %s %s", preview.name, preview.connector_id,
|
||||
preview.driver, preview.node)
|
||||
|
||||
need = frame_bytes(args.width, args.height)
|
||||
os.makedirs(args.shm_dir, exist_ok=True)
|
||||
video_w = LatestFrameWriter(
|
||||
os.path.join(args.shm_dir, f"{args.identity}.i420"),
|
||||
args.width, args.height)
|
||||
|
||||
hop_n = hop_bytes(args.audio_rate, 2, args.hop_s)
|
||||
audio_w = None
|
||||
audio_r = audio_wr = None
|
||||
audio_proc = None
|
||||
audio_buf = bytearray()
|
||||
if args.audio_card >= 0:
|
||||
os.makedirs(args.audio_shm_dir, exist_ok=True)
|
||||
audio_w = HopWriter(
|
||||
os.path.join(args.audio_shm_dir, f"{args.identity}.pcm"), hop_n)
|
||||
audio_r, audio_wr = _pipe()
|
||||
|
||||
vr, vw = _pipe()
|
||||
vcmd = video_cmd(args.device, args.width, args.height, args.fps, vw, preview)
|
||||
log.info("gst video %s", " ".join(vcmd))
|
||||
vproc = subprocess.Popen(
|
||||
vcmd, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL,
|
||||
stderr=None, pass_fds=(vw,))
|
||||
os.close(vw)
|
||||
|
||||
if audio_r is not None and audio_wr is not None:
|
||||
acmd = audio_cmd(args.audio_card, args.audio_rate, audio_wr)
|
||||
log.info("gst audio %s", " ".join(acmd))
|
||||
audio_proc = subprocess.Popen(
|
||||
acmd, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL,
|
||||
stderr=None, pass_fds=(audio_wr,))
|
||||
os.close(audio_wr)
|
||||
os.set_blocking(audio_r, False)
|
||||
|
||||
os.set_blocking(vr, False)
|
||||
follower = None if args.no_follow else Follower(
|
||||
RallyV4L2(args.device), args.width, args.height)
|
||||
stop = False
|
||||
|
||||
def _stop(_signum=None, _frame=None):
|
||||
nonlocal stop
|
||||
stop = True
|
||||
|
||||
signal.signal(signal.SIGTERM, _stop)
|
||||
signal.signal(signal.SIGINT, _stop)
|
||||
|
||||
vbuf = bytearray()
|
||||
seq = 0
|
||||
last = time.monotonic()
|
||||
try:
|
||||
while not stop:
|
||||
if vproc.poll() is not None:
|
||||
log.error("video gst exited rc=%s", vproc.returncode)
|
||||
break
|
||||
if audio_proc is not None and audio_proc.poll() is not None:
|
||||
log.error("audio gst exited rc=%s", audio_proc.returncode)
|
||||
break
|
||||
fds = [vr]
|
||||
if audio_r is not None:
|
||||
fds.append(audio_r)
|
||||
ready, _, _ = select.select(fds, [], [], 0.2)
|
||||
if vr in ready:
|
||||
while True:
|
||||
try:
|
||||
chunk = os.read(vr, max(need, need - len(vbuf)))
|
||||
except BlockingIOError:
|
||||
break
|
||||
if not chunk:
|
||||
break
|
||||
vbuf.extend(chunk)
|
||||
while len(vbuf) >= need:
|
||||
frame = bytes(vbuf[:need])
|
||||
del vbuf[:need]
|
||||
seq = video_w.write(frame)
|
||||
if follower is not None:
|
||||
follower.on_i420(frame)
|
||||
if audio_r is not None and audio_r in ready and audio_w is not None:
|
||||
while True:
|
||||
try:
|
||||
chunk = os.read(audio_r, max(hop_n, hop_n - len(audio_buf)))
|
||||
except BlockingIOError:
|
||||
break
|
||||
if not chunk:
|
||||
break
|
||||
audio_buf.extend(chunk)
|
||||
while len(audio_buf) >= hop_n:
|
||||
hop = bytes(audio_buf[:hop_n])
|
||||
del audio_buf[:hop_n]
|
||||
audio_w.write(hop)
|
||||
now = time.monotonic()
|
||||
if now - last >= 5.0:
|
||||
last = now
|
||||
log.info("video seq=%d preview=%s follow=%s",
|
||||
seq, preview.name if preview else "livekit",
|
||||
follower is not None)
|
||||
finally:
|
||||
if follower is not None:
|
||||
follower.ptz.stop()
|
||||
for proc in (vproc, audio_proc):
|
||||
if proc is None or proc.poll() is not None:
|
||||
continue
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
video_w.close()
|
||||
if audio_w is not None:
|
||||
audio_w.close()
|
||||
for fd in (vr, audio_r):
|
||||
if fd is None:
|
||||
continue
|
||||
try:
|
||||
os.close(fd)
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -21,19 +21,183 @@ import time
|
||||
from pathlib import Path
|
||||
|
||||
from config import load_config, _load_dotenv
|
||||
from discovery import discover_cameras, select_cameras
|
||||
from discovery import Camera, discover_cameras, select_cameras
|
||||
from latency import effective_enhance_mode
|
||||
from participant_tags import token_attributes
|
||||
from rally import discover_rally
|
||||
from tokens import make_token
|
||||
|
||||
log = logging.getLogger("cameras.main")
|
||||
|
||||
|
||||
def worker_args_for(cfg, cam) -> list[str]:
|
||||
return [
|
||||
"--camera-json", json.dumps(cam.__dict__),
|
||||
"--room", cfg.room,
|
||||
def spawn_enhance_daemon(cfg) -> subprocess.Popen:
|
||||
from enhance_daemon import wait_for_socket
|
||||
|
||||
cmd = [
|
||||
sys.executable, str(Path(__file__).parent / "enhance_daemon.py"),
|
||||
"--socket", cfg.enhance_socket,
|
||||
"--enhance-mode", "off" if cfg.enhance_mode == "off" else "force",
|
||||
"--enhance-model", cfg.enhance_model,
|
||||
"--enhance-chunk-s", str(cfg.enhance_chunk_s),
|
||||
"--enhance-hop-s", str(cfg.enhance_hop_s),
|
||||
"--enhance-infer", cfg.enhance_infer,
|
||||
"--beamform-mode", cfg.beamform_mode,
|
||||
"--torch-num-threads", str(max(2, cfg.torch_num_threads)),
|
||||
"--tdoa-every", str(cfg.beamform_tdoa_every),
|
||||
"--audio-rate", str(cfg.audio_rate),
|
||||
]
|
||||
log.info("starting shared enhance daemon at %s", cfg.enhance_socket)
|
||||
proc = subprocess.Popen(cmd)
|
||||
if not wait_for_socket(cfg.enhance_socket, timeout=90):
|
||||
proc.kill()
|
||||
raise RuntimeError(
|
||||
f"enhance daemon did not listen on {cfg.enhance_socket}")
|
||||
log.info("enhance daemon ready pid=%s", proc.pid)
|
||||
return proc
|
||||
|
||||
|
||||
def spawn_capture_daemon(cfg, cameras) -> subprocess.Popen:
|
||||
from frame_shm import wait_for_ready
|
||||
|
||||
payload = [{"identity": c.identity, "video_device": c.video_device}
|
||||
for c in cameras]
|
||||
cmd = [
|
||||
sys.executable, str(Path(__file__).parent / "capture_daemon.py"),
|
||||
"--cameras-json", json.dumps(payload),
|
||||
"--width", str(cfg.width),
|
||||
"--height", str(cfg.height),
|
||||
"--fps", str(cfg.fps),
|
||||
"--shm-dir", cfg.capture_shm_dir,
|
||||
"--vaapi-device", cfg.vaapi_device,
|
||||
]
|
||||
log.info("starting shared capture daemon shm=%s cams=%d",
|
||||
cfg.capture_shm_dir, len(payload))
|
||||
env = dict(os.environ)
|
||||
env["GST_REGISTRY_UPDATE"] = "no"
|
||||
env.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
|
||||
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
|
||||
proc = subprocess.Popen(cmd, env=env)
|
||||
ready = str(Path(cfg.capture_shm_dir) / "ready")
|
||||
if not wait_for_ready(ready, timeout=30):
|
||||
proc.kill()
|
||||
raise RuntimeError(f"capture daemon did not become ready at {ready}")
|
||||
log.info("capture daemon ready pid=%s", proc.pid)
|
||||
return proc
|
||||
|
||||
|
||||
def spawn_audio_daemon(cfg, cameras) -> subprocess.Popen:
|
||||
from frame_shm import wait_for_ready
|
||||
|
||||
payload = [{"identity": c.identity, "audio_card": c.audio_card}
|
||||
for c in cameras if c.audio_card is not None]
|
||||
if not payload:
|
||||
raise RuntimeError("no cameras with audio cards")
|
||||
cmd = [
|
||||
sys.executable, str(Path(__file__).parent / "audio_daemon.py"),
|
||||
"--cameras-json", json.dumps(payload),
|
||||
"--rate", str(cfg.audio_rate),
|
||||
"--hop-s", str(cfg.enhance_hop_s),
|
||||
"--shm-dir", cfg.audio_shm_dir,
|
||||
]
|
||||
log.info("starting shared audio daemon shm=%s mics=%d",
|
||||
cfg.audio_shm_dir, len(payload))
|
||||
proc = subprocess.Popen(cmd)
|
||||
ready = str(Path(cfg.audio_shm_dir) / "ready")
|
||||
if not wait_for_ready(ready, timeout=30):
|
||||
proc.kill()
|
||||
raise RuntimeError(f"audio daemon did not become ready at {ready}")
|
||||
log.info("audio daemon ready pid=%s", proc.pid)
|
||||
return proc
|
||||
|
||||
|
||||
def spawn_encode_daemon(cfg, cameras, portrait_dir: str = "") -> subprocess.Popen:
|
||||
from frame_shm import wait_for_ready
|
||||
|
||||
payload = []
|
||||
for c in cameras:
|
||||
payload.append({
|
||||
"identity": c.identity,
|
||||
"width": int(c.width or cfg.width),
|
||||
"height": int(c.height or cfg.height),
|
||||
})
|
||||
cmd = [
|
||||
sys.executable, str(Path(__file__).parent / "encode_daemon.py"),
|
||||
"--cameras-json", json.dumps(payload),
|
||||
"--i420-dir", cfg.capture_shm_dir,
|
||||
"--h264-dir", cfg.encode_shm_dir,
|
||||
"--fps", str(cfg.fps),
|
||||
"--bitrate", str(cfg.video_bitrate),
|
||||
"--hi-bitrate", str(cfg.rally_video_bitrate),
|
||||
"--vaapi-device", cfg.vaapi_device,
|
||||
"--codec", cfg.video_codec,
|
||||
]
|
||||
if portrait_dir:
|
||||
cmd.extend(["--portrait-dir", portrait_dir])
|
||||
log.info("starting encode daemon h264=%s cams=%d codec=%s portrait=%s",
|
||||
cfg.encode_shm_dir, len(payload), cfg.video_codec,
|
||||
portrait_dir or "off")
|
||||
env = dict(os.environ)
|
||||
env["GST_REGISTRY_UPDATE"] = "no"
|
||||
env.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
|
||||
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
|
||||
proc = subprocess.Popen(cmd, env=env)
|
||||
ready = str(Path(cfg.encode_shm_dir) / "ready")
|
||||
if not wait_for_ready(ready, timeout=60):
|
||||
proc.kill()
|
||||
raise RuntimeError(f"encode daemon did not become ready at {ready}")
|
||||
log.info("encode daemon ready pid=%s", proc.pid)
|
||||
return proc
|
||||
|
||||
|
||||
def spawn_portrait_daemon(cfg, cameras) -> subprocess.Popen:
|
||||
from frame_shm import wait_for_ready
|
||||
|
||||
payload = []
|
||||
for c in cameras:
|
||||
if c.identity == "rally":
|
||||
continue
|
||||
payload.append({
|
||||
"identity": c.identity,
|
||||
"width": int(c.width or cfg.width),
|
||||
"height": int(c.height or cfg.height),
|
||||
})
|
||||
cmd = [
|
||||
sys.executable, str(Path(__file__).parent / "portrait_daemon.py"),
|
||||
"--cameras-json", json.dumps(payload),
|
||||
"--in-dir", cfg.capture_shm_dir,
|
||||
"--out-dir", cfg.portrait_shm_dir,
|
||||
"--hz", str(cfg.portrait_hz),
|
||||
"--blur-px", str(cfg.portrait_blur_px),
|
||||
"--hold-s", str(cfg.portrait_hold_s),
|
||||
]
|
||||
log.info("starting portrait daemon shm=%s cams=%d hz=%.1f",
|
||||
cfg.portrait_shm_dir, len(payload), cfg.portrait_hz)
|
||||
env = dict(os.environ)
|
||||
env["OMP_NUM_THREADS"] = "1"
|
||||
env["OPENBLAS_NUM_THREADS"] = "1"
|
||||
env["MKL_NUM_THREADS"] = "1"
|
||||
proc = subprocess.Popen(cmd, env=env)
|
||||
ready = str(Path(cfg.portrait_shm_dir) / "ready")
|
||||
if not wait_for_ready(ready, timeout=45):
|
||||
proc.kill()
|
||||
raise RuntimeError(f"portrait daemon did not become ready at {ready}")
|
||||
log.info("portrait daemon ready pid=%s", proc.pid)
|
||||
return proc
|
||||
|
||||
|
||||
def publisher_args_for(cfg, cameras) -> list[str]:
|
||||
payload = []
|
||||
for cam in cameras:
|
||||
d = dict(cam.__dict__)
|
||||
d["token"] = make_token(
|
||||
cfg.url, cfg.api_key, cfg.api_secret,
|
||||
cfg.room, cam.identity,
|
||||
attributes=token_attributes(cfg.participant_tags))
|
||||
payload.append(d)
|
||||
args = [
|
||||
"--cameras-json", json.dumps(payload),
|
||||
"--url", cfg.url,
|
||||
"--token", make_token(cfg.url, cfg.api_key, cfg.api_secret,
|
||||
cfg.room, cam.identity),
|
||||
"--room", cfg.room,
|
||||
"--width", str(cfg.width),
|
||||
"--height", str(cfg.height),
|
||||
"--fps", str(cfg.fps),
|
||||
@@ -50,7 +214,128 @@ def worker_args_for(cfg, cam) -> list[str]:
|
||||
"--beamform-mode", cfg.beamform_mode,
|
||||
"--torch-num-threads", str(cfg.torch_num_threads),
|
||||
"--tdoa-every", str(cfg.beamform_tdoa_every),
|
||||
"--video-shm-dir", cfg.capture_shm_dir,
|
||||
"--audio-shm-dir", cfg.audio_shm_dir,
|
||||
"--h264-shm-dir", cfg.encode_shm_dir,
|
||||
"--speaker-camera", cfg.display_speaker_camera,
|
||||
"--audio-gate-open-db", str(cfg.audio_gate_open_db),
|
||||
"--audio-gate-close-db", str(cfg.audio_gate_close_db),
|
||||
"--audio-gate-hold-s", str(cfg.audio_gate_hold_s),
|
||||
]
|
||||
if cfg.audio_queue_ms is not None:
|
||||
args.extend(["--audio-queue-ms", str(cfg.audio_queue_ms)])
|
||||
if cfg.enhance_daemon and cfg.enhance_socket:
|
||||
args.extend(["--enhance-socket", cfg.enhance_socket])
|
||||
if cfg.enhance_speaker_only:
|
||||
args.append("--enhance-speaker-only")
|
||||
if cfg.audio_gate:
|
||||
args.append("--audio-gate")
|
||||
return args
|
||||
|
||||
|
||||
def fleet_cameras(cameras: list[Camera]) -> list[Camera]:
|
||||
return [c for c in cameras if c.identity != "rally"]
|
||||
|
||||
|
||||
def spawn_rally_follow(cfg, cam: Camera) -> subprocess.Popen:
|
||||
cmd = [
|
||||
sys.executable, str(Path(__file__).parent / "rally_follow.py"),
|
||||
"--device", cam.video_device,
|
||||
"--width", str(cam.width or 1920),
|
||||
"--height", str(cam.height or 1080),
|
||||
"--fps", "30",
|
||||
"--shm-dir", cfg.capture_shm_dir,
|
||||
"--audio-shm-dir", cfg.audio_shm_dir,
|
||||
"--identity", cam.identity,
|
||||
"--audio-rate", str(cfg.audio_rate),
|
||||
"--hop-s", str(cfg.enhance_hop_s),
|
||||
]
|
||||
if cam.audio_card is not None:
|
||||
cmd.extend(["--audio-card", str(cam.audio_card)])
|
||||
log.info("starting rally follow video=%s audio=%s (preview=LiveKit hairpin)",
|
||||
cam.video_device,
|
||||
f"hw:{cam.audio_card},0" if cam.audio_card is not None else "none")
|
||||
return subprocess.Popen(cmd)
|
||||
|
||||
|
||||
def spawn_publisher(cfg, cameras) -> subprocess.Popen:
|
||||
args = publisher_args_for(cfg, cameras)
|
||||
env = dict(os.environ)
|
||||
env.setdefault("LIBVA_DRIVER_NAME", "iHD")
|
||||
env.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
|
||||
log.info("starting single publisher cams=%d", len(cameras))
|
||||
return subprocess.Popen(
|
||||
[sys.executable, str(Path(__file__).parent / "run_publisher.py"), *args],
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, bufsize=1, env=env,
|
||||
)
|
||||
|
||||
|
||||
def stop_proc(proc: subprocess.Popen | None) -> None:
|
||||
if proc is None or proc.poll() is not None:
|
||||
return
|
||||
proc.send_signal(signal.SIGTERM)
|
||||
with contextlib.suppress(Exception):
|
||||
proc.wait(timeout=10)
|
||||
if proc.poll() is None:
|
||||
proc.kill()
|
||||
|
||||
|
||||
def worker_args_for(cfg, cam) -> list[str]:
|
||||
enhance_mode = effective_enhance_mode(
|
||||
cfg.enhance_mode,
|
||||
cam.identity,
|
||||
cfg.enhance_speaker_only,
|
||||
cfg.display_speaker_camera,
|
||||
)
|
||||
beamform_mode = cfg.beamform_mode
|
||||
if cfg.enhance_speaker_only and enhance_mode == "off":
|
||||
beamform_mode = "off"
|
||||
args = [
|
||||
"--camera-json", json.dumps(cam.__dict__),
|
||||
"--room", cfg.room,
|
||||
"--url", cfg.url,
|
||||
"--token", make_token(
|
||||
cfg.url, cfg.api_key, cfg.api_secret,
|
||||
cfg.room, cam.identity,
|
||||
attributes=token_attributes(cfg.participant_tags)),
|
||||
"--width", str(cfg.width),
|
||||
"--height", str(cfg.height),
|
||||
"--fps", str(cfg.fps),
|
||||
"--bitrate", str(cfg.video_bitrate),
|
||||
"--video-codec", cfg.video_codec,
|
||||
"--video-encoder", cfg.video_encoder,
|
||||
"--vaapi-device", cfg.vaapi_device,
|
||||
"--audio-rate", str(cfg.audio_rate),
|
||||
"--enhance-mode", enhance_mode,
|
||||
"--enhance-model", cfg.enhance_model,
|
||||
"--enhance-chunk-s", str(cfg.enhance_chunk_s),
|
||||
"--enhance-hop-s", str(cfg.enhance_hop_s),
|
||||
"--enhance-infer", cfg.enhance_infer,
|
||||
"--beamform-mode", beamform_mode,
|
||||
"--torch-num-threads", str(cfg.torch_num_threads),
|
||||
"--tdoa-every", str(cfg.beamform_tdoa_every),
|
||||
"--speaker-camera", cfg.display_speaker_camera,
|
||||
"--audio-gate-open-db", str(cfg.audio_gate_open_db),
|
||||
"--audio-gate-close-db", str(cfg.audio_gate_close_db),
|
||||
"--audio-gate-hold-s", str(cfg.audio_gate_hold_s),
|
||||
]
|
||||
if cfg.audio_queue_ms is not None:
|
||||
args.extend(["--audio-queue-ms", str(cfg.audio_queue_ms)])
|
||||
if cfg.video_hold_s is not None:
|
||||
args.extend(["--video-hold-s", str(cfg.video_hold_s)])
|
||||
if cfg.worker_split_executor:
|
||||
args.append("--split-executor")
|
||||
if cfg.enhance_daemon and cfg.enhance_socket:
|
||||
args.extend(["--enhance-socket", cfg.enhance_socket])
|
||||
if cfg.audio_gate:
|
||||
args.append("--audio-gate")
|
||||
if cfg.capture_daemon:
|
||||
args.extend([
|
||||
"--video-shm",
|
||||
f"{cfg.capture_shm_dir.rstrip('/')}/{cam.identity}.i420",
|
||||
])
|
||||
return args
|
||||
|
||||
|
||||
class Supervisor:
|
||||
@@ -128,6 +413,10 @@ def main() -> int:
|
||||
log.error("LIVEKIT_URL / LIVEKIT_API_KEY / LIVEKIT_API_SECRET "
|
||||
"not set (create .env from .env.example)")
|
||||
return 2
|
||||
if not cfg.url.startswith("wss://"):
|
||||
log.error("LIVEKIT_URL must be wss:// (refusing %s)", cfg.url)
|
||||
return 2
|
||||
log.info("livekit destination url=%s room=%s", cfg.url, cfg.room)
|
||||
|
||||
cameras = discover_cameras()
|
||||
log.info("discovered %d cameras:", len(cameras))
|
||||
@@ -138,41 +427,101 @@ def main() -> int:
|
||||
c.usb_port or "-")
|
||||
|
||||
cameras = select_cameras(cameras, cfg.cameras)
|
||||
if cfg.cameras == ["all"]:
|
||||
rally = discover_rally()
|
||||
if rally is not None:
|
||||
cameras = list(cameras) + [rally]
|
||||
log.info(" %s video=%s audio=hw:%s usb=%s",
|
||||
rally.identity, rally.video_device,
|
||||
rally.audio_card if rally.audio_card is not None else "-",
|
||||
rally.usb_port or "-")
|
||||
if not cameras:
|
||||
log.error("no cameras match --cameras %s", args.cameras)
|
||||
return 2
|
||||
|
||||
fleet = fleet_cameras(cameras)
|
||||
daemon = None
|
||||
capture = None
|
||||
audio = None
|
||||
encode = None
|
||||
rally_proc = None
|
||||
portrait = None
|
||||
if cfg.enhance_daemon and cfg.enhance_mode != "off":
|
||||
try:
|
||||
daemon = spawn_enhance_daemon(cfg)
|
||||
except Exception as exc:
|
||||
log.error("enhance daemon failed: %s", exc)
|
||||
return 2
|
||||
if cfg.capture_daemon:
|
||||
try:
|
||||
capture = spawn_capture_daemon(cfg, fleet)
|
||||
except Exception as exc:
|
||||
log.error("capture daemon failed: %s", exc)
|
||||
stop_proc(daemon)
|
||||
return 2
|
||||
if cfg.audio_daemon:
|
||||
try:
|
||||
audio = spawn_audio_daemon(cfg, fleet)
|
||||
except Exception as exc:
|
||||
log.error("audio daemon failed: %s", exc)
|
||||
stop_proc(capture)
|
||||
stop_proc(daemon)
|
||||
return 2
|
||||
rally_cam = next((c for c in cameras if c.identity == "rally"), None)
|
||||
if rally_cam is not None:
|
||||
try:
|
||||
rally_proc = spawn_rally_follow(cfg, rally_cam)
|
||||
shm = Path(cfg.capture_shm_dir) / "rally.i420"
|
||||
from frame_shm import wait_for_ready
|
||||
if not wait_for_ready(str(shm), timeout=20):
|
||||
raise RuntimeError(f"rally follow did not create {shm}")
|
||||
except Exception as exc:
|
||||
log.error("rally follow failed: %s", exc)
|
||||
stop_proc(rally_proc)
|
||||
stop_proc(audio)
|
||||
stop_proc(capture)
|
||||
stop_proc(daemon)
|
||||
return 2
|
||||
portrait_dir = ""
|
||||
if cfg.portrait_blur and cfg.capture_daemon:
|
||||
try:
|
||||
portrait = spawn_portrait_daemon(cfg, fleet)
|
||||
portrait_dir = cfg.portrait_shm_dir
|
||||
except Exception as exc:
|
||||
log.error("portrait daemon failed, encode uses raw: %s", exc)
|
||||
stop_proc(portrait)
|
||||
portrait = None
|
||||
portrait_dir = ""
|
||||
if cfg.encode_daemon and cfg.capture_daemon:
|
||||
try:
|
||||
encode = spawn_encode_daemon(cfg, cameras, portrait_dir=portrait_dir)
|
||||
except Exception as exc:
|
||||
log.error("encode daemon failed: %s", exc)
|
||||
stop_proc(portrait)
|
||||
stop_proc(rally_proc)
|
||||
stop_proc(audio)
|
||||
stop_proc(capture)
|
||||
stop_proc(daemon)
|
||||
return 2
|
||||
|
||||
use_pub = cfg.single_publisher and cfg.capture_daemon
|
||||
sup = Supervisor(cfg, cameras)
|
||||
for cam in cameras:
|
||||
sup.spawn(cam)
|
||||
log.info("spawned %d workers", len(cameras))
|
||||
pub = None
|
||||
if use_pub:
|
||||
pub = spawn_publisher(cfg, cameras)
|
||||
log.info("spawned single publisher for %d cameras", len(cameras))
|
||||
else:
|
||||
for cam in cameras:
|
||||
sup.spawn(cam)
|
||||
log.info("spawned %d workers", len(cameras))
|
||||
|
||||
if args.once:
|
||||
deadline = time.monotonic() + cfg.publish_timeout_s
|
||||
while time.monotonic() < deadline:
|
||||
sup.pump()
|
||||
sup.reap()
|
||||
time.sleep(0.5)
|
||||
# final report
|
||||
up = [i for i, p in sup.procs.items() if p.poll() is None]
|
||||
log.info("STATUS: %d/%d workers alive", len(up), len(sup.procs))
|
||||
for ident in up:
|
||||
pass
|
||||
# stop everything
|
||||
for p in sup.procs.values():
|
||||
p.send_signal(signal.SIGTERM)
|
||||
for p in sup.procs.values():
|
||||
def _shutdown():
|
||||
if pub is not None and pub.poll() is None:
|
||||
pub.send_signal(signal.SIGTERM)
|
||||
with contextlib.suppress(Exception):
|
||||
p.wait(timeout=10)
|
||||
return 0 if len(up) == len(sup.procs) else 1
|
||||
|
||||
try:
|
||||
while True:
|
||||
sup.pump()
|
||||
sup.reap()
|
||||
time.sleep(0.2)
|
||||
except KeyboardInterrupt:
|
||||
log.info("shutting down ...")
|
||||
pub.wait(timeout=10)
|
||||
if pub.poll() is None:
|
||||
pub.kill()
|
||||
for p in sup.procs.values():
|
||||
if p.poll() is None:
|
||||
p.send_signal(signal.SIGTERM)
|
||||
@@ -182,6 +531,57 @@ def main() -> int:
|
||||
for p in sup.procs.values():
|
||||
if p.poll() is None:
|
||||
p.kill()
|
||||
stop_proc(daemon)
|
||||
stop_proc(capture)
|
||||
stop_proc(audio)
|
||||
stop_proc(encode)
|
||||
stop_proc(portrait)
|
||||
stop_proc(rally_proc)
|
||||
|
||||
def _pump_pub():
|
||||
nonlocal pub
|
||||
if pub is None:
|
||||
return
|
||||
line = pub.stdout.readline() if pub.stdout else ""
|
||||
if line:
|
||||
sys.stdout.write(line)
|
||||
sys.stdout.flush()
|
||||
rc = pub.poll()
|
||||
if rc is not None:
|
||||
log.warning("publisher exited rc=%s; restarting", rc)
|
||||
time.sleep(1.0)
|
||||
pub = spawn_publisher(cfg, cameras)
|
||||
|
||||
if args.once:
|
||||
deadline = time.monotonic() + cfg.publish_timeout_s
|
||||
while time.monotonic() < deadline:
|
||||
if use_pub:
|
||||
_pump_pub()
|
||||
else:
|
||||
sup.pump()
|
||||
sup.reap()
|
||||
time.sleep(0.5)
|
||||
if use_pub:
|
||||
alive = pub is not None and pub.poll() is None
|
||||
log.info("STATUS: publisher %s", "UP" if alive else "DOWN")
|
||||
_shutdown()
|
||||
return 0 if alive else 1
|
||||
up = [i for i, p in sup.procs.items() if p.poll() is None]
|
||||
log.info("STATUS: %d/%d workers alive", len(up), len(sup.procs))
|
||||
_shutdown()
|
||||
return 0 if len(up) == len(sup.procs) else 1
|
||||
|
||||
try:
|
||||
while True:
|
||||
if use_pub:
|
||||
_pump_pub()
|
||||
else:
|
||||
sup.pump()
|
||||
sup.reap()
|
||||
time.sleep(0.2)
|
||||
except KeyboardInterrupt:
|
||||
log.info("shutting down ...")
|
||||
_shutdown()
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
+454
-54
@@ -14,9 +14,11 @@ from __future__ import annotations
|
||||
import argparse
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import signal
|
||||
import sys
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
@@ -26,19 +28,37 @@ from config import load_config
|
||||
from display_grid import (
|
||||
CameraGrid,
|
||||
ROLES,
|
||||
active_speaker_ids,
|
||||
bgra_from_livekit,
|
||||
camera_ids,
|
||||
is_camera_identity,
|
||||
collapse_bleed,
|
||||
confirm_speakers,
|
||||
describe_livekit_error,
|
||||
format_livekit_status,
|
||||
hold_speakers,
|
||||
placeholder,
|
||||
select_grid_speakers,
|
||||
update_noise_floor,
|
||||
)
|
||||
from drm_outputs import DrmOutput, format_outputs, list_outputs, select_outputs
|
||||
from gst_sink import KmsPipeline
|
||||
from gst_sink import KmsPipeline, VaGridPipeline
|
||||
from hand_raise import HandRaiseMonitor
|
||||
from latency import video_stream_kwargs
|
||||
from participant_tags import want_display_video
|
||||
from tokens import make_viewer_token
|
||||
|
||||
log = logging.getLogger("cameras.display")
|
||||
|
||||
H264_DIR = "/run/livekit-cameras/h264"
|
||||
|
||||
|
||||
def _h264_fifo_paths(identities: list[str]) -> list[str] | None:
|
||||
# Sidecar H.264 encode removed: grid tiles come from LiveKit I420.
|
||||
del identities
|
||||
return None
|
||||
|
||||
_PATTERNS = ("smpte", "ball", "snow")
|
||||
_PLACE_W, _PLACE_H = 1280, 720
|
||||
_PLACE_W, _PLACE_H = 1920, 1080
|
||||
|
||||
|
||||
def _public_url(cfg) -> str:
|
||||
@@ -71,6 +91,69 @@ def _bind_roles(roles: list[str], connectors: list[str]) -> list[tuple[str, DrmO
|
||||
return slots
|
||||
|
||||
|
||||
class HopLevels:
|
||||
"""Latest hop RMS + idle floor per grid identity (does not mute publish)."""
|
||||
|
||||
def __init__(self, shm_dir: str, identities: list[str], rise_db: float = 6.0) -> None:
|
||||
from hop_shm import HopReader
|
||||
from audio_gate import hop_mono_int16, rms_dbfs
|
||||
|
||||
self._mono = hop_mono_int16
|
||||
self._rms = rms_dbfs
|
||||
self.rise_db = float(rise_db)
|
||||
self._db = {i: -120.0 for i in identities}
|
||||
self._wave: dict[str, np.ndarray | None] = {i: None for i in identities}
|
||||
self._floor: dict[str, float | None] = {i: None for i in identities}
|
||||
self._n = {i: 0 for i in identities}
|
||||
self._readers = {
|
||||
i: HopReader(os.path.join(shm_dir, f"{i}.pcm")) for i in identities
|
||||
}
|
||||
|
||||
def poll(self, speaking: set[str] | None = None) -> None:
|
||||
hot = speaking or set()
|
||||
for ident, r in self._readers.items():
|
||||
blob = None
|
||||
while True:
|
||||
nxt = r.read()
|
||||
if nxt is None:
|
||||
break
|
||||
blob = nxt
|
||||
if blob is None:
|
||||
continue
|
||||
pcm = np.frombuffer(blob, dtype=np.int16)
|
||||
if pcm.size % 2 == 0:
|
||||
pcm = pcm.reshape(-1, 2)
|
||||
db = self._rms(self._mono(pcm))
|
||||
self._db[ident] = db
|
||||
self._wave[ident] = np.asarray(self._mono(pcm), dtype=np.float32)
|
||||
fl = self._floor[ident]
|
||||
self._n[ident] += 1
|
||||
if fl is None:
|
||||
self._floor[ident] = db
|
||||
else:
|
||||
self._floor[ident] = update_noise_floor(
|
||||
fl, db, speaking=ident in hot, rise_db=self.rise_db)
|
||||
|
||||
def levels(self, identities: list[str]) -> list[tuple[str, float]]:
|
||||
return [
|
||||
(i, self._db.get(i, -120.0))
|
||||
for i in identities
|
||||
if i in self._readers and self._n.get(i, 0) >= 3
|
||||
]
|
||||
|
||||
def waves(self, identities: list[str]) -> dict[str, np.ndarray]:
|
||||
out: dict[str, np.ndarray] = {}
|
||||
for i in identities:
|
||||
w = self._wave.get(i)
|
||||
if w is not None:
|
||||
out[i] = w
|
||||
return out
|
||||
|
||||
@property
|
||||
def floors(self) -> dict[str, float]:
|
||||
return {i: fl for i, fl in self._floor.items() if fl is not None}
|
||||
|
||||
|
||||
class Wall:
|
||||
def __init__(self, cfg, slots: list[tuple[str, DrmOutput]]):
|
||||
self.cfg = cfg
|
||||
@@ -82,13 +165,23 @@ class Wall:
|
||||
height=cfg.display_grid_height,
|
||||
cols=cfg.display_grid_cols,
|
||||
rows=cfg.display_grid_rows,
|
||||
identities=camera_ids(n, prefix),
|
||||
identities=[],
|
||||
)
|
||||
self.speaker_camera = (cfg.display_speaker_camera or "cam-01").strip()
|
||||
self.grid.set_status(format_livekit_status(url=cfg.url, state="connecting"))
|
||||
self._retry_s = 5.0
|
||||
# LiveKit FFI retries ~3 times (~12s) on ENETUNREACH; stay above that
|
||||
# so the overlay can show "host unreachable" instead of a bare timeout.
|
||||
self._connect_timeout_s = 20.0
|
||||
self.speaker_camera = (cfg.display_speaker_camera or "rally").strip()
|
||||
self.active_speaker: Optional[str] = None
|
||||
self._last_speakers: tuple[str, ...] = ()
|
||||
self._speaker_frame: Optional[np.ndarray] = None
|
||||
self._tasks: dict[str, asyncio.Task] = {}
|
||||
self._stop = asyncio.Event()
|
||||
self._pump_pool = ThreadPoolExecutor(
|
||||
max_workers=max(1, int(cfg.display_pump_workers)),
|
||||
thread_name_prefix="disp-pump",
|
||||
)
|
||||
# Screenshare is a stub until production LiveKit is online.
|
||||
self._share_placeholder = placeholder(_PLACE_W, _PLACE_H, [
|
||||
"Screen share",
|
||||
@@ -98,22 +191,162 @@ class Wall:
|
||||
cam = self.speaker_camera
|
||||
self._speaker_placeholder = placeholder(_PLACE_W, _PLACE_H, [
|
||||
"Speaker camera",
|
||||
"Logitech Rally",
|
||||
cam if cam.lower() != "active" else "(following active speaker)",
|
||||
"not connected",
|
||||
])
|
||||
self.va_grid: Optional[VaGridPipeline] = None
|
||||
self.hand_raise = HandRaiseMonitor(
|
||||
enabled=cfg.display_hand_raise,
|
||||
hz=cfg.display_hand_raise_hz,
|
||||
hold_s=cfg.display_hand_raise_hold_s,
|
||||
on_change=self.grid.set_highlight,
|
||||
model=cfg.display_hand_raise_model,
|
||||
)
|
||||
self._hop_levels = HopLevels(
|
||||
cfg.audio_shm_dir, self.grid.identities,
|
||||
rise_db=cfg.display_speak_rise_db)
|
||||
self._speak_cur: list[str] = []
|
||||
self._speak_peak = -120.0
|
||||
self._speak_hold_until = 0.0
|
||||
self._speak_counts: dict[str, int] = {}
|
||||
|
||||
def _refresh_speakers(self) -> None:
|
||||
"""Hop RMS is the detector; LiveKit is a hint, not a gate."""
|
||||
tiles = list(self.grid.identities)
|
||||
ranked = self._hop_levels.levels(tiles)
|
||||
picked = select_grid_speakers(
|
||||
ranked,
|
||||
min_db=self.cfg.display_speak_min_db,
|
||||
margin_db=self.cfg.display_speak_margin_db,
|
||||
floors=self._hop_levels.floors,
|
||||
rise_db=self.cfg.display_speak_rise_db,
|
||||
)
|
||||
picked = collapse_bleed(
|
||||
picked,
|
||||
self._hop_levels.waves(picked),
|
||||
dict(ranked),
|
||||
corr_min=self.cfg.display_speak_corr,
|
||||
)
|
||||
picked, self._speak_counts = confirm_speakers(
|
||||
self._speak_counts, picked,
|
||||
need=self.cfg.display_speak_confirm,
|
||||
already=self._speak_cur,
|
||||
)
|
||||
new_peak = max((db for i, db in ranked if i in picked), default=-120.0)
|
||||
now = time.monotonic()
|
||||
chrome, self._speak_hold_until = hold_speakers(
|
||||
self._speak_cur, picked,
|
||||
prev_peak=self._speak_peak, new_peak=new_peak,
|
||||
now=now, hold_until=self._speak_hold_until,
|
||||
hold_s=self.cfg.display_speak_hold_s,
|
||||
)
|
||||
if chrome == picked:
|
||||
self._speak_peak = new_peak
|
||||
self._speak_cur = chrome
|
||||
self.grid.set_speakers(chrome)
|
||||
key = tuple(chrome)
|
||||
if key != self._last_speakers:
|
||||
self._last_speakers = key
|
||||
log.info("active-speakers %s", ",".join(chrome) if chrome else "-")
|
||||
|
||||
def speaker_identity(self) -> Optional[str]:
|
||||
if self.speaker_camera.lower() == "active":
|
||||
return self.active_speaker
|
||||
return self.speaker_camera
|
||||
|
||||
def _push_sink(self, role: str, data: bytes, width: int, height: int, fps: int) -> None:
|
||||
def _set_lk_status(self, state: str, detail: str = "", retry_s: float | None = None) -> None:
|
||||
lines = format_livekit_status(
|
||||
url=self.cfg.url, state=state, detail=detail, retry_s=retry_s,
|
||||
room=self.cfg.room)
|
||||
if state == "connected":
|
||||
if self.grid.identities:
|
||||
self.grid.set_status(None)
|
||||
else:
|
||||
self.grid.set_status(["NO LIVE CAMERAS", f"room {self.cfg.room}"])
|
||||
return
|
||||
self.grid.set_status(lines)
|
||||
|
||||
def _seat_participant(self, participant) -> None:
|
||||
ident = getattr(participant, "identity", "") or ""
|
||||
if not ident or not self._want_video(participant):
|
||||
return
|
||||
if ident == self.speaker_identity():
|
||||
return
|
||||
if not self.grid.add_identity(ident):
|
||||
log.warning("grid full, skip %s", ident)
|
||||
return
|
||||
if self.grid.identities:
|
||||
self.grid.set_status(None)
|
||||
|
||||
def _unseat_participant(self, identity: str) -> None:
|
||||
ident = (identity or "").strip()
|
||||
if not ident:
|
||||
return
|
||||
self.grid.remove_identity(ident)
|
||||
self.grid.clear(ident)
|
||||
if ident == self.speaker_identity():
|
||||
self._speaker_frame = None
|
||||
# Keep mosaic if other tiles remain; otherwise show empty-room text
|
||||
# only while the SFU session is still up (status overlay owns errors).
|
||||
if not self.grid.identities and not self.grid._status_lines:
|
||||
self.grid.set_status(["NO LIVE CAMERAS", f"room {self.cfg.room}"])
|
||||
|
||||
def _tile_index(self, identity: str) -> Optional[int]:
|
||||
try:
|
||||
return self.grid.identities.index(identity)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
def _start_va_grid(self) -> None:
|
||||
# vacompositor prerolls one HDMI frame then never flips (crtc CRC unique=1).
|
||||
# CPU mosaic through appsrc+kmssink is the path that actually presents.
|
||||
if os.environ.get("DISPLAY_VA_GRID", "0").strip() != "1":
|
||||
log.info("va-grid skipped; CPU mosaic -> kmssink")
|
||||
return
|
||||
sink = self.slots.get("grid")
|
||||
if sink is None or not sink.out.connected:
|
||||
return
|
||||
self.va_grid = VaGridPipeline(sink.out)
|
||||
self.va_grid.start(
|
||||
self.grid,
|
||||
ingest_w=max(16, int(self.cfg.width)),
|
||||
ingest_h=max(16, int(self.cfg.height)),
|
||||
fps=max(1, int(self.cfg.fps) or 15),
|
||||
queue_buffers=self.cfg.display_gst_queue_buffers,
|
||||
h264_paths=_h264_fifo_paths(self.grid.identities),
|
||||
)
|
||||
if not self.va_grid.alive:
|
||||
log.warning("va-grid failed to start; falling back to CPU mosaic")
|
||||
self.va_grid = None
|
||||
|
||||
def _want_video(self, participant) -> bool:
|
||||
return want_display_video(
|
||||
identity=getattr(participant, "identity", "") or "",
|
||||
attributes=getattr(participant, "attributes", None) or {},
|
||||
metadata=getattr(participant, "metadata", "") or "",
|
||||
hide_tags=self.cfg.display_hide_tags,
|
||||
participant_prefix=self.cfg.participant_prefix or "cam",
|
||||
display_identity=self.cfg.display_identity,
|
||||
speaker_identity=self.speaker_identity() or "",
|
||||
)
|
||||
|
||||
def _push_sink(self, role: str, data: bytes, width: int, height: int, fps: int,
|
||||
pixel_format: str = "bgra") -> None:
|
||||
sink = self.slots.get(role)
|
||||
if sink is None or not sink.out.connected:
|
||||
return
|
||||
if not sink.alive or sink.width != width or sink.height != height:
|
||||
sink.start_appsrc(width, height, fps)
|
||||
if (not sink.alive or sink.width != width or sink.height != height
|
||||
or getattr(sink, "pixel_format", "bgra") != pixel_format):
|
||||
sink.start_appsrc(
|
||||
width, height, fps,
|
||||
queue_buffers=self.cfg.display_gst_queue_buffers,
|
||||
pixel_format=pixel_format)
|
||||
if not sink.push(data):
|
||||
sink.start_appsrc(width, height, fps)
|
||||
sink.start_appsrc(
|
||||
width, height, fps,
|
||||
queue_buffers=self.cfg.display_gst_queue_buffers,
|
||||
pixel_format=pixel_format)
|
||||
sink.push(data)
|
||||
err = sink.gst_errors()
|
||||
if err:
|
||||
@@ -121,26 +354,56 @@ class Wall:
|
||||
log.warning("%s gst: %s", role, line)
|
||||
|
||||
async def _output_loops(self) -> None:
|
||||
self._start_va_grid()
|
||||
grid_fps = max(1, int(self.cfg.display_grid_fps))
|
||||
grid_period = 1.0 / grid_fps
|
||||
spk_period = 1.0 / max(1, int(self.cfg.fps))
|
||||
next_grid = time.monotonic()
|
||||
next_spk = time.monotonic()
|
||||
next_hop = time.monotonic()
|
||||
while not self._stop.is_set():
|
||||
now = time.monotonic()
|
||||
if now >= next_hop:
|
||||
self._hop_levels.poll(speaking=set(self._speak_cur))
|
||||
self._refresh_speakers()
|
||||
next_hop = now + 0.2
|
||||
if now >= next_grid:
|
||||
self._push_sink(
|
||||
"grid", self.grid.render(),
|
||||
self.grid.width, self.grid.height, grid_fps,
|
||||
)
|
||||
if self.va_grid is not None:
|
||||
if not self.va_grid.alive:
|
||||
log.warning("va-grid died; falling back to CPU mosaic")
|
||||
self.va_grid.stop()
|
||||
self.va_grid = None
|
||||
else:
|
||||
err = self.va_grid.gst_errors()
|
||||
if err:
|
||||
for line in err.strip().splitlines()[-6:]:
|
||||
log.warning("va-grid gst: %s", line)
|
||||
if self.va_grid is None:
|
||||
i420 = self.cfg.display_video_format in ("i420", "yuv420p")
|
||||
if i420:
|
||||
self._push_sink(
|
||||
"grid", self.grid.render_i420(),
|
||||
self.grid.width, self.grid.height, grid_fps,
|
||||
pixel_format="i420",
|
||||
)
|
||||
else:
|
||||
self._push_sink(
|
||||
"grid", self.grid.render(),
|
||||
self.grid.width, self.grid.height, grid_fps,
|
||||
)
|
||||
next_grid = now + grid_period
|
||||
if now >= next_spk:
|
||||
spk = self._speaker_frame
|
||||
arrival = self.cfg.display_speaker_push == "arrival"
|
||||
if spk is None:
|
||||
img = self._speaker_placeholder
|
||||
else:
|
||||
img = spk
|
||||
self._push_sink("speaker", img.tobytes(), img.shape[1], img.shape[0], self.cfg.fps)
|
||||
self._push_sink(
|
||||
"speaker", img.tobytes(), img.shape[1], img.shape[0],
|
||||
self.cfg.fps)
|
||||
elif not arrival:
|
||||
self._push_sink(
|
||||
"speaker", spk.tobytes(), spk.shape[1], spk.shape[0],
|
||||
self.cfg.fps)
|
||||
share = self._share_placeholder
|
||||
self._push_sink("screenshare", share.tobytes(), share.shape[1], share.shape[0], self.cfg.fps)
|
||||
next_spk = now + spk_period
|
||||
@@ -149,23 +412,60 @@ class Wall:
|
||||
async def _pump_camera(self, track, identity: str) -> None:
|
||||
import livekit.rtc as rtc
|
||||
|
||||
stream = rtc.VideoStream(track)
|
||||
stream_kw = video_stream_kwargs(
|
||||
self.cfg.display_video_capacity, self.cfg.display_video_format)
|
||||
if stream_kw.get("format") == "bgra":
|
||||
stream_kw["format"] = rtc.VideoBufferType.BGRA
|
||||
elif stream_kw.get("format") in ("i420", "yuv420p"):
|
||||
stream_kw["format"] = rtc.VideoBufferType.I420
|
||||
stream = rtc.VideoStream(track, **stream_kw)
|
||||
want_bgra = self.cfg.display_video_format == "bgra"
|
||||
use_i420 = self.cfg.display_video_format in ("i420", "yuv420p")
|
||||
loop = asyncio.get_running_loop()
|
||||
try:
|
||||
async for event in stream:
|
||||
frame = event.frame
|
||||
if frame.width <= 0 or frame.height <= 0:
|
||||
continue
|
||||
try:
|
||||
converted = frame.convert(rtc.VideoBufferType.BGRA)
|
||||
except Exception as exc:
|
||||
log.warning("%s: convert failed: %s", identity, exc)
|
||||
if (self.va_grid is not None
|
||||
and getattr(self.va_grid, "codec", "") == "i420"):
|
||||
idx = self._tile_index(identity)
|
||||
if (idx is not None
|
||||
and frame.width == self.va_grid.ingest_w
|
||||
and frame.height == self.va_grid.ingest_h):
|
||||
self.va_grid.push_tile(idx, frame.data)
|
||||
is_spk = identity == self.speaker_identity()
|
||||
cpu_grid = self.va_grid is None
|
||||
if cpu_grid and use_i420 and not is_spk:
|
||||
blob = bytes(frame.data)
|
||||
await loop.run_in_executor(
|
||||
self._pump_pool,
|
||||
self.grid.set_frame_i420,
|
||||
identity, blob, frame.width, frame.height)
|
||||
self.hand_raise.offer(identity, blob, frame.width, frame.height)
|
||||
continue
|
||||
bgra = bgra_from_livekit(converted.data, converted.width, converted.height)
|
||||
self.grid.set_frame(identity, bgra)
|
||||
if identity == self.speaker_identity():
|
||||
self._speaker_frame = bgra
|
||||
if cpu_grid or is_spk:
|
||||
try:
|
||||
if frame.type == rtc.VideoBufferType.BGRA:
|
||||
converted = frame
|
||||
else:
|
||||
converted = frame.convert(rtc.VideoBufferType.BGRA)
|
||||
except Exception as exc:
|
||||
log.warning("%s: convert failed: %s", identity, exc)
|
||||
continue
|
||||
bgra = bgra_from_livekit(
|
||||
converted.data, converted.width, converted.height)
|
||||
if cpu_grid:
|
||||
self.grid.set_frame(identity, bgra)
|
||||
if is_spk:
|
||||
self._speaker_frame = bgra
|
||||
if self.cfg.display_speaker_push == "arrival":
|
||||
self._push_sink(
|
||||
"speaker", bgra.tobytes(),
|
||||
bgra.shape[1], bgra.shape[0], self.cfg.fps)
|
||||
finally:
|
||||
await stream.aclose()
|
||||
self.hand_raise.forget(identity)
|
||||
self.grid.clear(identity)
|
||||
if identity == self.speaker_identity():
|
||||
self._speaker_frame = None
|
||||
@@ -175,27 +475,29 @@ class Wall:
|
||||
|
||||
def attach(self, track, publication, participant, rtc) -> None:
|
||||
ident = participant.identity
|
||||
if ident == self.cfg.display_identity:
|
||||
if not self._want_video(participant):
|
||||
return
|
||||
is_spk = ident == self.speaker_identity()
|
||||
if not is_spk:
|
||||
self._seat_participant(participant)
|
||||
if ident not in self.grid.identities:
|
||||
return
|
||||
key = self._track_key(ident, track)
|
||||
old = self._tasks.pop(key, None)
|
||||
if old:
|
||||
old.cancel()
|
||||
# Screenshare subscribe/render is stubbed until production LiveKit.
|
||||
if is_camera_identity(ident, self.cfg.participant_prefix):
|
||||
self._tasks[key] = asyncio.create_task(
|
||||
self._pump_camera(track, ident), name=f"cam-{ident}")
|
||||
self._tasks[key] = asyncio.create_task(
|
||||
self._pump_camera(track, ident), name=f"cam-{ident}")
|
||||
|
||||
def subscribe_if_wanted(self, participant, rtc) -> None:
|
||||
ident = participant.identity
|
||||
if ident == self.cfg.display_identity:
|
||||
if not self._want_video(participant):
|
||||
return
|
||||
cam = is_camera_identity(ident, self.cfg.participant_prefix)
|
||||
for pub in participant.track_publications.values():
|
||||
if pub.kind != rtc.TrackKind.KIND_VIDEO:
|
||||
continue
|
||||
want = cam
|
||||
if want and not pub.subscribed:
|
||||
if not pub.subscribed:
|
||||
log.info("subscribe %s from %s (source=%s)", pub.sid, ident, pub.source)
|
||||
pub.set_subscribed(True)
|
||||
|
||||
@@ -206,9 +508,69 @@ class Wall:
|
||||
task.cancel()
|
||||
|
||||
async def run(self) -> int:
|
||||
loops = asyncio.create_task(self._output_loops(), name="output-loops")
|
||||
loop = asyncio.get_running_loop()
|
||||
for sig in (signal.SIGINT, signal.SIGTERM):
|
||||
try:
|
||||
loop.add_signal_handler(sig, self._stop.set)
|
||||
except NotImplementedError:
|
||||
pass
|
||||
self.hand_raise.start()
|
||||
self._set_lk_status("connecting")
|
||||
log.info(
|
||||
"wall up (livekit optional): grid=%dx%d speaker_camera=%s "
|
||||
"video_capacity=%d format=%s gst_queue=%d speaker_push=%s "
|
||||
"pump_workers=%d hide_tags=%s hand_raise=%s url=%s",
|
||||
self.cfg.display_grid_cols, self.cfg.display_grid_rows, self.speaker_camera,
|
||||
self.cfg.display_video_capacity,
|
||||
self.cfg.display_video_format or "native",
|
||||
self.cfg.display_gst_queue_buffers,
|
||||
self.cfg.display_speaker_push,
|
||||
self.cfg.display_pump_workers,
|
||||
",".join(self.cfg.display_hide_tags) or "off",
|
||||
"on" if self.hand_raise.enabled else "off",
|
||||
self.cfg.url,
|
||||
)
|
||||
try:
|
||||
while not self._stop.is_set():
|
||||
try:
|
||||
await self._session()
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
detail = describe_livekit_error(exc)
|
||||
log.error("livekit %s: %s", detail, exc or type(exc).__name__)
|
||||
self.grid.set_identities([])
|
||||
self._speaker_frame = None
|
||||
self._set_lk_status(
|
||||
"disconnected", detail=detail, retry_s=self._retry_s)
|
||||
if self._stop.is_set():
|
||||
break
|
||||
try:
|
||||
await asyncio.wait_for(self._stop.wait(), timeout=self._retry_s)
|
||||
except asyncio.TimeoutError:
|
||||
pass
|
||||
if not self._stop.is_set():
|
||||
self._set_lk_status("connecting")
|
||||
finally:
|
||||
loops.cancel()
|
||||
for t in self._tasks.values():
|
||||
t.cancel()
|
||||
self._tasks.clear()
|
||||
for sink in self.slots.values():
|
||||
sink.stop()
|
||||
if self.va_grid is not None:
|
||||
self.va_grid.stop()
|
||||
self.va_grid = None
|
||||
self._pump_pool.shutdown(wait=False, cancel_futures=True)
|
||||
self.hand_raise.stop()
|
||||
return 0
|
||||
|
||||
async def _session(self) -> None:
|
||||
import livekit.rtc as rtc
|
||||
|
||||
room = rtc.Room()
|
||||
session_done = asyncio.Event()
|
||||
|
||||
@room.on("track_subscribed")
|
||||
def _on_sub(track, publication, participant):
|
||||
@@ -227,48 +589,80 @@ class Wall:
|
||||
@room.on("participant_connected")
|
||||
def _on_pc(participant):
|
||||
log.info("participant joined: %s", participant.identity)
|
||||
self._seat_participant(participant)
|
||||
self.subscribe_if_wanted(participant, rtc)
|
||||
|
||||
@room.on("participant_disconnected")
|
||||
def _on_pd(participant):
|
||||
ident = getattr(participant, "identity", "") or ""
|
||||
log.info("participant left: %s", ident)
|
||||
self._unseat_participant(ident)
|
||||
|
||||
@room.on("disconnected")
|
||||
def _on_disc(*args):
|
||||
reason = ""
|
||||
if args:
|
||||
reason = str(args[0] or "")
|
||||
detail = describe_livekit_error(RuntimeError(reason)) if reason else "server closed the session"
|
||||
if reason:
|
||||
low = reason.lower()
|
||||
if "disconnect" in low or "close" in low:
|
||||
detail = reason[:96]
|
||||
log.warning("livekit disconnected: %s", reason or "no reason")
|
||||
self.grid.set_identities([])
|
||||
self._speaker_frame = None
|
||||
self._set_lk_status(
|
||||
"disconnected", detail=detail, retry_s=self._retry_s)
|
||||
session_done.set()
|
||||
|
||||
@room.on("active_speakers_changed")
|
||||
def _on_spk(speakers):
|
||||
cams = [
|
||||
p.identity for p in speakers
|
||||
if is_camera_identity(p.identity, self.cfg.participant_prefix)
|
||||
]
|
||||
cams = active_speaker_ids(speakers, want=self._want_video)
|
||||
self.active_speaker = cams[0] if cams else None
|
||||
self._hop_levels.poll(speaking=set(self._speak_cur))
|
||||
self._refresh_speakers()
|
||||
|
||||
token = make_viewer_token(
|
||||
self.cfg.api_key, self.cfg.api_secret, self.cfg.room, self.cfg.display_identity)
|
||||
log.info("connecting to %s room=%s as %s", self.cfg.url, self.cfg.room, self.cfg.display_identity)
|
||||
await room.connect(self.cfg.url, token, rtc.RoomOptions(auto_subscribe=False))
|
||||
self._set_lk_status("connecting")
|
||||
try:
|
||||
await asyncio.wait_for(
|
||||
room.connect(self.cfg.url, token, rtc.RoomOptions(auto_subscribe=False)),
|
||||
timeout=self._connect_timeout_s,
|
||||
)
|
||||
except Exception:
|
||||
try:
|
||||
await room.disconnect()
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
|
||||
self._set_lk_status("connected")
|
||||
for rp in room.remote_participants.values():
|
||||
self._seat_participant(rp)
|
||||
self.subscribe_if_wanted(rp, rtc)
|
||||
for pub in rp.track_publications.values():
|
||||
if pub.track and pub.track.kind == rtc.TrackKind.KIND_VIDEO:
|
||||
self.attach(pub.track, pub, rp, rtc)
|
||||
|
||||
log.info(
|
||||
"wall up: grid=%dx%d speaker_camera=%s screenshare=STUB",
|
||||
self.grid.cols, self.grid.rows, self.speaker_camera,
|
||||
"livekit connected: tiles=%d speaker_camera=%s",
|
||||
len(self.grid.identities), self.speaker_camera,
|
||||
)
|
||||
loops = asyncio.create_task(self._output_loops(), name="output-loops")
|
||||
loop = asyncio.get_running_loop()
|
||||
for sig in (signal.SIGINT, signal.SIGTERM):
|
||||
try:
|
||||
loop.add_signal_handler(sig, self._stop.set)
|
||||
except NotImplementedError:
|
||||
pass
|
||||
stop_task = asyncio.create_task(self._stop.wait())
|
||||
disc_task = asyncio.create_task(session_done.wait())
|
||||
try:
|
||||
await self._stop.wait()
|
||||
await asyncio.wait({stop_task, disc_task}, return_when=asyncio.FIRST_COMPLETED)
|
||||
finally:
|
||||
loops.cancel()
|
||||
stop_task.cancel()
|
||||
disc_task.cancel()
|
||||
for t in self._tasks.values():
|
||||
t.cancel()
|
||||
for sink in self.slots.values():
|
||||
sink.stop()
|
||||
await room.disconnect()
|
||||
return 0
|
||||
self._tasks.clear()
|
||||
try:
|
||||
await room.disconnect()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def run_test_pattern(slots: list[tuple[str, DrmOutput]]) -> int:
|
||||
@@ -326,6 +720,8 @@ def main(argv: Optional[list[str]] = None) -> int:
|
||||
|
||||
cfg = load_config()
|
||||
logging.getLogger().setLevel(getattr(logging, cfg.log_level, logging.INFO))
|
||||
if cfg.url or cfg.room:
|
||||
log.info("livekit destination url=%s room=%s", cfg.url, cfg.room)
|
||||
if args.speaker_camera:
|
||||
cfg.display_speaker_camera = args.speaker_camera
|
||||
roles = [p.strip() for p in args.roles.split(",") if p.strip()] or cfg.display_roles
|
||||
@@ -341,6 +737,10 @@ def main(argv: Optional[list[str]] = None) -> int:
|
||||
|
||||
if args.test_pattern:
|
||||
return run_test_pattern(slots)
|
||||
if not (cfg.url or "").startswith("wss://"):
|
||||
log.error("LIVEKIT_URL must be wss:// (refusing %s)", cfg.url)
|
||||
return 2
|
||||
log.info("livekit destination url=%s room=%s", cfg.url, cfg.room)
|
||||
return asyncio.run(Wall(cfg, slots).run())
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,356 @@
|
||||
"""One process: N LiveKit identities, shared video SHM + audio hop rings."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import signal
|
||||
import sys
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
|
||||
log = logging.getLogger("cameras.publisher")
|
||||
|
||||
|
||||
def _rss_anon_kb() -> int:
|
||||
try:
|
||||
with open("/proc/self/status", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
if line.startswith("RssAnon:"):
|
||||
return int(line.split()[1])
|
||||
except OSError:
|
||||
return 0
|
||||
return 0
|
||||
|
||||
|
||||
def _pub_opts(rtc, args, height: int = 0):
|
||||
from livekit.rtc import TrackPublishOptions, VideoEncoding, VideoCodec, TrackSource
|
||||
from livekit.rtc._proto.room_pb2 import (
|
||||
ENCODER_BACKEND_AUTO,
|
||||
ENCODER_BACKEND_SOFTWARE,
|
||||
ENCODER_BACKEND_HARDWARE,
|
||||
ENCODER_BACKEND_NVENC,
|
||||
ENCODER_BACKEND_VAAPI,
|
||||
ENCODER_BACKEND_VIDEOTOOLBOX,
|
||||
DEGRADATION_PREFERENCE_MAINTAIN_RESOLUTION,
|
||||
)
|
||||
encoder_map = {
|
||||
"auto": ENCODER_BACKEND_AUTO,
|
||||
"software": ENCODER_BACKEND_SOFTWARE,
|
||||
"hardware": ENCODER_BACKEND_HARDWARE,
|
||||
"nvenc": ENCODER_BACKEND_NVENC,
|
||||
"vaapi": ENCODER_BACKEND_VAAPI,
|
||||
"videotoolbox": ENCODER_BACKEND_VIDEOTOOLBOX,
|
||||
}
|
||||
codec_map = {
|
||||
"vp8": VideoCodec.VP8,
|
||||
"h264": VideoCodec.H264,
|
||||
"av1": VideoCodec.AV1,
|
||||
"vp9": VideoCodec.VP9,
|
||||
"h265": VideoCodec.H265,
|
||||
}
|
||||
encoder = encoder_map.get(args.video_encoder, ENCODER_BACKEND_AUTO)
|
||||
codec = codec_map.get(args.video_codec, VideoCodec.H264)
|
||||
br = int(args.bitrate)
|
||||
if int(height) >= 720:
|
||||
br = max(br, int(os.environ.get("VIDEO_RALLY_BITRATE", "20000000")))
|
||||
video = TrackPublishOptions(
|
||||
source=TrackSource.SOURCE_CAMERA,
|
||||
simulcast=False,
|
||||
video_codec=codec,
|
||||
video_encoding=VideoEncoding(
|
||||
max_bitrate=br,
|
||||
max_framerate=int(args.fps),
|
||||
),
|
||||
video_encoder=encoder,
|
||||
degradation_preference=DEGRADATION_PREFERENCE_MAINTAIN_RESOLUTION,
|
||||
)
|
||||
audio = TrackPublishOptions(source=TrackSource.SOURCE_MICROPHONE, dtx=False)
|
||||
return video, audio
|
||||
|
||||
|
||||
async def publish_one(rtc, cam: dict, args, stop: asyncio.Event,
|
||||
connect_gate: asyncio.Semaphore, enc_state: dict) -> None:
|
||||
from audio_cleanup import AudioCleaner
|
||||
from frame_shm import LatestFrameReader, frame_bytes
|
||||
from hop_shm import HopReader
|
||||
from audio_gate import make_gate
|
||||
from latency import (
|
||||
audio_queue_size_ms,
|
||||
effective_enhance_mode,
|
||||
next_video_capture,
|
||||
should_recycle_encoders,
|
||||
video_capture_gate_acquire,
|
||||
video_capture_gate_release,
|
||||
)
|
||||
|
||||
identity = cam["identity"]
|
||||
clog = logging.getLogger(f"cameras.publisher.{identity}")
|
||||
w = int(cam.get("width") or args.width)
|
||||
h = int(cam.get("height") or args.height)
|
||||
if args.video_codec in ("av1", "h264"):
|
||||
h = h - (h % 16)
|
||||
w = w - (w % 16)
|
||||
enhance_mode = effective_enhance_mode(
|
||||
args.enhance_mode, identity, args.enhance_speaker_only,
|
||||
args.speaker_camera)
|
||||
beamform = args.beamform_mode
|
||||
if args.enhance_speaker_only and enhance_mode == "off":
|
||||
beamform = "off"
|
||||
|
||||
cleaner = AudioCleaner(
|
||||
mode=enhance_mode,
|
||||
model_source=args.enhance_model,
|
||||
sample_rate=args.audio_rate,
|
||||
chunk_s=args.enhance_chunk_s,
|
||||
hop_s=args.enhance_hop_s,
|
||||
beamform=beamform,
|
||||
torch_num_threads=args.torch_num_threads,
|
||||
tdoa_every=args.tdoa_every,
|
||||
enhance_infer=args.enhance_infer,
|
||||
enhance_socket=args.enhance_socket,
|
||||
identity=identity,
|
||||
gate=make_gate(
|
||||
identity,
|
||||
args.speaker_camera,
|
||||
bool(args.audio_gate),
|
||||
args.audio_gate_open_db,
|
||||
args.audio_gate_close_db,
|
||||
args.audio_gate_hold_s,
|
||||
args.enhance_hop_s,
|
||||
),
|
||||
)
|
||||
clog.info("audio backend=%s enhance=%s beamform=%s gate=%s",
|
||||
cleaner.backend, enhance_mode, beamform,
|
||||
"on" if cleaner._gate is not None else "off")
|
||||
|
||||
room = rtc.Room()
|
||||
nbytes = frame_bytes(w, h)
|
||||
shm_path = os.path.join(args.video_shm_dir, f"{identity}.i420")
|
||||
shm = LatestFrameReader(shm_path)
|
||||
scratch = bytearray(nbytes)
|
||||
video_frames = 0
|
||||
video_gate = {"in_flight": 0, "dropped": 0, "max_in_flight": 3}
|
||||
from encoded_pub import capture_encoded
|
||||
from nal_shm import NalReader
|
||||
if args.video_codec == "av1":
|
||||
from av1_obu import is_keyframe
|
||||
else:
|
||||
from annexb import is_keyframe
|
||||
nal = NalReader(os.path.join(args.h264_shm_dir, f"{identity}.h264"))
|
||||
clog.info("video path=encoded-preencoded identity=%s codec=%s h264=%s",
|
||||
identity, args.video_codec, nal.path)
|
||||
audio_card = cam.get("audio_card")
|
||||
hops = HopReader(os.path.join(args.audio_shm_dir, f"{identity}.pcm")) if audio_card is not None else None
|
||||
atrack = None
|
||||
audio_source = None
|
||||
v_opts, a_opts = _pub_opts(rtc, args, h)
|
||||
|
||||
async with connect_gate:
|
||||
await room.connect(args.url, cam["token"], rtc.RoomOptions(auto_subscribe=False))
|
||||
clog.info("connected room=%r", room.name)
|
||||
video_source = rtc.VideoSource(w, h)
|
||||
video_track = rtc.LocalVideoTrack.create_video_track(
|
||||
f"{identity}-video", video_source)
|
||||
if hops is not None:
|
||||
audio_source = rtc.AudioSource(
|
||||
sample_rate=args.audio_rate, num_channels=1,
|
||||
queue_size_ms=audio_queue_size_ms(
|
||||
args.enhance_hop_s, args.audio_queue_ms))
|
||||
atrack = rtc.LocalAudioTrack.create_audio_track(
|
||||
f"{identity}-audio", audio_source)
|
||||
await room.local_participant.publish_track(video_track, v_opts)
|
||||
if atrack is not None:
|
||||
await room.local_participant.publish_track(atrack, a_opts)
|
||||
clog.info("status: UP identity=%s video_shm=%s audio=%s",
|
||||
identity, shm_path,
|
||||
f"hw:{audio_card},0" if audio_card is not None else "none")
|
||||
|
||||
async def recycle_video():
|
||||
nonlocal video_source, video_track
|
||||
sid = getattr(video_track, "sid", None) or ""
|
||||
clog.warning("encoder recycle rss_anon_kb=%d sid=%s",
|
||||
_rss_anon_kb(), sid)
|
||||
if sid:
|
||||
try:
|
||||
await room.local_participant.unpublish_track(sid)
|
||||
except Exception: # noqa: BLE001
|
||||
clog.exception("unpublish for encoder recycle failed")
|
||||
stagger = max(0, int(cam.get("index") or 0)) * 0.05
|
||||
if stagger:
|
||||
await asyncio.sleep(stagger)
|
||||
video_source = rtc.VideoSource(w, h)
|
||||
video_track = rtc.LocalVideoTrack.create_video_track(
|
||||
f"{identity}-video", video_source)
|
||||
await room.local_participant.publish_track(video_track, v_opts)
|
||||
|
||||
async def video_loop():
|
||||
nonlocal video_frames, video_source, video_track
|
||||
period = 1.0 / max(1, int(args.fps))
|
||||
last_seq = -1
|
||||
due = time.monotonic()
|
||||
local_gen = enc_state["gen"]
|
||||
loop = asyncio.get_running_loop()
|
||||
while not stop.is_set():
|
||||
now = time.monotonic()
|
||||
capture, due = next_video_capture(now, due, period)
|
||||
if capture:
|
||||
au = nal.read()
|
||||
if au is not None:
|
||||
last_seq, payload = au
|
||||
try:
|
||||
rc = capture_encoded(
|
||||
video_source,
|
||||
payload,
|
||||
w,
|
||||
h,
|
||||
is_keyframe(payload),
|
||||
timestamp_us=int(time.monotonic() * 1_000_000),
|
||||
codec=args.video_codec,
|
||||
)
|
||||
if rc == 0:
|
||||
video_frames += 1
|
||||
elif video_frames % 90 == 1:
|
||||
clog.warning("capture_encoded rc=%s seq=%s", rc, last_seq)
|
||||
except Exception: # noqa: BLE001
|
||||
clog.exception("capture_encoded failed")
|
||||
if args.encoder_rss_limit_kb and video_frames % 90 == 0:
|
||||
rss = _rss_anon_kb()
|
||||
async with enc_state["lock"]:
|
||||
if should_recycle_encoders(
|
||||
rss, args.encoder_rss_limit_kb, now,
|
||||
enc_state["last"],
|
||||
args.encoder_recycle_interval_s):
|
||||
enc_state["last"] = now
|
||||
enc_state["gen"] += 1
|
||||
if local_gen < enc_state["gen"]:
|
||||
await recycle_video()
|
||||
local_gen = enc_state["gen"]
|
||||
delay = due - time.monotonic()
|
||||
if delay > 0:
|
||||
await asyncio.sleep(delay)
|
||||
|
||||
async def audio_loop():
|
||||
assert hops is not None and audio_source is not None
|
||||
hop_n = int(round(args.enhance_hop_s * args.audio_rate))
|
||||
n = 0
|
||||
loop = asyncio.get_running_loop()
|
||||
while not stop.is_set():
|
||||
blob = hops.read()
|
||||
if blob is None:
|
||||
await asyncio.sleep(0.005)
|
||||
continue
|
||||
try:
|
||||
pcm = np.frombuffer(blob, dtype=np.int16)
|
||||
if pcm.size % 2 == 0:
|
||||
pcm = pcm.reshape(-1, 2)
|
||||
cleaned = await loop.run_in_executor(None, cleaner.process, pcm)
|
||||
await audio_source.capture_frame(
|
||||
rtc.AudioFrame(
|
||||
cleaned.tobytes(), args.audio_rate, 1, len(cleaned)))
|
||||
n += 1
|
||||
if n % 50 == 0:
|
||||
clog.info("audio hop process_ms=%.1f", cleaner.last_dt_ms)
|
||||
except Exception: # noqa: BLE001
|
||||
clog.exception("audio hop failed")
|
||||
|
||||
tasks = [asyncio.create_task(video_loop(), name=f"{identity}-video")]
|
||||
if hops is not None:
|
||||
tasks.append(asyncio.create_task(audio_loop(), name=f"{identity}-audio"))
|
||||
try:
|
||||
await stop.wait()
|
||||
finally:
|
||||
for t in tasks:
|
||||
t.cancel()
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
clog.info("stopped; delivered %d video frames", video_frames)
|
||||
try:
|
||||
await room.disconnect()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
|
||||
async def amain(args) -> int:
|
||||
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
|
||||
os.environ.pop("LIBVA_DRM_DEVICE", None)
|
||||
import livekit.rtc as rtc
|
||||
|
||||
cams = json.loads(args.cameras_json)
|
||||
stop = asyncio.Event()
|
||||
loop = asyncio.get_running_loop()
|
||||
for sig in (signal.SIGTERM, signal.SIGINT):
|
||||
try:
|
||||
loop.add_signal_handler(sig, stop.set)
|
||||
except NotImplementedError:
|
||||
pass
|
||||
log.info("publisher start cams=%d shm=%s pcm=%s speaker_only=%s "
|
||||
"encoder_rss_limit_kb=%d",
|
||||
len(cams), args.video_shm_dir, args.audio_shm_dir,
|
||||
args.enhance_speaker_only, args.encoder_rss_limit_kb)
|
||||
gate = asyncio.Semaphore(1)
|
||||
enc_state = {"lock": asyncio.Lock(), "last": 0.0, "gen": 0}
|
||||
tasks = [
|
||||
asyncio.create_task(
|
||||
publish_one(rtc, cam, args, stop, gate, enc_state),
|
||||
name=cam["identity"])
|
||||
for cam in cams
|
||||
]
|
||||
await stop.wait()
|
||||
for t in tasks:
|
||||
t.cancel()
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--cameras-json", required=True)
|
||||
ap.add_argument("--url", required=True)
|
||||
ap.add_argument("--room", required=True)
|
||||
ap.add_argument("--width", type=int, default=640)
|
||||
ap.add_argument("--height", type=int, default=360)
|
||||
ap.add_argument("--fps", type=int, default=30)
|
||||
ap.add_argument("--bitrate", type=int, default=1_200_000)
|
||||
ap.add_argument("--video-codec", default="h264")
|
||||
ap.add_argument("--video-encoder", default="vaapi")
|
||||
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
|
||||
ap.add_argument("--audio-rate", type=int, default=16000)
|
||||
ap.add_argument("--enhance-mode", default="auto")
|
||||
ap.add_argument("--enhance-model",
|
||||
default="speechbrain/metricgan-plus-voicebank")
|
||||
ap.add_argument("--enhance-chunk-s", type=float, default=1.0)
|
||||
ap.add_argument("--enhance-hop-s", type=float, default=0.1)
|
||||
ap.add_argument("--enhance-infer", default="auto")
|
||||
ap.add_argument("--beamform-mode", default="auto")
|
||||
ap.add_argument("--torch-num-threads", type=int, default=1)
|
||||
ap.add_argument("--tdoa-every", type=int, default=5)
|
||||
ap.add_argument("--audio-queue-ms", type=int, default=None)
|
||||
ap.add_argument("--enhance-socket", default="")
|
||||
ap.add_argument("--video-shm-dir", default="/run/livekit-cameras/raw")
|
||||
ap.add_argument("--audio-shm-dir", default="/run/livekit-cameras/pcm")
|
||||
ap.add_argument("--h264-shm-dir", default="/run/livekit-cameras/h264")
|
||||
ap.add_argument("--enhance-speaker-only", action="store_true")
|
||||
ap.add_argument("--speaker-camera", default="rally")
|
||||
ap.add_argument("--audio-gate", action="store_true")
|
||||
ap.add_argument("--audio-gate-open-db", type=float, default=-28.0)
|
||||
ap.add_argument("--audio-gate-close-db", type=float, default=-34.0)
|
||||
ap.add_argument("--audio-gate-hold-s", type=float, default=0.3)
|
||||
ap.add_argument("--encoder-rss-limit-kb", type=int, default=0,
|
||||
help="Recycle VAAPI VideoSource tracks when RssAnon exceeds this (0=off)")
|
||||
ap.add_argument("--encoder-recycle-interval-s", type=float, default=120.0)
|
||||
args = ap.parse_args()
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [%(name)s] %(levelname)s %(message)s")
|
||||
try:
|
||||
return asyncio.run(amain(args))
|
||||
except Exception: # noqa: BLE001
|
||||
log.exception("publisher fatal")
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
+155
-67
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import concurrent.futures
|
||||
import contextlib
|
||||
import json
|
||||
import logging
|
||||
@@ -35,10 +36,10 @@ def main() -> int:
|
||||
ap.add_argument("--room", required=True)
|
||||
ap.add_argument("--url", required=True)
|
||||
ap.add_argument("--token", required=True)
|
||||
ap.add_argument("--width", type=int, default=640)
|
||||
ap.add_argument("--height", type=int, default=360)
|
||||
ap.add_argument("--fps", type=int, default=30)
|
||||
ap.add_argument("--bitrate", type=int, default=800_000)
|
||||
ap.add_argument("--width", type=int, default=320)
|
||||
ap.add_argument("--height", type=int, default=180)
|
||||
ap.add_argument("--fps", type=int, default=15)
|
||||
ap.add_argument("--bitrate", type=int, default=400_000)
|
||||
ap.add_argument("--video-codec", default="h264")
|
||||
ap.add_argument("--video-encoder", default="vaapi")
|
||||
ap.add_argument("--vaapi-device", default="/dev/dri/renderD129")
|
||||
@@ -52,6 +53,17 @@ def main() -> int:
|
||||
ap.add_argument("--beamform-mode", default="auto")
|
||||
ap.add_argument("--torch-num-threads", type=int, default=1)
|
||||
ap.add_argument("--tdoa-every", type=int, default=5)
|
||||
ap.add_argument("--audio-queue-ms", type=int, default=None)
|
||||
ap.add_argument("--video-hold-s", type=float, default=None)
|
||||
ap.add_argument("--split-executor", action="store_true")
|
||||
ap.add_argument("--enhance-socket", default="")
|
||||
ap.add_argument("--speaker-camera", default="rally")
|
||||
ap.add_argument("--audio-gate", action="store_true")
|
||||
ap.add_argument("--audio-gate-open-db", type=float, default=-28.0)
|
||||
ap.add_argument("--audio-gate-close-db", type=float, default=-34.0)
|
||||
ap.add_argument("--audio-gate-hold-s", type=float, default=0.3)
|
||||
ap.add_argument("--video-shm", default="",
|
||||
help="latest-frame I420 mmap from capture_daemon")
|
||||
args = ap.parse_args()
|
||||
|
||||
cam = json.loads(args.camera_json)
|
||||
@@ -62,15 +74,16 @@ def main() -> int:
|
||||
|
||||
from config import Config # noqa: F401 (ensures .env is loaded)
|
||||
from audio_cleanup import AudioCleaner, apply_torch_thread_limits
|
||||
apply_torch_thread_limits(args.torch_num_threads)
|
||||
import livekit.rtc as rtc
|
||||
|
||||
from audio_gate import make_gate
|
||||
from latency import audio_queue_size_ms, video_hold_seconds
|
||||
identity = cam["identity"]
|
||||
os.environ["LIBVA_DRIVER_NAME"] = "iHD"
|
||||
os.environ.pop("LIBVA_DRM_DEVICE", None)
|
||||
if not args.enhance_socket:
|
||||
apply_torch_thread_limits(args.torch_num_threads)
|
||||
import livekit.rtc as rtc
|
||||
video_dev = cam["video_device"]
|
||||
audio_card = cam.get("audio_card")
|
||||
if args.vaapi_device:
|
||||
os.environ.setdefault("LIBVA_DRM_DEVICE", args.vaapi_device)
|
||||
os.environ.setdefault("LIBVA_DRIVER_NAME", "iHD")
|
||||
|
||||
async def run() -> int:
|
||||
cleaner = AudioCleaner(
|
||||
@@ -83,76 +96,145 @@ def main() -> int:
|
||||
torch_num_threads=args.torch_num_threads,
|
||||
tdoa_every=args.tdoa_every,
|
||||
enhance_infer=args.enhance_infer,
|
||||
enhance_socket=args.enhance_socket,
|
||||
identity=identity,
|
||||
gate=make_gate(
|
||||
identity,
|
||||
args.speaker_camera,
|
||||
bool(args.audio_gate),
|
||||
args.audio_gate_open_db,
|
||||
args.audio_gate_close_db,
|
||||
args.audio_gate_hold_s,
|
||||
args.enhance_hop_s,
|
||||
),
|
||||
)
|
||||
log.info("audio backend: %s", cleaner.backend)
|
||||
log.info("audio backend: %s gate=%s", cleaner.backend,
|
||||
"on" if cleaner._gate is not None else "off")
|
||||
|
||||
io_ex = None
|
||||
enhance_ex = None
|
||||
split_owned: list[concurrent.futures.Executor] = []
|
||||
if args.split_executor:
|
||||
io_ex = concurrent.futures.ThreadPoolExecutor(
|
||||
max_workers=2, thread_name_prefix=f"{identity}-io")
|
||||
enhance_ex = concurrent.futures.ThreadPoolExecutor(
|
||||
max_workers=1, thread_name_prefix=f"{identity}-enh")
|
||||
split_owned.extend([io_ex, enhance_ex])
|
||||
log.info("split executor: ffmpeg io vs enhance")
|
||||
|
||||
room = rtc.Room()
|
||||
await room.connect(args.url, args.token)
|
||||
log.info("connected to room %r as %r", room.name, identity)
|
||||
await room.connect(
|
||||
args.url, args.token, rtc.RoomOptions(auto_subscribe=False))
|
||||
log.info("connected to room %r as %r (auto_subscribe=False)",
|
||||
room.name, identity)
|
||||
|
||||
# ---------------- video ----------------
|
||||
video_source = rtc.VideoSource(args.width, args.height)
|
||||
video_track = rtc.LocalVideoTrack.create_video_track(
|
||||
f"{identity}-video", video_source)
|
||||
|
||||
vproc = subprocess.Popen(
|
||||
["ffmpeg", "-hide_banner", "-loglevel", "error",
|
||||
"-fflags", "nobuffer", "-flags", "low_delay",
|
||||
"-thread_queue_size", "4",
|
||||
"-f", "v4l2",
|
||||
"-input_format", "mjpeg",
|
||||
"-video_size", f"{args.width}x{args.height}",
|
||||
"-framerate", str(args.fps),
|
||||
"-i", video_dev,
|
||||
"-f", "rawvideo", "-pix_fmt", "yuv420p",
|
||||
"pipe:1"],
|
||||
stdout=subprocess.PIPE,
|
||||
)
|
||||
frame_bytes = args.width * args.height * 3 // 2
|
||||
video_frames = 0
|
||||
# Audio is published in enhance hops (default 200 ms). Hold video by
|
||||
# that same amount so lipsync at the subscriber: a frame captured at
|
||||
# T is sent when the audio hop covering T is sent.
|
||||
video_hold_s = float(args.enhance_hop_s) if audio_card is not None else 0.0
|
||||
vproc: subprocess.Popen | None = None
|
||||
shm_path = (args.video_shm or "").strip()
|
||||
video_hold_s = video_hold_seconds(
|
||||
args.enhance_hop_s,
|
||||
audio_card is not None,
|
||||
0.0 if shm_path else args.video_hold_s,
|
||||
)
|
||||
pending: deque[tuple[float, bytes]] = deque()
|
||||
max_pending = max(8, int(args.fps * (video_hold_s + 0.15)) + 4)
|
||||
if video_hold_s > 0:
|
||||
log.info("av sync: holding video %.0f ms to match audio hop",
|
||||
video_hold_s * 1000.0)
|
||||
|
||||
async def video_loop():
|
||||
nonlocal video_frames
|
||||
loop = asyncio.get_running_loop()
|
||||
while vproc.poll() is None:
|
||||
data = await loop.run_in_executor(
|
||||
None, vproc.stdout.read, frame_bytes)
|
||||
if not data or len(data) < frame_bytes:
|
||||
if not data:
|
||||
break
|
||||
keep = b""
|
||||
while len(keep) < frame_bytes:
|
||||
more = await loop.run_in_executor(
|
||||
None, vproc.stdout.read,
|
||||
frame_bytes - len(keep))
|
||||
if not more:
|
||||
return
|
||||
keep += more
|
||||
data = keep
|
||||
now = time.monotonic()
|
||||
pending.append((now, data))
|
||||
while len(pending) > max_pending:
|
||||
pending.popleft()
|
||||
for payload in drain_held_frames(pending, now, video_hold_s):
|
||||
if shm_path:
|
||||
from frame_shm import LatestFrameReader
|
||||
shm_reader = LatestFrameReader(shm_path)
|
||||
log.info("video from shared capture shm=%s hold=0 (latest frame)",
|
||||
shm_path)
|
||||
|
||||
async def video_loop():
|
||||
nonlocal video_frames
|
||||
period = 1.0 / max(1, int(args.fps))
|
||||
last_seq = -1
|
||||
next_t = time.monotonic()
|
||||
scratch = bytearray(frame_bytes)
|
||||
while True:
|
||||
seq = shm_reader.copy_into(scratch)
|
||||
if seq is not None and seq != last_seq:
|
||||
last_seq = seq
|
||||
try:
|
||||
video_source.capture_frame(
|
||||
rtc.VideoFrame(
|
||||
args.width, args.height,
|
||||
rtc.VideoBufferType.I420, scratch),
|
||||
timestamp_us=int(time.monotonic() * 1_000_000),
|
||||
)
|
||||
video_frames += 1
|
||||
except Exception: # noqa: BLE001
|
||||
log.exception("video capture_frame failed")
|
||||
next_t += period
|
||||
delay = next_t - time.monotonic()
|
||||
if delay > 0:
|
||||
await asyncio.sleep(delay)
|
||||
else:
|
||||
next_t = time.monotonic()
|
||||
else:
|
||||
from capture_va import video_capture_cmd
|
||||
vproc = subprocess.Popen(
|
||||
video_capture_cmd(
|
||||
video_dev, args.width, args.height, args.fps, fd=1),
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
)
|
||||
log.info("capture gst jpegdec Arc (no sidecar h264)")
|
||||
if video_hold_s > 0:
|
||||
log.info("av sync: holding video %.0f ms to match audio hop",
|
||||
video_hold_s * 1000.0)
|
||||
|
||||
async def video_loop():
|
||||
nonlocal video_frames
|
||||
assert vproc is not None
|
||||
loop = asyncio.get_running_loop()
|
||||
stdout = vproc.stdout
|
||||
assert stdout is not None
|
||||
while vproc.poll() is None:
|
||||
data = await loop.run_in_executor(
|
||||
io_ex, stdout.read, frame_bytes)
|
||||
if not data or len(data) < frame_bytes:
|
||||
if not data:
|
||||
break
|
||||
keep = b""
|
||||
while len(keep) < frame_bytes:
|
||||
more = await loop.run_in_executor(
|
||||
io_ex, stdout.read,
|
||||
frame_bytes - len(keep))
|
||||
if not more:
|
||||
return
|
||||
keep += more
|
||||
data = keep
|
||||
now = time.monotonic()
|
||||
pending.append((now, data))
|
||||
while len(pending) > max_pending:
|
||||
pending.popleft()
|
||||
for payload in drain_held_frames(pending, now, video_hold_s):
|
||||
try:
|
||||
video_source.capture_frame(
|
||||
rtc.VideoFrame(
|
||||
args.width, args.height,
|
||||
rtc.VideoBufferType.I420, payload),
|
||||
timestamp_us=int(time.monotonic() * 1_000_000),
|
||||
)
|
||||
video_frames += 1
|
||||
except Exception: # noqa: BLE001
|
||||
log.exception("video capture_frame failed")
|
||||
err = b""
|
||||
if vproc.stderr:
|
||||
try:
|
||||
video_source.capture_frame(
|
||||
rtc.VideoFrame(
|
||||
args.width, args.height,
|
||||
rtc.VideoBufferType.I420, payload),
|
||||
timestamp_us=int(time.monotonic() * 1_000_000),
|
||||
)
|
||||
video_frames += 1
|
||||
err = vproc.stderr.read() or b""
|
||||
except Exception: # noqa: BLE001
|
||||
log.exception("video capture_frame failed")
|
||||
pass
|
||||
log.error("capture gst exited rc=%s stderr=%s",
|
||||
vproc.returncode,
|
||||
err.decode("utf-8", "replace")[-800:])
|
||||
|
||||
# ---------------- audio ----------------
|
||||
aproc: subprocess.Popen | None = None
|
||||
@@ -160,9 +242,12 @@ def main() -> int:
|
||||
if audio_card is not None:
|
||||
audio_source = rtc.AudioSource(
|
||||
sample_rate=args.audio_rate, num_channels=1,
|
||||
queue_size_ms=max(250, int(args.enhance_hop_s * 1000) + 50))
|
||||
queue_size_ms=audio_queue_size_ms(
|
||||
args.enhance_hop_s, args.audio_queue_ms))
|
||||
atrack = rtc.LocalAudioTrack.create_audio_track(
|
||||
f"{identity}-audio", audio_source)
|
||||
log.info("audio queue_size_ms=%d", audio_queue_size_ms(
|
||||
args.enhance_hop_s, args.audio_queue_ms))
|
||||
|
||||
hop_s = args.enhance_hop_s
|
||||
chunk = int(hop_s * args.audio_rate) # samples per hop
|
||||
@@ -177,7 +262,7 @@ def main() -> int:
|
||||
hops = 0
|
||||
while aproc.poll() is None:
|
||||
data = await loop.run_in_executor(
|
||||
None, aproc.stdout.read, 4096)
|
||||
io_ex, aproc.stdout.read, 4096)
|
||||
if not data:
|
||||
break
|
||||
buf += data
|
||||
@@ -189,7 +274,7 @@ def main() -> int:
|
||||
usable = (n_frames // chunk) * chunk
|
||||
if usable:
|
||||
cleaned = await loop.run_in_executor(
|
||||
None, cleaner.process,
|
||||
enhance_ex, cleaner.process,
|
||||
pcm_stereo[:usable])
|
||||
t_cap = time.perf_counter()
|
||||
await audio_source.capture_frame(
|
||||
@@ -292,6 +377,9 @@ def main() -> int:
|
||||
for t in tasks:
|
||||
t.cancel()
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
for ex in split_owned:
|
||||
with contextlib.suppress(Exception):
|
||||
ex.shutdown(wait=False, cancel_futures=True)
|
||||
for p in (vproc, aproc):
|
||||
if p is not None and p.poll() is None:
|
||||
with contextlib.suppress(Exception):
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Throwaway: one software-encoder publisher reading cam-01 I420 SHM.
|
||||
|
||||
Does not open /dev/camNN. Identity spike-sw so it does not kick cam-01.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(ROOT))
|
||||
os.chdir(ROOT)
|
||||
|
||||
from config import load_config # noqa: E402
|
||||
from tokens import make_token # noqa: E402
|
||||
|
||||
SHM_DIR = "/run/livekit-cameras/raw"
|
||||
SRC = f"{SHM_DIR}/cam-01.i420"
|
||||
LINK = f"{SHM_DIR}/spike-sw.i420"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
cfg = load_config()
|
||||
if not os.path.exists(SRC):
|
||||
print("missing", SRC, file=sys.stderr)
|
||||
return 1
|
||||
try:
|
||||
os.remove(LINK)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
os.symlink(SRC, LINK)
|
||||
cam = {
|
||||
"index": 0,
|
||||
"identity": "spike-sw",
|
||||
"video_device": "/dev/null",
|
||||
"audio_card": None,
|
||||
"usb_port": "",
|
||||
"serial": "",
|
||||
"extra_video_devices": [],
|
||||
"width": 640,
|
||||
"height": 360,
|
||||
"token": make_token(cfg.url, cfg.api_key, cfg.api_secret, cfg.room, "spike-sw"),
|
||||
}
|
||||
os.environ.setdefault(
|
||||
"LIVEKIT_LIB_PATH",
|
||||
str(ROOT / "native" / "liblivekit_ffi.so"),
|
||||
)
|
||||
os.environ.setdefault("LIBVA_DRM_DEVICE", cfg.vaapi_device)
|
||||
os.environ.setdefault("LIBVA_DRIVER_NAME", "iHD")
|
||||
argv = [
|
||||
sys.executable,
|
||||
str(ROOT / "run_publisher.py"),
|
||||
"--cameras-json", json.dumps([cam]),
|
||||
"--url", cfg.url,
|
||||
"--room", cfg.room,
|
||||
"--width", "640",
|
||||
"--height", "360",
|
||||
"--fps", "30",
|
||||
"--bitrate", str(cfg.video_bitrate),
|
||||
"--video-codec", "h264",
|
||||
"--video-encoder", "software",
|
||||
"--vaapi-device", cfg.vaapi_device,
|
||||
"--audio-rate", str(cfg.audio_rate),
|
||||
"--enhance-mode", "off",
|
||||
"--beamform-mode", "off",
|
||||
"--video-shm-dir", SHM_DIR,
|
||||
"--audio-shm-dir", cfg.audio_shm_dir,
|
||||
"--encoder-rss-limit-kb", "99000000",
|
||||
]
|
||||
os.execv(sys.executable, argv)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
BIN
Binary file not shown.
@@ -0,0 +1,23 @@
|
||||
"""Annex-B access-unit split."""
|
||||
from annexb import AnnexBSplitter, is_keyframe, nal_unit_type
|
||||
|
||||
|
||||
def test_splitter_emits_complete_nal():
|
||||
s = AnnexBSplitter()
|
||||
assert s.push(b"") == []
|
||||
first = b"\x00\x00\x00\x01\x67SPS"
|
||||
second = b"\x00\x00\x00\x01\x68PPS"
|
||||
assert s.push(first + second) == [first]
|
||||
assert s.push(b"\x00\x00\x00\x01\x65IDR") == [second]
|
||||
|
||||
|
||||
def test_is_keyframe_idr():
|
||||
au = b"\x00\x00\x00\x01\x67\x00\x00\x00\x01\x65\x88"
|
||||
assert is_keyframe(au) is True
|
||||
assert nal_unit_type(au) == 7
|
||||
|
||||
|
||||
def test_is_keyframe_delta():
|
||||
au = b"\x00\x00\x00\x01\x41\x00"
|
||||
assert is_keyframe(au) is False
|
||||
assert nal_unit_type(au) == 1
|
||||
@@ -0,0 +1,73 @@
|
||||
"""Hop energy gate so neighboring C920s do not publish bleed."""
|
||||
import numpy as np
|
||||
|
||||
from audio_gate import HopGate, gate_enabled_for, rms_dbfs
|
||||
|
||||
|
||||
def _sine(n=1600, amp=8000):
|
||||
return (np.sin(2 * np.pi * 440 * np.arange(n) / 16000) * amp).astype(np.int16)
|
||||
|
||||
|
||||
def test_rms_dbfs_silence_is_floor():
|
||||
assert rms_dbfs(np.zeros(1600, dtype=np.int16)) <= -120.0
|
||||
|
||||
|
||||
def test_rms_dbfs_sine_is_near_minus_15():
|
||||
db = rms_dbfs(_sine())
|
||||
assert -16.5 < db < -14.0
|
||||
|
||||
|
||||
def test_gate_silences_quiet_hop():
|
||||
g = HopGate(open_db=-32.0, close_db=-40.0, hold_hops=3)
|
||||
quiet = (np.random.randn(1600) * 80).astype(np.int16)
|
||||
out = g.apply(quiet)
|
||||
assert int(np.max(np.abs(out))) == 0
|
||||
assert g.open is False
|
||||
|
||||
|
||||
def test_gate_passes_loud_hop():
|
||||
g = HopGate(open_db=-32.0, close_db=-40.0, hold_hops=3)
|
||||
loud = _sine()
|
||||
out = g.apply(loud)
|
||||
assert int(np.max(np.abs(out))) > 1000
|
||||
assert g.open is True
|
||||
|
||||
|
||||
def test_gate_hold_then_closes():
|
||||
g = HopGate(open_db=-32.0, close_db=-40.0, hold_hops=2)
|
||||
g.apply(_sine())
|
||||
quiet = np.zeros(1600, dtype=np.int16)
|
||||
assert int(np.max(np.abs(g.apply(quiet)))) == 0 or g.open is True
|
||||
g.apply(quiet)
|
||||
g.apply(quiet)
|
||||
assert g.open is False
|
||||
assert int(np.max(np.abs(g.apply(quiet)))) == 0
|
||||
|
||||
|
||||
def test_gate_skips_rally():
|
||||
assert gate_enabled_for("rally", "rally", True) is False
|
||||
assert gate_enabled_for("cam-07", "rally", True) is True
|
||||
assert gate_enabled_for("cam-07", "rally", False) is False
|
||||
|
||||
|
||||
def test_cleaner_gates_quiet_after_dsp():
|
||||
from audio_cleanup import AudioCleaner
|
||||
c = AudioCleaner(mode="off", beamform="off", gate=HopGate(open_db=-32.0, close_db=-40.0, hold_hops=1))
|
||||
quiet = (np.random.randn(1600) * 80).astype(np.int16)
|
||||
out = c.process(quiet)
|
||||
assert out.shape == (1600,)
|
||||
assert int(np.max(np.abs(out))) == 0
|
||||
loud = _sine(1600, 8000)
|
||||
out = c.process(loud)
|
||||
assert int(np.max(np.abs(out))) > 1000
|
||||
|
||||
|
||||
def test_cleaner_gate_uses_raw_energy_not_dsp_makeup():
|
||||
"""Room floor ~-34 dBFS must stay closed even though DSP makeup is ~+3 dB."""
|
||||
from audio_cleanup import AudioCleaner
|
||||
c = AudioCleaner(mode="off", beamform="off", gate=HopGate(open_db=-32.0, close_db=-40.0, hold_hops=1))
|
||||
# sine RMS -34 dBFS => amplitude ~924
|
||||
floor = _sine(1600, 924)
|
||||
assert -35.5 < rms_dbfs(floor) < -32.5
|
||||
out = c.process(floor)
|
||||
assert int(np.max(np.abs(out))) == 0
|
||||
@@ -0,0 +1,19 @@
|
||||
from audio_gst import grouped_audio, shared_audio_cmd
|
||||
|
||||
|
||||
def test_shared_audio_cmd_no_video_elements():
|
||||
cams = [("cam-01", 4), ("cam-02", 5)]
|
||||
cmd = shared_audio_cmd(cams, 16000, [8, 9])
|
||||
joined = " ".join(cmd)
|
||||
assert joined.count("alsasrc") == 2
|
||||
assert "device=hw:4,0" in joined
|
||||
assert "device=hw:5,0" in joined
|
||||
assert "h264enc" not in joined
|
||||
assert "v4l2src" not in joined
|
||||
|
||||
|
||||
def test_grouped_audio_splits():
|
||||
cams = [(f"cam-{i:02d}", i + 1) for i in range(1, 21)]
|
||||
fds = list(range(20))
|
||||
groups = grouped_audio(cams, fds, 5)
|
||||
assert len(groups) == 4
|
||||
@@ -0,0 +1,50 @@
|
||||
from capture_va import grouped_cameras, shared_capture_cmd, video_capture_cmd
|
||||
|
||||
|
||||
def test_video_capture_cmd_latest_frame_no_sidecar_h264():
|
||||
cmd = video_capture_cmd("/dev/cam01", 640, 360, 30)
|
||||
joined = " ".join(cmd)
|
||||
assert "v4l2src" in joined
|
||||
assert "device=/dev/cam01" in joined
|
||||
assert "jpegdec" in joined
|
||||
assert "videoconvert" in joined
|
||||
assert "I420" in joined
|
||||
assert "leaky=downstream" in joined
|
||||
assert "max-size-buffers=1" in joined
|
||||
assert "max-size-time=0" in joined
|
||||
assert "fdsink" in joined
|
||||
assert "h264enc" not in joined
|
||||
assert "tee" not in joined
|
||||
|
||||
|
||||
def test_capture_does_not_gst_crop_or_box():
|
||||
"""MCU crop is Python-side on 07/14/15 only; gst stays full 360."""
|
||||
for dev in ("/dev/cam01", "/dev/cam07"):
|
||||
joined = " ".join(video_capture_cmd(dev, 640, 360, 30))
|
||||
assert "videocrop" not in joined
|
||||
assert "videobox" not in joined
|
||||
assert "videoscale" not in joined
|
||||
|
||||
|
||||
def test_shared_capture_cmd_one_process_no_sidecar_h264():
|
||||
cams = [("cam-01", "/dev/cam01"), ("cam-02", "/dev/cam02")]
|
||||
cmd = shared_capture_cmd(cams, 640, 360, 30, [8, 9])
|
||||
joined = " ".join(cmd)
|
||||
assert cmd[0].endswith("gst-launch-1.0") or "gst-launch-1.0" in cmd[0]
|
||||
assert joined.count("v4l2src") == 2
|
||||
assert joined.count("jpegdec") == 2
|
||||
assert joined.count("videoconvert") == 2
|
||||
assert joined.count("fdsink") == 2
|
||||
assert "fd=8" in joined and "fd=9" in joined
|
||||
assert "h264enc" not in joined
|
||||
assert "tee" not in joined
|
||||
|
||||
|
||||
def test_grouped_cameras_splits_into_small_pool():
|
||||
cams = [(f"cam-{i:02d}", f"/dev/cam{i:02d}") for i in range(1, 21)]
|
||||
fds = list(range(20, 40))
|
||||
groups = grouped_cameras(cams, fds, 5)
|
||||
assert len(groups) == 4
|
||||
assert all(len(c) == 5 and len(f) == 5 for c, f in groups)
|
||||
assert groups[0][0][0][0] == "cam-01"
|
||||
assert groups[-1][0][-1][0] == "cam-20"
|
||||
+124
-1
@@ -1,4 +1,4 @@
|
||||
from config import Config, _parse_cameras
|
||||
from config import Config, _parse_cameras, load_config
|
||||
|
||||
|
||||
def test_parse_cameras_list():
|
||||
@@ -17,3 +17,126 @@ def test_torch_num_threads_default_one():
|
||||
|
||||
def test_enhance_infer_default_auto():
|
||||
assert Config().enhance_infer == "auto"
|
||||
|
||||
|
||||
def test_latency_flag_defaults():
|
||||
c = Config()
|
||||
assert c.audio_queue_ms is None
|
||||
assert c.video_hold_s is None
|
||||
assert c.worker_split_executor is False
|
||||
assert c.enhance_speaker_only is False
|
||||
assert c.audio_gate is False
|
||||
assert c.audio_gate_open_db == -28.0
|
||||
assert c.audio_gate_close_db == -34.0
|
||||
assert c.audio_gate_hold_s == 0.3
|
||||
assert c.display_video_capacity == 0
|
||||
assert c.display_video_format == ""
|
||||
assert c.display_gst_queue_buffers == 2
|
||||
assert c.display_speaker_push == "timer"
|
||||
assert c.participant_tags == ["uwh"]
|
||||
assert c.display_hide_tags == []
|
||||
assert c.display_hand_raise is True
|
||||
assert c.display_hand_raise_hz == 4.0
|
||||
assert c.display_hand_raise_hold_s == 1.0
|
||||
assert c.display_hand_raise_model == "thunder"
|
||||
assert c.display_speak_min_db == -40.0
|
||||
assert c.display_speak_margin_db == 6.0
|
||||
assert c.display_speak_hold_s == 0.6
|
||||
assert c.display_speak_rise_db == 3.0
|
||||
assert c.display_speak_corr == 0.4
|
||||
assert c.display_speak_confirm == 2
|
||||
assert c.portrait_blur is True
|
||||
assert c.portrait_shm_dir == "/run/livekit-cameras/portrait"
|
||||
assert c.portrait_hz == 8.0
|
||||
assert c.portrait_blur_px == 7
|
||||
assert c.portrait_hold_s == 0.8
|
||||
|
||||
|
||||
def test_enhance_daemon_defaults():
|
||||
c = Config()
|
||||
assert c.enhance_daemon is True
|
||||
assert c.enhance_socket == "/tmp/livekit-enhance.sock"
|
||||
assert c.capture_daemon is True
|
||||
assert c.capture_shm_dir == "/run/livekit-cameras/raw"
|
||||
assert c.audio_daemon is True
|
||||
assert c.audio_shm_dir == "/run/livekit-cameras/pcm"
|
||||
assert c.single_publisher is True
|
||||
|
||||
|
||||
def test_load_config_enhance_daemon_env(monkeypatch, tmp_path):
|
||||
env = tmp_path / ".env"
|
||||
env.write_text("")
|
||||
monkeypatch.setenv("ENHANCE_DAEMON", "0")
|
||||
monkeypatch.setenv("ENHANCE_SOCKET", "/run/foo.sock")
|
||||
cfg = load_config(env)
|
||||
assert cfg.enhance_daemon is False
|
||||
assert cfg.enhance_socket == "/run/foo.sock"
|
||||
|
||||
|
||||
def test_load_config_latency_env(monkeypatch, tmp_path):
|
||||
env = tmp_path / ".env"
|
||||
env.write_text("")
|
||||
monkeypatch.setenv("AUDIO_QUEUE_MS", "60")
|
||||
monkeypatch.setenv("VIDEO_HOLD_S", "0")
|
||||
monkeypatch.setenv("WORKER_SPLIT_EXECUTOR", "1")
|
||||
monkeypatch.setenv("ENHANCE_SPEAKER_ONLY", "true")
|
||||
monkeypatch.setenv("AUDIO_GATE", "0")
|
||||
monkeypatch.setenv("AUDIO_GATE_OPEN_DB", "-28")
|
||||
monkeypatch.setenv("AUDIO_GATE_HOLD_S", "0.5")
|
||||
monkeypatch.setenv("DISPLAY_VIDEO_CAPACITY", "1")
|
||||
monkeypatch.setenv("DISPLAY_VIDEO_FORMAT", "BGRA")
|
||||
monkeypatch.setenv("DISPLAY_GST_QUEUE_BUFFERS", "1")
|
||||
monkeypatch.setenv("DISPLAY_SPEAKER_PUSH", "arrival")
|
||||
cfg = load_config(env)
|
||||
assert cfg.audio_queue_ms == 60
|
||||
assert cfg.video_hold_s == 0.0
|
||||
assert cfg.worker_split_executor is True
|
||||
assert cfg.enhance_speaker_only is True
|
||||
assert cfg.audio_gate is False
|
||||
assert cfg.audio_gate_open_db == -28.0
|
||||
assert cfg.audio_gate_hold_s == 0.5
|
||||
assert cfg.display_video_capacity == 1
|
||||
assert cfg.display_video_format == "bgra"
|
||||
assert cfg.display_gst_queue_buffers == 1
|
||||
assert cfg.display_speaker_push == "arrival"
|
||||
|
||||
|
||||
def test_load_config_site_tags(monkeypatch, tmp_path):
|
||||
env = tmp_path / ".env"
|
||||
env.write_text("")
|
||||
monkeypatch.setenv("PARTICIPANT_TAGS", "uwh")
|
||||
monkeypatch.setenv("DISPLAY_HIDE_TAGS", "")
|
||||
cfg = load_config(env)
|
||||
assert cfg.participant_tags == ["uwh"]
|
||||
assert cfg.display_hide_tags == []
|
||||
monkeypatch.setenv("DISPLAY_HIDE_TAGS", "uwh")
|
||||
cfg = load_config(env)
|
||||
assert cfg.display_hide_tags == ["uwh"]
|
||||
|
||||
|
||||
def test_load_config_hand_raise_env(monkeypatch, tmp_path):
|
||||
env = tmp_path / ".env"
|
||||
env.write_text("")
|
||||
monkeypatch.setenv("DISPLAY_HAND_RAISE", "0")
|
||||
monkeypatch.setenv("DISPLAY_HAND_RAISE_HZ", "3")
|
||||
monkeypatch.setenv("DISPLAY_HAND_RAISE_HOLD_S", "1.2")
|
||||
cfg = load_config(env)
|
||||
assert cfg.display_hand_raise is False
|
||||
assert cfg.display_hand_raise_hz == 3.0
|
||||
assert cfg.display_hand_raise_hold_s == 1.2
|
||||
|
||||
|
||||
def test_load_config_portrait_env(monkeypatch, tmp_path):
|
||||
env = tmp_path / ".env"
|
||||
env.write_text("")
|
||||
monkeypatch.setenv("PORTRAIT_BLUR", "0")
|
||||
monkeypatch.setenv("PORTRAIT_SHM_DIR", "/tmp/portrait")
|
||||
monkeypatch.setenv("PORTRAIT_HZ", "5")
|
||||
monkeypatch.setenv("PORTRAIT_BLUR_PX", "11")
|
||||
monkeypatch.setenv("PORTRAIT_HOLD_S", "0.4")
|
||||
cfg = load_config(env)
|
||||
assert cfg.portrait_blur is False
|
||||
assert cfg.portrait_shm_dir == "/tmp/portrait"
|
||||
assert cfg.portrait_hz == 5.0
|
||||
assert cfg.portrait_blur_px == 11
|
||||
assert cfg.portrait_hold_s == 0.4
|
||||
|
||||
@@ -3,11 +3,18 @@ import numpy as np
|
||||
|
||||
from display_grid import (
|
||||
CameraGrid,
|
||||
bgra_to_i420,
|
||||
camera_ids,
|
||||
describe_livekit_error,
|
||||
display_name,
|
||||
format_livekit_status,
|
||||
grid_shape,
|
||||
i420_nbytes,
|
||||
is_camera_identity,
|
||||
letterbox_bgra,
|
||||
letterbox_i420,
|
||||
placeholder,
|
||||
status_screen,
|
||||
)
|
||||
|
||||
|
||||
@@ -75,7 +82,147 @@ def test_gap_separates_tiles():
|
||||
assert tuple(canvas[y, x]) == (10, 10, 10, 255)
|
||||
|
||||
|
||||
def test_inner_rect_cam01_is_top_left_inside_border():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4, gap=8, border=3)
|
||||
x, y, w, h = g.inner_rect(0)
|
||||
assert (x, y) == (g.gap + g.border, g.gap + g.border)
|
||||
assert (w, h) == (g.inner_w, g.inner_h)
|
||||
x2, y2, _, _ = g.inner_rect(1)
|
||||
assert x2 == g.gap + (g.tile_w + g.gap) + g.border
|
||||
assert y2 == y
|
||||
|
||||
|
||||
def test_placeholder_draws_text():
|
||||
img = placeholder(640, 360, ["Screen share", "stub"])
|
||||
assert img.shape == (360, 640, 4)
|
||||
assert img.sum() > 0
|
||||
|
||||
|
||||
def test_letterbox_i420_size_and_even_uv():
|
||||
src = np.zeros((360, 640, 4), dtype=np.uint8)
|
||||
src[:] = (0, 255, 0, 255)
|
||||
blob = bgra_to_i420(src)
|
||||
out = letterbox_i420(blob, 640, 360, 368, 226)
|
||||
assert len(out) == i420_nbytes(368, 226)
|
||||
y = np.frombuffer(out, dtype=np.uint8, count=368 * 226).reshape(226, 368)
|
||||
assert 0 not in y[0]
|
||||
assert 0 not in y[-1]
|
||||
|
||||
|
||||
def test_grid_i420_render_and_set_frame():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
buf = g.render_i420()
|
||||
assert len(buf) == i420_nbytes(1920, 1080)
|
||||
src = np.zeros((360, 640, 4), dtype=np.uint8)
|
||||
src[:] = (255, 0, 0, 255)
|
||||
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
|
||||
buf2 = g.render_i420()
|
||||
assert buf2 != buf
|
||||
g.clear("cam-07")
|
||||
assert g.render_i420() == buf
|
||||
|
||||
|
||||
def test_ensure_slot_binds_remote_to_empty_tile():
|
||||
g = CameraGrid(width=640, height=360, cols=2, rows=1, identities=["cam-01", "cam-02"])
|
||||
assert g.ensure_slot("alice") is True
|
||||
assert "alice" in g._tiles
|
||||
assert g.identities[0] == "alice"
|
||||
frame = np.zeros((90, 160, 4), dtype=np.uint8)
|
||||
g.set_frame("alice", frame)
|
||||
assert "alice" in g._live
|
||||
assert g.ensure_slot("alice") is True
|
||||
|
||||
|
||||
def test_grid_shape_shrinks_with_live_count():
|
||||
assert grid_shape(0) == (1, 1)
|
||||
assert grid_shape(1) == (1, 1)
|
||||
assert grid_shape(2) == (2, 1)
|
||||
assert grid_shape(4) == (2, 2)
|
||||
assert grid_shape(5) == (3, 2)
|
||||
assert grid_shape(6) == (3, 2)
|
||||
assert grid_shape(9) == (3, 3)
|
||||
assert grid_shape(12) == (4, 3)
|
||||
assert grid_shape(20) == (5, 4)
|
||||
cols, rows = grid_shape(5)
|
||||
assert cols * rows >= 5
|
||||
assert cols <= 5 and rows <= 4
|
||||
|
||||
|
||||
def test_set_identities_rebuilds_compact_layout():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
g.set_identities(["cam-08", "cam-09", "cam-10", "cam-11", "cam-13"])
|
||||
assert g.identities == ["cam-08", "cam-09", "cam-10", "cam-11", "cam-13"]
|
||||
assert (g.cols, g.rows) == (3, 2)
|
||||
assert g.tile_w > CameraGrid(width=1920, height=1080, cols=5, rows=4).tile_w
|
||||
g.set_identities(["cam-08", "cam-09"])
|
||||
assert (g.cols, g.rows) == (2, 1)
|
||||
assert g.identities == ["cam-08", "cam-09"]
|
||||
g.set_identities([])
|
||||
assert g.identities == []
|
||||
assert (g.cols, g.rows) == (1, 1)
|
||||
|
||||
|
||||
def test_status_overlay_replaces_mosaic_until_cleared():
|
||||
g = CameraGrid(width=320, height=180, cols=2, rows=1, identities=["cam-01", "cam-02"])
|
||||
mosaic = g.render()
|
||||
g.set_status(["LIVEKIT DISCONNECTED", "ws://10.200.0.21:7880", "host unreachable"])
|
||||
overlay = g.render()
|
||||
assert overlay != mosaic
|
||||
assert len(overlay) == 320 * 180 * 4
|
||||
i420 = g.render_i420()
|
||||
assert len(i420) == i420_nbytes(320, 180)
|
||||
g.set_status(None)
|
||||
assert g.render() == mosaic
|
||||
|
||||
|
||||
def test_format_livekit_status_and_errors():
|
||||
assert format_livekit_status(url="ws://10.200.0.21:7880", state="connected") == []
|
||||
connecting = format_livekit_status(url="ws://10.200.0.21:7880", state="connecting")
|
||||
assert connecting[0] == "CONNECTING TO LIVEKIT"
|
||||
assert "10.200.0.21:7880" in connecting[1]
|
||||
lines = format_livekit_status(
|
||||
url="ws://10.200.0.21:7880",
|
||||
state="disconnected",
|
||||
detail="host unreachable",
|
||||
retry_s=5,
|
||||
)
|
||||
assert lines[0] == "LIVEKIT DISCONNECTED"
|
||||
assert "host unreachable" in lines
|
||||
assert any("retry" in ln.lower() for ln in lines)
|
||||
assert "unreachable" in describe_livekit_error(
|
||||
OSError("No route to host")).lower() or "host" in describe_livekit_error(
|
||||
OSError("[Errno 113] No route to host")).lower()
|
||||
assert "timed out" in describe_livekit_error(TimeoutError("timed out")).lower()
|
||||
assert "refused" in describe_livekit_error(ConnectionRefusedError("Connection refused")).lower()
|
||||
|
||||
|
||||
def test_status_screen_disconnected_uses_composed_layout():
|
||||
img = status_screen(1920, 1080, [
|
||||
"LIVEKIT DISCONNECTED",
|
||||
"ws://10.200.0.21:7880",
|
||||
"host unreachable",
|
||||
"retrying in 5s",
|
||||
])
|
||||
assert img.shape == (1080, 1920, 4)
|
||||
assert img.dtype == np.uint8
|
||||
# Composed: background, accent rail, card — not a flat fill.
|
||||
assert len(np.unique(img.reshape(-1, 4), axis=0)) > 12
|
||||
bg = img[24, 80, :3]
|
||||
card = img[540, 960, :3]
|
||||
assert tuple(int(x) for x in bg) != tuple(int(x) for x in card)
|
||||
# Disconnected accent rail is warm amber (BGRA: R > G > B).
|
||||
stripe = img[540, :8, :3].mean(axis=0)
|
||||
b, g, r = (float(x) for x in stripe)
|
||||
assert r > g > b
|
||||
assert r > 180
|
||||
|
||||
|
||||
def test_status_screen_connecting_uses_cool_accent():
|
||||
img = status_screen(1920, 1080, [
|
||||
"CONNECTING TO LIVEKIT",
|
||||
"ws://10.200.0.21:7880",
|
||||
])
|
||||
stripe = img[540, :8, :3].mean(axis=0)
|
||||
b, g, r = (float(x) for x in stripe)
|
||||
assert b > r
|
||||
assert b > 160
|
||||
|
||||
@@ -19,3 +19,29 @@ def test_select_three_prefers_xe_hdmi():
|
||||
assert three[0].name == "HDMI-A-7" or three[0].connected
|
||||
ids = [o.connector_id for o in three]
|
||||
assert len(set(ids)) == 3
|
||||
|
||||
|
||||
def test_named_disconnected_falls_back_to_connected_xe():
|
||||
from drm_outputs import DrmOutput, pick_outputs
|
||||
|
||||
wall = DrmOutput(
|
||||
name="HDMI-A-7", sysfs="card0-HDMI-A-7", node="/dev/dri/card0",
|
||||
driver="xe", connector_id=538, status="connected", width=3840, height=2160,
|
||||
)
|
||||
stale_grid = DrmOutput(
|
||||
name="HDMI-A-3", sysfs="card1-HDMI-A-3", node="/dev/dri/card1",
|
||||
driver="i915", connector_id=530, status="disconnected",
|
||||
)
|
||||
stale_spk = DrmOutput(
|
||||
name="HDMI-A-5", sysfs="card0-HDMI-A-5", node="/dev/dri/card0",
|
||||
driver="xe", connector_id=520, status="disconnected",
|
||||
)
|
||||
picked = pick_outputs(
|
||||
[stale_grid, stale_spk, wall],
|
||||
n=2,
|
||||
names=["HDMI-A-3", "HDMI-A-5"],
|
||||
)
|
||||
assert len(picked) == 1
|
||||
assert picked[0].name == "HDMI-A-7"
|
||||
assert picked[0].connected
|
||||
assert picked[0].driver == "xe"
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
from encode_daemon import _one_enc
|
||||
|
||||
|
||||
def test_h264_uses_igpu_va_not_x264_or_arc_av1():
|
||||
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD129")
|
||||
s = " ".join(cmd)
|
||||
assert "varenderD129h264enc" in cmd
|
||||
assert "x264enc" not in cmd
|
||||
assert "vaav1enc" not in cmd
|
||||
assert "format=nv12" in s
|
||||
assert "key-int-max=1" in cmd
|
||||
assert "b-frames=0" in cmd
|
||||
|
||||
|
||||
def test_h264_factory_follows_render_node():
|
||||
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD128")
|
||||
assert "varenderD128h264enc" in cmd
|
||||
|
||||
|
||||
def test_c920_stays_fast_gop1():
|
||||
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD129")
|
||||
assert "bitrate=2000" in cmd
|
||||
assert "key-int-max=1" in cmd
|
||||
assert "target-usage=7" in cmd
|
||||
|
||||
|
||||
def test_rally_1080_uses_higher_bitrate_and_no_motion_trails():
|
||||
cmd = _one_enc(1920, 1088, 30, 2000, 3, 4, "h264", render="/dev/dri/renderD129")
|
||||
assert "bitrate=20000" in cmd
|
||||
assert "key-int-max=1" in cmd
|
||||
assert "target-usage=2" in cmd
|
||||
assert "b-frames=0" in cmd
|
||||
|
||||
|
||||
def test_rally_bitrate_rollbacks_via_hi_kbps():
|
||||
cmd = _one_enc(1920, 1088, 30, 2000, 3, 4, "h264",
|
||||
render="/dev/dri/renderD129", hi_kbps=12000)
|
||||
assert "bitrate=12000" in cmd
|
||||
|
||||
|
||||
def test_av1_stays_on_arc_vaav1enc():
|
||||
cmd = _one_enc(640, 368, 30, 2000, 3, 4, "av1")
|
||||
assert "vaav1enc" in cmd
|
||||
assert "varenderD129h264enc" not in cmd
|
||||
@@ -0,0 +1,135 @@
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
|
||||
def test_frame_roundtrip_stereo_hop():
|
||||
from enhance_ipc import dump_frame, load_frame
|
||||
|
||||
hop = 1600
|
||||
t = np.arange(hop, dtype=np.int16)
|
||||
pcm = np.stack([t, t + 1], axis=1)
|
||||
blob = dump_frame("cam-07", pcm)
|
||||
ident, out = load_frame(blob)
|
||||
assert ident == "cam-07"
|
||||
assert out.dtype == np.int16
|
||||
assert out.shape == (hop, 2)
|
||||
np.testing.assert_array_equal(out, pcm)
|
||||
|
||||
|
||||
def test_unix_client_gets_handler_output(tmp_path):
|
||||
from enhance_daemon import serve_unix
|
||||
from enhance_ipc import EnhanceClient
|
||||
|
||||
sock = str(tmp_path / "enhance.sock")
|
||||
stop = threading.Event()
|
||||
|
||||
def handler(ident, pcm):
|
||||
assert ident == "cam-03"
|
||||
# fake "enhance": average channels, invert
|
||||
if pcm.ndim == 2:
|
||||
mono = pcm.mean(axis=1)
|
||||
else:
|
||||
mono = pcm
|
||||
return (-mono).astype(np.int16)
|
||||
|
||||
t = threading.Thread(target=serve_unix, args=(sock, handler, stop), daemon=True)
|
||||
t.start()
|
||||
deadline = time.time() + 2
|
||||
while time.time() < deadline and not (tmp_path / "enhance.sock").exists():
|
||||
time.sleep(0.01)
|
||||
hop = 800
|
||||
pcm = np.stack([np.full(hop, 1000, np.int16), np.full(hop, 3000, np.int16)], axis=1)
|
||||
client = EnhanceClient(sock)
|
||||
try:
|
||||
out = client.process("cam-03", pcm)
|
||||
finally:
|
||||
client.close()
|
||||
stop.set()
|
||||
assert out.dtype == np.int16
|
||||
assert out.shape == (hop,)
|
||||
assert int(out[0]) == -2000
|
||||
|
||||
|
||||
def _start_echo_daemon(sock, stop):
|
||||
from enhance_daemon import serve_unix
|
||||
|
||||
def handler(ident, pcm):
|
||||
if pcm.ndim == 2:
|
||||
mono = pcm.mean(axis=1)
|
||||
else:
|
||||
mono = pcm
|
||||
return np.asarray(mono, dtype=np.int16)
|
||||
|
||||
t = threading.Thread(target=serve_unix, args=(sock, handler, stop), daemon=True)
|
||||
t.start()
|
||||
deadline = time.time() + 2
|
||||
while time.time() < deadline and not os.path.exists(sock):
|
||||
time.sleep(0.01)
|
||||
return t
|
||||
|
||||
|
||||
def test_audio_cleaner_remote_does_not_load_model(tmp_path):
|
||||
from audio_cleanup import AudioCleaner
|
||||
|
||||
sock = str(tmp_path / "enhance.sock")
|
||||
stop = threading.Event()
|
||||
_start_echo_daemon(sock, stop)
|
||||
try:
|
||||
c = AudioCleaner(
|
||||
mode="force", beamform="force", enhance_socket=sock,
|
||||
identity="cam-09")
|
||||
assert c._enhancer is None
|
||||
assert c._beamformer is None
|
||||
assert c.backend == "daemon"
|
||||
hop = 1600
|
||||
stereo = np.stack([
|
||||
np.full(hop, 400, np.int16),
|
||||
np.full(hop, 800, np.int16),
|
||||
], axis=1)
|
||||
out = c.process(stereo)
|
||||
assert out.shape == (hop,)
|
||||
assert out.dtype == np.int16
|
||||
assert int(out[0]) == 600
|
||||
finally:
|
||||
stop.set()
|
||||
|
||||
|
||||
def test_shared_engine_constructs_one_cleaner_for_two_cams():
|
||||
from audio_cleanup import AudioCleaner
|
||||
from enhance_daemon import SharedEnhanceEngine
|
||||
|
||||
n = {"c": 0}
|
||||
|
||||
def make():
|
||||
n["c"] += 1
|
||||
return AudioCleaner(mode="off", beamform="off")
|
||||
|
||||
eng = SharedEnhanceEngine(make_cleaner=make)
|
||||
hop = np.zeros(800, dtype=np.int16)
|
||||
out_a = eng.process("cam-01", hop)
|
||||
out_b = eng.process("cam-02", hop)
|
||||
out_a2 = eng.process("cam-01", hop)
|
||||
assert n["c"] == 1
|
||||
assert set(eng.identities()) == {"cam-01", "cam-02"}
|
||||
assert out_a.shape == (800,)
|
||||
assert out_b.shape == (800,)
|
||||
assert out_a2.shape == (800,)
|
||||
|
||||
|
||||
def test_wait_for_socket_false_then_true(tmp_path):
|
||||
from enhance_daemon import wait_for_socket, serve_unix
|
||||
|
||||
sock = str(tmp_path / "missing.sock")
|
||||
assert wait_for_socket(sock, timeout=0.15) is False
|
||||
stop = threading.Event()
|
||||
t = threading.Thread(
|
||||
target=serve_unix, args=(sock, lambda i, p: p, stop), daemon=True)
|
||||
t.start()
|
||||
try:
|
||||
assert wait_for_socket(sock, timeout=2.0) is True
|
||||
finally:
|
||||
stop.set()
|
||||
@@ -0,0 +1,23 @@
|
||||
from frame_shm import LatestFrameReader, LatestFrameWriter, frame_bytes
|
||||
|
||||
|
||||
def test_latest_frame_roundtrip(tmp_path):
|
||||
path = str(tmp_path / "cam-01.i420")
|
||||
w, h = 16, 16
|
||||
n = frame_bytes(w, h)
|
||||
writer = LatestFrameWriter(path, w, h)
|
||||
payload = bytes(range(256)) * (n // 256 + 1)
|
||||
payload = payload[:n]
|
||||
seq = writer.write(payload)
|
||||
assert seq == 1
|
||||
reader = LatestFrameReader(path)
|
||||
got = reader.latest()
|
||||
assert got is not None
|
||||
assert got[0] == 1
|
||||
assert got[1] == payload
|
||||
seq2 = writer.write(bytes([3]) * n)
|
||||
got2 = reader.latest()
|
||||
assert got2 is not None
|
||||
assert got2[0] == seq2
|
||||
assert got2[1][0] == 3
|
||||
writer.close()
|
||||
@@ -0,0 +1,66 @@
|
||||
from display_grid import CameraGrid
|
||||
from drm_outputs import DrmOutput
|
||||
from gst_sink import appsrc_cmd, va_grid_cmd
|
||||
|
||||
|
||||
def _out():
|
||||
return DrmOutput(
|
||||
name="HDMI-A-7",
|
||||
sysfs="card0-HDMI-A-7",
|
||||
node="/dev/dri/card0",
|
||||
driver="xe",
|
||||
connector_id=538,
|
||||
status="connected",
|
||||
)
|
||||
|
||||
|
||||
def test_appsrc_queue_buffers_default_two():
|
||||
cmd = appsrc_cmd(_out(), 1920, 1080, 30)
|
||||
assert "max-size-buffers=2" in cmd
|
||||
|
||||
|
||||
def test_appsrc_queue_buffers_one():
|
||||
cmd = appsrc_cmd(_out(), 1920, 1080, 30, queue_buffers=1)
|
||||
assert "max-size-buffers=1" in cmd
|
||||
assert "max-size-buffers=2" not in cmd
|
||||
|
||||
|
||||
def test_appsrc_i420_blocksize():
|
||||
cmd = appsrc_cmd(_out(), 1920, 1080, 30, pixel_format="i420")
|
||||
joined = " ".join(cmd)
|
||||
assert "format=i420" in joined
|
||||
assert "blocksize=3110400" in joined # 1920*1080*3//2
|
||||
assert "format=bgra" not in joined
|
||||
|
||||
|
||||
def test_va_grid_cmd_pins_arc_compositor_and_twenty_tiles():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
tile_fds = list(range(4, 24))
|
||||
cmd = va_grid_cmd(
|
||||
_out(), g, ingest_w=640, ingest_h=360, fps=15,
|
||||
tile_fds=tile_fds, chrome_fd=3, queue_buffers=1)
|
||||
joined = " ".join(cmd)
|
||||
assert "varenderD129compositor" in joined
|
||||
assert "varenderD129postproc" in joined
|
||||
assert "kmssink" in joined
|
||||
assert "connector-id=538" in joined
|
||||
assert "driver-name=xe" in joined
|
||||
assert joined.count("fdsrc") == 21 # chrome + 20 tiles
|
||||
x, y, w, h = g.inner_rect(0)
|
||||
assert f"xpos={x}" in joined
|
||||
assert f"ypos={y}" in joined
|
||||
assert "format=i420" in joined
|
||||
assert "blocksize=345600" in joined # 640*360*1.5
|
||||
|
||||
|
||||
def test_va_grid_cmd_h264_uses_arc_decoder():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
paths = [f"/tmp/cam-{i:02d}.h264" for i in range(1, 21)]
|
||||
cmd = va_grid_cmd(
|
||||
_out(), g, ingest_w=640, ingest_h=360, fps=15,
|
||||
tile_h264_paths=paths, chrome_fd=3)
|
||||
joined = " ".join(cmd)
|
||||
assert "varenderD129h264dec" in joined
|
||||
assert joined.count("h264parse") == 20
|
||||
assert "fdsrc" in joined # chrome only
|
||||
assert joined.count("filesrc") == 20
|
||||
@@ -0,0 +1,30 @@
|
||||
import os
|
||||
import stat
|
||||
|
||||
from h264_fifo import hold_fifo
|
||||
|
||||
|
||||
def test_hold_fifo_creates_rdwr_pipe(tmp_path):
|
||||
path = str(tmp_path / "cam-01.h264")
|
||||
fd = hold_fifo(path)
|
||||
try:
|
||||
st = os.stat(path)
|
||||
assert stat.S_ISFIFO(st.st_mode)
|
||||
os.write(fd, b"\x00\x01")
|
||||
assert os.read(fd, 2) == b"\x00\x01"
|
||||
finally:
|
||||
os.close(fd)
|
||||
|
||||
|
||||
def test_hold_fifo_does_not_replace_existing_fifo(tmp_path):
|
||||
path = str(tmp_path / "cam-02.h264")
|
||||
a = hold_fifo(path)
|
||||
ino = os.stat(path).st_ino
|
||||
b = hold_fifo(path)
|
||||
try:
|
||||
assert os.stat(path).st_ino == ino
|
||||
os.write(a, b"ab")
|
||||
assert os.read(b, 2) == b"ab"
|
||||
finally:
|
||||
os.close(a)
|
||||
os.close(b)
|
||||
@@ -0,0 +1,171 @@
|
||||
"""Hand-raise rule, latch, and yellow I420 chrome."""
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
|
||||
from display_grid import FRAME_RAISE, CameraGrid, bgra_to_i420, i420_nbytes
|
||||
from hand_raise import RaiseLatch, letterbox_rgb, wrist_is_raised
|
||||
|
||||
|
||||
def _kpts() -> np.ndarray:
|
||||
k = np.zeros((17, 3), dtype=np.float32)
|
||||
k[:, 2] = 0.8
|
||||
k[0] = [0.35, 0.50, 0.9] # nose
|
||||
k[5] = [0.50, 0.35, 0.9] # L shoulder
|
||||
k[6] = [0.50, 0.65, 0.9] # R shoulder
|
||||
k[7] = [0.62, 0.30, 0.9] # L elbow
|
||||
k[8] = [0.62, 0.70, 0.9]
|
||||
k[9] = [0.70, 0.28, 0.9] # L wrist down
|
||||
k[10] = [0.70, 0.72, 0.9]
|
||||
return k
|
||||
|
||||
|
||||
def test_wrist_down_is_not_raised():
|
||||
assert wrist_is_raised(_kpts()) is False
|
||||
|
||||
|
||||
def test_left_wrist_above_shoulder_is_raised():
|
||||
k = _kpts()
|
||||
k[7] = [0.32, 0.30, 0.9] # elbow up
|
||||
k[9] = [0.18, 0.32, 0.9] # wrist above shoulder (y 0.50)
|
||||
assert wrist_is_raised(k) is True
|
||||
|
||||
|
||||
def test_modest_wrist_lift_is_raised():
|
||||
"""Webcam framing: hand just above the shoulder still counts."""
|
||||
k = _kpts()
|
||||
k[7] = [0.46, 0.30, 0.9]
|
||||
k[9] = [0.43, 0.32, 0.9] # 0.07 above shoulder y=0.50
|
||||
assert wrist_is_raised(k) is True
|
||||
|
||||
|
||||
def test_wrist_on_face_rejected():
|
||||
k = _kpts()
|
||||
k[7] = [0.38, 0.48, 0.9]
|
||||
k[9] = [0.28, 0.50, 0.9] # up, but next to the nose
|
||||
assert wrist_is_raised(k) is False
|
||||
|
||||
|
||||
def test_wrist_above_head_without_shoulders():
|
||||
"""Close-up webcam: shoulders off-frame, hand still clearly up."""
|
||||
k = np.zeros((17, 3), dtype=np.float32)
|
||||
k[0] = [0.55, 0.50, 0.85] # nose
|
||||
k[9] = [0.18, 0.62, 0.7] # wrist high, not on the face
|
||||
assert wrist_is_raised(k) is True
|
||||
|
||||
|
||||
def test_low_score_not_raised():
|
||||
k = _kpts()
|
||||
k[:, 2] = 0.05
|
||||
assert wrist_is_raised(k) is False
|
||||
assert wrist_is_raised(np.zeros((17, 3), dtype=np.float32)) is False
|
||||
|
||||
|
||||
def test_raise_latch_needs_hits_then_holds():
|
||||
latch = RaiseLatch(hits=2, hold_s=0.4, window=2)
|
||||
t = 10.0
|
||||
assert latch.update(True, t) is False
|
||||
assert latch.update(True, t + 0.1) is True
|
||||
assert latch.update(False, t + 0.2) is True
|
||||
assert latch.update(False, t + 0.55) is False
|
||||
|
||||
|
||||
def test_raise_latch_survives_one_miss():
|
||||
latch = RaiseLatch(hits=3, hold_s=1.0, window=5)
|
||||
t = 10.0
|
||||
assert latch.update(True, t) is False
|
||||
assert latch.update(True, t + 0.1) is False
|
||||
assert latch.update(False, t + 0.2) is False
|
||||
assert latch.update(True, t + 0.3) is True
|
||||
assert latch.update(True, t + 0.4) is True
|
||||
assert latch.update(False, t + 0.5) is True
|
||||
|
||||
|
||||
def test_letterbox_rgb_pads_16x9():
|
||||
rgb = np.full((360, 640, 3), 200, dtype=np.uint8)
|
||||
out = letterbox_rgb(rgb, 192)
|
||||
assert out.shape == (192, 192, 3)
|
||||
assert int(out[0, 96, 0]) == 0
|
||||
assert int(out[96, 96, 0]) == 200
|
||||
thunder = letterbox_rgb(rgb, 256)
|
||||
assert thunder.shape == (256, 256, 3)
|
||||
|
||||
|
||||
def test_resolve_movenet_thunder_is_256():
|
||||
from pathlib import Path
|
||||
from hand_raise import resolve_movenet
|
||||
path, size, name = resolve_movenet("thunder")
|
||||
assert name == "thunder"
|
||||
assert size == 256
|
||||
assert path.name == "movenet_singlepose_thunder.onnx"
|
||||
path, size, name = resolve_movenet("lightning")
|
||||
assert name == "lightning" and size == 192
|
||||
assert Path(path).name == "movenet_singlepose_lightning.onnx"
|
||||
|
||||
|
||||
def test_set_highlight_paints_yellow_i420_border():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
src = np.zeros((360, 640, 4), dtype=np.uint8)
|
||||
src[:] = (40, 40, 40, 255)
|
||||
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
|
||||
before = np.frombuffer(g.render_i420(), dtype=np.uint8).copy()
|
||||
g.set_highlight("cam-07", True)
|
||||
assert g.highlighted("cam-07") is True
|
||||
after = np.frombuffer(g.render_i420(), dtype=np.uint8)
|
||||
assert not np.array_equal(before, after)
|
||||
idx = g.identities.index("cam-07")
|
||||
ox, oy = g.tile_origin(idx)
|
||||
yplane = after[: g.width * g.height].reshape(g.height, g.width)
|
||||
yv, _, _ = g._yuv_raise
|
||||
# Top edge of the tile chrome should be yellow-luma, not gray live.
|
||||
assert int(yplane[oy, ox + g.tile_w // 2]) == yv
|
||||
assert int(yplane[oy - 2, ox + g.tile_w // 2]) == yv
|
||||
ix, iy, iw, _ih = g.inner_rect(idx)
|
||||
assert int(yplane[iy, ix + iw // 2]) != yv
|
||||
g.set_highlight("cam-07", False)
|
||||
assert g.highlighted("cam-07") is False
|
||||
restored = np.frombuffer(g.render_i420(), dtype=np.uint8)
|
||||
assert int(restored[: g.width * g.height].reshape(g.height, g.width)[oy, ox + g.tile_w // 2]) != yv
|
||||
|
||||
|
||||
def test_clear_drops_yellow():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
src = np.zeros((360, 640, 4), dtype=np.uint8)
|
||||
g.set_frame_i420("cam-01", bgra_to_i420(src), 640, 360)
|
||||
g.set_highlight("cam-01", True)
|
||||
g.clear("cam-01")
|
||||
assert g.highlighted("cam-01") is False
|
||||
assert len(g.render_i420()) == i420_nbytes(1920, 1080)
|
||||
|
||||
|
||||
def test_fp32_movenet_sees_person_not_raised():
|
||||
"""INT8 Xenova weights scored ~0.05 on a real person; FP32 must lock on."""
|
||||
from pathlib import Path
|
||||
from hand_raise import HandRaiseMonitor, MODEL_PATH
|
||||
|
||||
fixture = Path(__file__).parent / "fixtures" / "movenet_warrior_192.npy"
|
||||
assert MODEL_PATH.name == "movenet_singlepose_lightning.onnx"
|
||||
rgb = np.load(fixture)
|
||||
mon = HandRaiseMonitor(enabled=True, model="lightning")
|
||||
assert mon.enabled
|
||||
k = mon.infer_rgb(rgb)
|
||||
assert float(k[:, 2].max()) > 0.4
|
||||
assert float(k[0, 2]) > 0.3 # nose
|
||||
assert wrist_is_raised(k) is False # warrior: arms out, not up
|
||||
|
||||
|
||||
def test_fp32_thunder_sees_person_not_raised():
|
||||
from pathlib import Path
|
||||
from hand_raise import HandRaiseMonitor, resolve_movenet
|
||||
|
||||
path, size, name = resolve_movenet("thunder")
|
||||
assert path.is_file()
|
||||
assert size == 256 and name == "thunder"
|
||||
rgb = np.load(Path(__file__).parent / "fixtures" / "movenet_warrior_192.npy")
|
||||
mon = HandRaiseMonitor(enabled=True, model="thunder")
|
||||
assert mon.enabled
|
||||
assert mon._input_size == 256
|
||||
k = mon.infer_rgb(rgb)
|
||||
assert float(k[:, 2].max()) > 0.4
|
||||
assert float(k[0, 2]) > 0.3
|
||||
assert wrist_is_raised(k) is False
|
||||
@@ -0,0 +1,47 @@
|
||||
from hop_shm import HopReader, HopWriter, hop_bytes
|
||||
|
||||
|
||||
def test_hop_ring_preserves_order(tmp_path):
|
||||
path = str(tmp_path / "cam-01.pcm")
|
||||
n = hop_bytes(16000, 2, 0.1)
|
||||
w = HopWriter(path, n, nslots=4)
|
||||
r = HopReader(path)
|
||||
for i in range(3):
|
||||
payload = bytes([i]) * n
|
||||
w.write(payload)
|
||||
got = [r.read() for _ in range(3)]
|
||||
assert got[0] is not None and got[0][0] == 0
|
||||
assert got[1] is not None and got[1][0] == 1
|
||||
assert got[2] is not None and got[2][0] == 2
|
||||
assert r.read() is None
|
||||
w.close()
|
||||
|
||||
|
||||
def test_hop_ring_skips_when_reader_lags(tmp_path):
|
||||
path = str(tmp_path / "cam-01.pcm")
|
||||
n = 32
|
||||
w = HopWriter(path, n, nslots=4)
|
||||
r = HopReader(path)
|
||||
for i in range(10):
|
||||
w.write(bytes([i]) * n)
|
||||
last = r.read()
|
||||
assert last is not None
|
||||
w.close()
|
||||
|
||||
|
||||
def test_reader_follows_writer_restart(tmp_path):
|
||||
"""Cameras-only restart zeros seq; the wall must not stay stuck on _last."""
|
||||
path = str(tmp_path / "cam-01.pcm")
|
||||
n = 32
|
||||
w = HopWriter(path, n, nslots=4)
|
||||
r = HopReader(path)
|
||||
for i in range(8):
|
||||
w.write(bytes([i]) * n)
|
||||
assert r.read() is not None
|
||||
w.close()
|
||||
w2 = HopWriter(path, n, nslots=4)
|
||||
w2.write(bytes([99]) * n)
|
||||
got = r.read()
|
||||
assert got is not None and got[0] == 99
|
||||
assert r.read() is None
|
||||
w2.close()
|
||||
@@ -0,0 +1,56 @@
|
||||
from i420_nv12 import (
|
||||
coded_dim,
|
||||
drop_va_pad_i420,
|
||||
i420_to_nv12_into,
|
||||
nv12_bytes,
|
||||
va_visible_height,
|
||||
)
|
||||
|
||||
def test_coded_dim_pads_up():
|
||||
assert coded_dim(360) == 368
|
||||
assert coded_dim(1080) == 1088
|
||||
assert coded_dim(640) == 640
|
||||
assert coded_dim(352) == 352
|
||||
|
||||
|
||||
def test_i420_to_nv12_pads_last_row():
|
||||
w, h = 640, 360
|
||||
ysz = w * h
|
||||
uv = (w // 2) * (h // 2)
|
||||
src = bytearray(ysz + 2 * uv)
|
||||
src[:ysz] = b"\x80" * ysz
|
||||
src[(h - 1) * w:ysz] = b"\x90" * w
|
||||
src[ysz:ysz + uv] = b"\x40" * uv
|
||||
src[ysz + uv:] = b"\xc0" * uv
|
||||
cw, ch = 640, 368
|
||||
out = bytearray(nv12_bytes(cw, ch))
|
||||
n = i420_to_nv12_into(src, w, h, out)
|
||||
assert n == nv12_bytes(cw, ch)
|
||||
assert out[0] == 0x80
|
||||
# cloned pad rows 360-367
|
||||
assert out[360 * cw] == 0x90
|
||||
assert out[367 * cw] == 0x90
|
||||
assert out[cw * ch] == 0x40
|
||||
assert out[cw * ch + 1] == 0xc0
|
||||
|
||||
|
||||
def test_drop_va_pad_restores_360():
|
||||
w, h = 640, 368
|
||||
vis = 360
|
||||
ysz = w * h
|
||||
uv = (w // 2) * (h // 2)
|
||||
src = bytearray(ysz + 2 * uv)
|
||||
src[: w * vis] = b"\x77" * (w * vis)
|
||||
src[w * vis:ysz] = b"\x11" * (w * 8)
|
||||
out, ow, oh = drop_va_pad_i420(src, w, h)
|
||||
assert (ow, oh) == (w, vis)
|
||||
assert len(out) == w * vis * 3 // 2
|
||||
assert out[0] == 0x77
|
||||
assert 0x11 not in out[: w * vis]
|
||||
|
||||
|
||||
def test_va_visible_height():
|
||||
assert va_visible_height(640, 368) == 360
|
||||
assert va_visible_height(1920, 1088) == 1080
|
||||
assert va_visible_height(640, 360) == 360
|
||||
assert va_visible_height(640, 352) == 352
|
||||
@@ -0,0 +1,99 @@
|
||||
from capture_va import (
|
||||
crop_scale_i420_drop_bottom,
|
||||
jpeg_mcu_bottom_crop,
|
||||
needs_jpeg_mcu_fix,
|
||||
repeat_i420_bottom_from_last_good,
|
||||
)
|
||||
|
||||
|
||||
def _i420(width, height, y=16, u=128, v=128):
|
||||
y_size = width * height
|
||||
uv = (width // 2) * (height // 2)
|
||||
buf = bytearray(y_size + 2 * uv)
|
||||
buf[:y_size] = bytes([y]) * y_size
|
||||
buf[y_size:y_size + uv] = bytes([u]) * uv
|
||||
buf[y_size + uv:] = bytes([v]) * uv
|
||||
return buf
|
||||
|
||||
|
||||
def test_repeat_clones_last_good_y_and_chroma_over_black_pad():
|
||||
w, h, crop = 640, 360, 8
|
||||
buf = _i420(w, h, y=16, u=128, v=128)
|
||||
last_good = h - crop - 1
|
||||
y_stride = w
|
||||
# last good Y row = 90; last good UV row = 40 (chroma 4:2:0)
|
||||
src = last_good * y_stride
|
||||
buf[src:src + y_stride] = bytes([90]) * y_stride
|
||||
uv_w = w // 2
|
||||
uv_h = h // 2
|
||||
uv_crop = crop // 2
|
||||
last_uv = uv_h - uv_crop - 1
|
||||
y_size = w * h
|
||||
u0 = y_size + last_uv * uv_w
|
||||
v0 = y_size + uv_w * uv_h + last_uv * uv_w
|
||||
buf[u0:u0 + uv_w] = bytes([40]) * uv_w
|
||||
buf[v0:v0 + uv_w] = bytes([200]) * uv_w
|
||||
|
||||
repeat_i420_bottom_from_last_good(buf, w, h, crop)
|
||||
|
||||
for r in range(h - crop, h):
|
||||
row = buf[r * y_stride:(r + 1) * y_stride]
|
||||
assert set(row) == {90}, r
|
||||
for r in range(uv_h - uv_crop, uv_h):
|
||||
u = buf[y_size + r * uv_w:y_size + (r + 1) * uv_w]
|
||||
v = buf[y_size + uv_w * uv_h + r * uv_w:
|
||||
y_size + uv_w * uv_h + (r + 1) * uv_w]
|
||||
assert set(u) == {40}, r
|
||||
assert set(v) == {200}, r
|
||||
# rows above the pad stay put
|
||||
mid = (h // 2) * y_stride
|
||||
assert set(buf[mid:mid + y_stride]) == {16}
|
||||
|
||||
|
||||
def test_360p_crops_partial_mcu_and_last_full_mcu():
|
||||
"""360 % 16 = 8 leftover; last full 16-row MCU is also garbage on 07/14/15."""
|
||||
assert jpeg_mcu_bottom_crop(360) == 24
|
||||
assert jpeg_mcu_bottom_crop(480) == 0
|
||||
assert jpeg_mcu_bottom_crop(352) == 0
|
||||
assert jpeg_mcu_bottom_crop(8) == 0
|
||||
buf = _i420(640, 480, y=77)
|
||||
before = bytes(buf)
|
||||
repeat_i420_bottom_from_last_good(buf, 640, 480, jpeg_mcu_bottom_crop(480))
|
||||
assert bytes(buf) == before
|
||||
|
||||
|
||||
def test_mcu_fix_cams_off():
|
||||
assert not needs_jpeg_mcu_fix("cam-07")
|
||||
assert not needs_jpeg_mcu_fix("cam-12")
|
||||
assert not needs_jpeg_mcu_fix("cam-14")
|
||||
assert not needs_jpeg_mcu_fix("cam-15")
|
||||
assert not needs_jpeg_mcu_fix("cam-01")
|
||||
assert not needs_jpeg_mcu_fix("cam-20")
|
||||
|
||||
|
||||
def test_crop_scale_drops_garbage_without_cloned_stripe():
|
||||
w, h = 640, 360
|
||||
crop = 24
|
||||
buf = _i420(w, h, y=40, u=100, v=140)
|
||||
y_size = w * h
|
||||
# garbage last 24 Y rows
|
||||
buf[(h - crop) * w:y_size] = bytes([250]) * (crop * w)
|
||||
out = crop_scale_i420_drop_bottom(buf, w, h, crop)
|
||||
assert len(out) == len(buf)
|
||||
y = memoryview(out)[:y_size]
|
||||
last = bytes(y[(h - 8) * w:h * w])
|
||||
assert last.count(250) == 0
|
||||
assert set(last) == {16}
|
||||
|
||||
|
||||
def test_live_path_clones_not_studio_black():
|
||||
w, h, crop = 640, 360, 24
|
||||
buf = _i420(w, h, y=40, u=100, v=140)
|
||||
last_good = h - crop - 1
|
||||
buf[last_good * w:(last_good + 1) * w] = bytes([90]) * w
|
||||
buf[(h - crop) * w:h * w] = bytes([250]) * (crop * w)
|
||||
repeat_i420_bottom_from_last_good(buf, w, h, crop)
|
||||
visible = bytes(buf[(h - crop) * w:h * w])
|
||||
assert set(visible) == {90}
|
||||
assert 16 not in visible
|
||||
assert 250 not in visible
|
||||
@@ -0,0 +1,64 @@
|
||||
"""Env-flagged low-latency knobs (hop/queue, hold, enhance scope)."""
|
||||
from latency import (
|
||||
audio_queue_size_ms,
|
||||
effective_enhance_mode,
|
||||
video_hold_seconds,
|
||||
video_stream_kwargs,
|
||||
)
|
||||
|
||||
|
||||
def test_audio_queue_defaults_to_legacy_floor():
|
||||
assert audio_queue_size_ms(0.1, audio_queue_ms=None) == 150
|
||||
assert audio_queue_size_ms(0.2, audio_queue_ms=None) == 250
|
||||
|
||||
|
||||
def test_audio_queue_env_overrides_floor():
|
||||
assert audio_queue_size_ms(0.1, audio_queue_ms=60) == 60
|
||||
assert audio_queue_size_ms(0.05, audio_queue_ms=60) == 60
|
||||
|
||||
|
||||
def test_video_hold_matches_hop_when_audio_present():
|
||||
assert video_hold_seconds(0.1, has_audio=True, video_hold_s=None) == 0.1
|
||||
assert video_hold_seconds(0.05, has_audio=True, video_hold_s=None) == 0.05
|
||||
|
||||
|
||||
def test_video_hold_zero_without_audio():
|
||||
assert video_hold_seconds(0.1, has_audio=False, video_hold_s=None) == 0.0
|
||||
assert video_hold_seconds(0.1, has_audio=False, video_hold_s=0.2) == 0.0
|
||||
|
||||
|
||||
def test_video_hold_env_overrides_including_zero():
|
||||
assert video_hold_seconds(0.1, has_audio=True, video_hold_s=0.0) == 0.0
|
||||
assert video_hold_seconds(0.1, has_audio=True, video_hold_s=0.05) == 0.05
|
||||
|
||||
|
||||
def test_enhance_speaker_only_rally_disables_c920s():
|
||||
assert effective_enhance_mode(
|
||||
"auto", "cam-01", speaker_only=True, speaker_camera="rally",
|
||||
) == "off"
|
||||
assert effective_enhance_mode(
|
||||
"auto", "rally", speaker_only=True, speaker_camera="rally",
|
||||
) == "auto"
|
||||
|
||||
|
||||
def test_enhance_speaker_only_inactive_leaves_mode():
|
||||
assert effective_enhance_mode(
|
||||
"auto", "cam-07", speaker_only=False, speaker_camera="cam-01",
|
||||
) == "auto"
|
||||
|
||||
|
||||
def test_enhance_speaker_only_ignored_when_speaker_is_active():
|
||||
assert effective_enhance_mode(
|
||||
"auto", "cam-07", speaker_only=True, speaker_camera="active",
|
||||
) == "auto"
|
||||
|
||||
|
||||
def test_video_stream_legacy_unbounded_native():
|
||||
assert video_stream_kwargs(capacity=0, format_name="") == {}
|
||||
|
||||
|
||||
def test_video_stream_latest_bgra():
|
||||
assert video_stream_kwargs(capacity=1, format_name="bgra") == {
|
||||
"capacity": 1,
|
||||
"format": "bgra",
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Latest-AU H.264 mmap: overwrite unread payloads; seq is monotonic."""
|
||||
from nal_shm import AU_MAX, NalReader, NalWriter
|
||||
|
||||
|
||||
def test_writer_seq_starts_at_one_and_increments(tmp_path):
|
||||
path = str(tmp_path / "cam-01.h264")
|
||||
w = NalWriter(path, max_payload=64)
|
||||
assert w.write(b"\x00\x00\x00\x01\x65" + b"x" * 10) == 1
|
||||
assert w.write(b"\x00\x00\x00\x01\x41" + b"y" * 10) == 2
|
||||
w.close()
|
||||
|
||||
|
||||
def test_reader_returns_latest_payload(tmp_path):
|
||||
path = str(tmp_path / "cam-01.h264")
|
||||
w = NalWriter(path, max_payload=64)
|
||||
w.write(b"first-access-unit-xxxx")
|
||||
w.write(b"second-access-unit-yyy")
|
||||
r = NalReader(path)
|
||||
seq, payload = r.read()
|
||||
assert seq == 2
|
||||
assert payload.startswith(b"second-access-unit")
|
||||
w.close()
|
||||
r.close()
|
||||
|
||||
|
||||
def test_reader_none_when_no_new_au(tmp_path):
|
||||
path = str(tmp_path / "cam-01.h264")
|
||||
w = NalWriter(path, max_payload=64)
|
||||
w.write(b"only-once")
|
||||
r = NalReader(path)
|
||||
assert r.read()[0] == 1
|
||||
assert r.read() is None
|
||||
w.close()
|
||||
r.close()
|
||||
|
||||
|
||||
def test_write_drops_oversize_payload(tmp_path):
|
||||
path = str(tmp_path / "cam-01.h264")
|
||||
w = NalWriter(path, max_payload=8)
|
||||
assert w.write(b"123456789") == 0
|
||||
w.close()
|
||||
|
||||
|
||||
def test_au_max_is_bounded():
|
||||
assert AU_MAX >= 64 * 1024
|
||||
assert AU_MAX <= 2 * 1024 * 1024
|
||||
@@ -0,0 +1,88 @@
|
||||
from participant_tags import (
|
||||
should_hide_participant,
|
||||
tags_from_fields,
|
||||
token_attributes,
|
||||
want_display_video,
|
||||
)
|
||||
|
||||
|
||||
def test_tags_from_attribute_tag():
|
||||
assert tags_from_fields({"tag": "uwh"}, "") == {"uwh"}
|
||||
|
||||
|
||||
def test_tags_from_comma_separated_tags_attr():
|
||||
assert tags_from_fields({"tags": "uwh,lab"}, None) == {"uwh", "lab"}
|
||||
|
||||
|
||||
def test_tags_from_json_metadata():
|
||||
assert tags_from_fields({}, '{"tag": "uwh"}') == {"uwh"}
|
||||
|
||||
|
||||
def test_hide_disabled_when_hide_tags_empty():
|
||||
assert should_hide_participant({"tag": "uwh"}, "", []) is False
|
||||
|
||||
|
||||
def test_hide_uwh_when_filter_enabled():
|
||||
assert should_hide_participant({"tag": "uwh"}, "", ["uwh"]) is True
|
||||
assert should_hide_participant({"tag": "nyc"}, "", ["uwh"]) is False
|
||||
assert should_hide_participant({}, "", ["uwh"]) is False
|
||||
|
||||
|
||||
def test_token_attributes_sets_tag_and_tags():
|
||||
assert token_attributes(["uwh"]) == {"tag": "uwh", "tags": "uwh"}
|
||||
assert token_attributes([]) == {}
|
||||
|
||||
|
||||
def test_want_display_keeps_cameras_when_filter_off():
|
||||
assert want_display_video(
|
||||
identity="cam-01",
|
||||
attributes={"tag": "uwh"},
|
||||
metadata="",
|
||||
hide_tags=[],
|
||||
participant_prefix="cam",
|
||||
display_identity="display-wall",
|
||||
) is True
|
||||
assert want_display_video(
|
||||
identity="alice",
|
||||
attributes={},
|
||||
metadata="",
|
||||
hide_tags=[],
|
||||
participant_prefix="cam",
|
||||
display_identity="display-wall",
|
||||
) is False
|
||||
assert want_display_video(
|
||||
identity="rally",
|
||||
attributes={},
|
||||
metadata="",
|
||||
hide_tags=[],
|
||||
participant_prefix="cam",
|
||||
display_identity="display-wall",
|
||||
speaker_identity="rally",
|
||||
) is True
|
||||
assert want_display_video(
|
||||
identity="rally",
|
||||
attributes={},
|
||||
metadata="",
|
||||
hide_tags=[],
|
||||
participant_prefix="cam",
|
||||
display_identity="display-wall",
|
||||
) is False
|
||||
|
||||
|
||||
def test_want_display_hides_uwh_and_shows_remote_when_filter_on():
|
||||
assert want_display_video(
|
||||
identity="cam-01",
|
||||
attributes={"tag": "uwh"},
|
||||
metadata="",
|
||||
hide_tags=["uwh"],
|
||||
participant_prefix="cam",
|
||||
display_identity="display-wall",
|
||||
) is False
|
||||
assert want_display_video(
|
||||
identity="alice",
|
||||
attributes={"tag": "nyc"},
|
||||
metadata="",
|
||||
hide_tags=["uwh"],
|
||||
participant_prefix="cam",
|
||||
display_identity="display-wall",
|
||||
) is True
|
||||
@@ -0,0 +1,209 @@
|
||||
"""Person-aware background blur: composite, hold, I420, encode path."""
|
||||
import numpy as np
|
||||
|
||||
from portrait import (
|
||||
MaskHold,
|
||||
apply_i420,
|
||||
apply_portrait,
|
||||
bgr_to_i420,
|
||||
composite_bgr,
|
||||
i420_nbytes,
|
||||
i420_source_path,
|
||||
i420_to_bgr,
|
||||
person_present,
|
||||
soft_blur_bgr,
|
||||
)
|
||||
|
||||
|
||||
def _checker(h: int, w: int) -> np.ndarray:
|
||||
img = np.zeros((h, w, 3), dtype=np.uint8)
|
||||
img[::4, :] = 255
|
||||
img[:, ::4] = 255
|
||||
return img
|
||||
|
||||
|
||||
def test_composite_keeps_person_pixels():
|
||||
h, w = 80, 128
|
||||
bg = _checker(h, w)
|
||||
person = bg.copy()
|
||||
person[16:64, 32:96] = (40, 180, 40)
|
||||
mask = np.zeros((h, w), dtype=np.float32)
|
||||
mask[16:64, 32:96] = 1.0
|
||||
out = composite_bgr(person, soft_blur_bgr(person, 9), mask)
|
||||
roi = out[24:56, 40:88]
|
||||
assert abs(int(roi[:, :, 1].mean()) - 180) < 8
|
||||
assert abs(int(roi[:, :, 0].mean()) - 40) < 8
|
||||
|
||||
|
||||
def test_composite_blurs_background_checker():
|
||||
h, w = 80, 128
|
||||
bg = _checker(h, w)
|
||||
person = bg.copy()
|
||||
person[16:64, 32:96] = (40, 180, 40)
|
||||
mask = np.zeros((h, w), dtype=np.float32)
|
||||
mask[16:64, 32:96] = 1.0
|
||||
sharp_var = float(bg[0:12, :].var())
|
||||
out = composite_bgr(person, soft_blur_bgr(person, 9), mask)
|
||||
blur_var = float(out[0:12, :].var())
|
||||
assert sharp_var > 1000
|
||||
assert blur_var < sharp_var * 0.5
|
||||
|
||||
|
||||
def test_empty_mask_is_fully_blurred():
|
||||
img = _checker(40, 64)
|
||||
mask = np.zeros((40, 64), dtype=np.float32)
|
||||
out = apply_portrait(img, mask)
|
||||
assert person_present(mask) is False
|
||||
assert float(out.var()) < float(img.var()) * 0.5
|
||||
assert not np.array_equal(out, img)
|
||||
|
||||
|
||||
def test_tiny_blob_is_not_a_person():
|
||||
mask = np.zeros((40, 64), dtype=np.float32)
|
||||
mask[10:12, 10:12] = 0.9
|
||||
assert person_present(mask) is False
|
||||
|
||||
|
||||
def test_mask_hold_keeps_last_on_miss():
|
||||
hold = MaskHold(hold_s=0.5, ema=1.0)
|
||||
mask = np.ones((8, 8), dtype=np.float32)
|
||||
got = hold.update(mask, True, 1.0)
|
||||
assert got is not None
|
||||
missed = hold.update(None, False, 1.2)
|
||||
assert missed is not None
|
||||
assert float(missed.mean()) == 1.0
|
||||
assert hold.update(None, False, 1.6) is None
|
||||
|
||||
|
||||
def test_ema_smooths_mask_flicker():
|
||||
hold = MaskHold(hold_s=1.0, ema=0.5)
|
||||
a = np.zeros((4, 4), dtype=np.float32)
|
||||
b = np.ones((4, 4), dtype=np.float32)
|
||||
hold.update(a, True, 0.0)
|
||||
out = hold.update(b, True, 0.1)
|
||||
assert 0.4 < float(out.mean()) < 0.6
|
||||
|
||||
|
||||
def test_i420_roundtrip_keeps_640x360():
|
||||
w, h = 640, 360
|
||||
bgr = np.zeros((h, w, 3), dtype=np.uint8)
|
||||
bgr[20:200, 80:400] = (30, 90, 200)
|
||||
payload = bgr_to_i420(bgr)
|
||||
assert len(payload) == i420_nbytes(w, h)
|
||||
mask = np.zeros((h, w), dtype=np.float32)
|
||||
mask[20:200, 80:400] = 1.0
|
||||
out = apply_i420(payload, w, h, mask, radius=9)
|
||||
assert len(out) == i420_nbytes(w, h)
|
||||
back = i420_to_bgr(out, w, h)
|
||||
assert back.shape == (h, w, 3)
|
||||
|
||||
|
||||
def test_missing_mask_blurs_whole_frame():
|
||||
w, h = 48, 32
|
||||
bgr = _checker(h, w)
|
||||
payload = bgr_to_i420(bgr)
|
||||
out = apply_i420(payload, w, h, None, radius=7)
|
||||
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
y1 = np.frombuffer(out, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
assert float(y1.var()) < float(y0.var()) * 0.6
|
||||
|
||||
|
||||
def test_apply_i420_blurs_y_background():
|
||||
w, h = 64, 48
|
||||
bgr = np.zeros((h, w, 3), dtype=np.uint8)
|
||||
bgr[::3, :] = 255
|
||||
bgr[:, ::3] = 255
|
||||
bgr[8:40, 16:48] = (20, 200, 20)
|
||||
payload = bgr_to_i420(bgr)
|
||||
mask = np.zeros((h, w), dtype=np.float32)
|
||||
mask[8:40, 16:48] = 1.0
|
||||
out = apply_i420(payload, w, h, mask, radius=9)
|
||||
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
y1 = np.frombuffer(out, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
assert float(y1[0:6, :].var()) < float(y0[0:6, :].var()) * 0.6
|
||||
assert abs(int(y1[12:36, 20:44].mean()) - int(y0[12:36, 20:44].mean())) < 12
|
||||
|
||||
|
||||
def test_c920_uses_portrait_dir_rally_stays_raw():
|
||||
assert i420_source_path("cam-01", "/run/raw", "/run/por") == "/run/por/cam-01.i420"
|
||||
assert i420_source_path("cam-15", "/run/raw", "/run/por") == "/run/por/cam-15.i420"
|
||||
assert i420_source_path("rally", "/run/raw", "/run/por") == "/run/raw/rally.i420"
|
||||
assert i420_source_path("cam-01", "/run/raw", "") == "/run/raw/cam-01.i420"
|
||||
|
||||
|
||||
def test_letterbox_mask_maps_back_to_16x9():
|
||||
from portrait import letterbox_bgr, unletterbox_mask
|
||||
|
||||
bgr = np.zeros((360, 640, 3), dtype=np.uint8)
|
||||
canvas, box = letterbox_bgr(bgr, 256)
|
||||
assert canvas.shape == (256, 256, 3)
|
||||
x0, y0, nw, nh = box
|
||||
assert nw > nh
|
||||
m = np.zeros((256, 256), dtype=np.float32)
|
||||
cy, cx = y0 + nh // 2, x0 + nw // 2
|
||||
m[cy - 2:cy + 3, cx - 2:cx + 3] = 1.0
|
||||
full = unletterbox_mask(m, box, 360, 640)
|
||||
assert full.shape == (360, 640)
|
||||
assert float(full[180, 320]) > 0.5
|
||||
assert float(full[2, 2]) < 0.1
|
||||
|
||||
|
||||
def test_selfie_cpu_sees_warrior_not_empty():
|
||||
import cv2
|
||||
from pathlib import Path
|
||||
|
||||
from portrait import PersonSeg
|
||||
|
||||
rgb = np.load(Path(__file__).parent / "fixtures" / "movenet_warrior_192.npy")
|
||||
bgr = cv2.cvtColor(rgb, cv2.COLOR_RGB2BGR)
|
||||
seg = PersonSeg()
|
||||
assert seg.device == "CPU"
|
||||
mask = seg.infer_bgr(bgr)
|
||||
assert person_present(mask)
|
||||
empty = np.full((360, 640, 3), 40, dtype=np.uint8)
|
||||
assert person_present(seg.infer_bgr(empty)) is False
|
||||
|
||||
|
||||
def test_pose_protect_keeps_wrists_and_face():
|
||||
from portrait import pose_protect_mask
|
||||
|
||||
k = np.zeros((17, 3), dtype=np.float32)
|
||||
k[0] = [0.30, 0.50, 0.9] # nose
|
||||
k[9] = [0.12, 0.18, 0.9] # L wrist
|
||||
k[10] = [0.12, 0.82, 0.9] # R wrist
|
||||
m = pose_protect_mask(100, 200, k)
|
||||
assert float(m[12, 36]) > 0.7
|
||||
assert float(m[12, 164]) > 0.7
|
||||
assert float(m[30, 100]) > 0.7
|
||||
assert float(m[90, 10]) < 0.2
|
||||
|
||||
|
||||
def test_union_masks_keeps_face_and_body():
|
||||
from portrait import union_masks
|
||||
|
||||
body = np.zeros((40, 40), dtype=np.float32)
|
||||
body[20:36, 10:30] = 1.0
|
||||
face = np.zeros((40, 40), dtype=np.float32)
|
||||
face[4:14, 14:26] = 1.0
|
||||
out = union_masks(body, face)
|
||||
assert float(out[8, 20]) == 1.0
|
||||
assert float(out[28, 20]) == 1.0
|
||||
assert float(out[2, 2]) == 0.0
|
||||
|
||||
|
||||
def test_lighter_blur_keeps_more_background_detail():
|
||||
w, h = 64, 48
|
||||
bgr = np.zeros((h, w, 3), dtype=np.uint8)
|
||||
bgr[::3, :] = 255
|
||||
bgr[:, ::3] = 255
|
||||
payload = bgr_to_i420(bgr)
|
||||
mask = np.zeros((h, w), dtype=np.float32)
|
||||
mask[8:40, 16:48] = 1.0
|
||||
heavy = apply_i420(payload, w, h, mask, radius=15)
|
||||
light = apply_i420(payload, w, h, mask, radius=7)
|
||||
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
yh = np.frombuffer(heavy, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
yl = np.frombuffer(light, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
bg0 = float(y0[0:6, :].var())
|
||||
assert float(yl[0:6, :].var()) > float(yh[0:6, :].var())
|
||||
assert float(yl[0:6, :].var()) < bg0
|
||||
@@ -0,0 +1,87 @@
|
||||
"""SHM round-trip for portrait daemon with a fake mask."""
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
|
||||
from frame_shm import LatestFrameReader, LatestFrameWriter, wait_for_ready
|
||||
from portrait import bgr_to_i420, i420_nbytes, i420_to_bgr, person_present
|
||||
from portrait_daemon import CamSlot, tick_once
|
||||
|
||||
|
||||
def test_tick_blurs_bg_and_keeps_person(tmp_path):
|
||||
w, h = 64, 48
|
||||
bgr = np.zeros((h, w, 3), dtype=np.uint8)
|
||||
bgr[::3, :] = 255
|
||||
bgr[:, ::3] = 255
|
||||
bgr[8:40, 16:48] = (20, 200, 20)
|
||||
payload = bgr_to_i420(bgr)
|
||||
src = tmp_path / "raw" / "cam-01.i420"
|
||||
dst = tmp_path / "por" / "cam-01.i420"
|
||||
writer = LatestFrameWriter(str(src), w, h)
|
||||
writer.write(payload)
|
||||
mask = np.zeros((h, w), dtype=np.float32)
|
||||
mask[8:40, 16:48] = 1.0
|
||||
|
||||
def infer(_bgr):
|
||||
return mask
|
||||
|
||||
slot = CamSlot(
|
||||
identity="cam-01",
|
||||
width=w,
|
||||
height=h,
|
||||
reader=LatestFrameReader(str(src)),
|
||||
writer=LatestFrameWriter(str(dst), w, h),
|
||||
infer_hz=30.0,
|
||||
hold_s=0.8,
|
||||
blur_px=9,
|
||||
)
|
||||
n = tick_once([slot], infer, time.monotonic(), infer_budget_s=0.05)
|
||||
assert n == 1
|
||||
out_r = LatestFrameReader(str(dst))
|
||||
buf = bytearray(i420_nbytes(w, h))
|
||||
seq = out_r.copy_into(buf)
|
||||
assert seq == 1
|
||||
out = i420_to_bgr(buf, w, h)
|
||||
roi = out[12:36, 20:44]
|
||||
assert abs(int(roi[:, :, 1].mean()) - 200) < 25
|
||||
bg = out[0:6, :]
|
||||
assert float(bg.var()) < float(bgr[0:6, :].var()) * 0.6
|
||||
|
||||
|
||||
def test_tick_blurs_when_empty(tmp_path):
|
||||
w, h = 32, 32
|
||||
bgr = np.zeros((h, w, 3), dtype=np.uint8)
|
||||
bgr[::3, :] = 255
|
||||
bgr[:, ::3] = 255
|
||||
payload = bgr_to_i420(bgr)
|
||||
src = tmp_path / "raw" / "cam-02.i420"
|
||||
dst = tmp_path / "por" / "cam-02.i420"
|
||||
LatestFrameWriter(str(src), w, h).write(payload)
|
||||
|
||||
def infer(_bgr):
|
||||
return np.zeros((h, w), dtype=np.float32)
|
||||
|
||||
slot = CamSlot(
|
||||
identity="cam-02",
|
||||
width=w,
|
||||
height=h,
|
||||
reader=LatestFrameReader(str(src)),
|
||||
writer=LatestFrameWriter(str(dst), w, h),
|
||||
infer_hz=30.0,
|
||||
hold_s=0.2,
|
||||
blur_px=9,
|
||||
)
|
||||
tick_once([slot], infer, time.monotonic(), infer_budget_s=0.05)
|
||||
buf = bytearray(i420_nbytes(w, h))
|
||||
assert LatestFrameReader(str(dst)).copy_into(buf) == 1
|
||||
y0 = np.frombuffer(payload, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
y1 = np.frombuffer(buf, dtype=np.uint8, count=w * h).reshape(h, w)
|
||||
assert float(y1.var()) < float(y0.var()) * 0.6
|
||||
assert person_present(np.zeros((h, w), dtype=np.float32)) is False
|
||||
|
||||
|
||||
def test_wait_ready_helper(tmp_path):
|
||||
p = tmp_path / "ready"
|
||||
assert wait_for_ready(str(p), timeout=0.05) is False
|
||||
p.write_text("ok\n")
|
||||
assert wait_for_ready(str(p), timeout=0.2) is True
|
||||
@@ -0,0 +1,36 @@
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
|
||||
from discovery import Camera
|
||||
from rally import dante_capture_card, discover_rally, rally_video_node
|
||||
from run import fleet_cameras
|
||||
from usb_ids import RALLY_IDENTITY
|
||||
|
||||
_HAVE_RALLY = rally_video_node() is not None
|
||||
|
||||
|
||||
@pytest.mark.skipif(not _HAVE_RALLY, reason="Rally V-R0010 not on USB")
|
||||
def test_rally_node_is_pinned_or_uvc():
|
||||
node = rally_video_node()
|
||||
assert node is not None
|
||||
assert Path(node).exists()
|
||||
|
||||
|
||||
@pytest.mark.skipif(not _HAVE_RALLY, reason="Rally V-R0010 not on USB")
|
||||
def test_discover_rally_not_a_cam_slot():
|
||||
cam = discover_rally()
|
||||
assert cam is not None
|
||||
assert cam.identity == RALLY_IDENTITY
|
||||
assert cam.width == 1920
|
||||
assert cam.height == 1080
|
||||
assert cam.identity != "cam-01"
|
||||
card = dante_capture_card()
|
||||
assert cam.audio_card == card
|
||||
|
||||
|
||||
def test_fleet_cameras_drops_rally():
|
||||
cams = [
|
||||
Camera(index=0, identity="cam-01", video_device="/dev/cam01", audio_card=1),
|
||||
Camera(index=20, identity="rally", video_device="/dev/rally", audio_card=0),
|
||||
]
|
||||
assert [c.identity for c in fleet_cameras(cams)] == ["cam-01"]
|
||||
@@ -0,0 +1,172 @@
|
||||
"""YuNet face gates for Rally follow — reject ceiling/stand/noise."""
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
|
||||
from rally_follow import FaceGate, accept_yunet_face, select_yunet_face
|
||||
|
||||
|
||||
def _face(
|
||||
*,
|
||||
x=400.0,
|
||||
y=200.0,
|
||||
w=80.0,
|
||||
h=100.0,
|
||||
score=0.93,
|
||||
re=None,
|
||||
le=None,
|
||||
nose=None,
|
||||
rm=None,
|
||||
lm=None,
|
||||
) -> np.ndarray:
|
||||
"""YuNet 15-vector: box + 5 landmarks + score, in 960x540 detect space."""
|
||||
if re is None:
|
||||
re = (x + 0.30 * w, y + 0.38 * h)
|
||||
if le is None:
|
||||
le = (x + 0.70 * w, y + 0.38 * h)
|
||||
if nose is None:
|
||||
nose = (x + 0.50 * w, y + 0.55 * h)
|
||||
if rm is None:
|
||||
rm = (x + 0.35 * w, y + 0.78 * h)
|
||||
if lm is None:
|
||||
lm = (x + 0.65 * w, y + 0.78 * h)
|
||||
row = np.zeros(15, dtype=np.float32)
|
||||
row[0:4] = (x, y, w, h)
|
||||
row[4:6] = re
|
||||
row[6:8] = le
|
||||
row[8:10] = nose
|
||||
row[10:12] = rm
|
||||
row[12:14] = lm
|
||||
row[14] = score
|
||||
return row
|
||||
|
||||
|
||||
FW, FH = 960, 540
|
||||
|
||||
|
||||
def test_good_face_accepted():
|
||||
assert accept_yunet_face(_face(), FW, FH) is True
|
||||
|
||||
|
||||
def test_tiny_blob_rejected():
|
||||
# Live FP: 25x34 @ 0.805, nose not between the eyes.
|
||||
row = _face(x=512, y=250, w=25, h=34, score=0.805,
|
||||
re=(528.5, 264), le=(532.1, 263), nose=(536.1, 271),
|
||||
rm=(529.7, 277), lm=(532.1, 278))
|
||||
assert accept_yunet_face(row, FW, FH) is False
|
||||
|
||||
|
||||
def test_low_score_rejected():
|
||||
assert accept_yunet_face(_face(score=0.61), FW, FH) is False
|
||||
|
||||
|
||||
def test_ceiling_band_rejected():
|
||||
row = _face(x=400, y=2, w=80, h=100)
|
||||
assert accept_yunet_face(row, FW, FH) is False
|
||||
|
||||
|
||||
def test_stand_band_rejected():
|
||||
row = _face(x=400, y=470, w=80, h=100)
|
||||
assert accept_yunet_face(row, FW, FH) is False
|
||||
|
||||
|
||||
def test_nose_not_between_eyes_rejected():
|
||||
row = _face(re=(430, 238), le=(450, 238), nose=(470, 255))
|
||||
assert accept_yunet_face(row, FW, FH) is False
|
||||
|
||||
|
||||
def test_upside_down_landmarks_rejected():
|
||||
row = _face(
|
||||
re=(424, 270), le=(456, 270), nose=(440, 240),
|
||||
rm=(428, 220), lm=(452, 220),
|
||||
)
|
||||
assert accept_yunet_face(row, FW, FH) is False
|
||||
|
||||
|
||||
def test_select_skips_large_invalid_for_smaller_valid():
|
||||
junk = _face(x=10, y=2, w=80, h=100, score=0.99) # ceiling
|
||||
good = _face(x=400, y=200, w=80, h=100, score=0.88)
|
||||
picked = select_yunet_face(np.stack([junk, good]), FW, FH)
|
||||
assert picked is not None
|
||||
assert float(picked[0]) == 400.0
|
||||
|
||||
|
||||
def test_select_empty():
|
||||
assert select_yunet_face(None, FW, FH) is None
|
||||
assert select_yunet_face(np.zeros((0, 15)), FW, FH) is None
|
||||
|
||||
|
||||
def test_face_gate_needs_confirm():
|
||||
gate = FaceGate(need=3, max_jump=0.18)
|
||||
box = (400, 200, 80, 100)
|
||||
assert gate.update(box, 960, 540) is None
|
||||
assert gate.update(box, 960, 540) is None
|
||||
assert gate.update(box, 960, 540) == box
|
||||
|
||||
|
||||
def test_face_gate_jump_resets():
|
||||
gate = FaceGate(need=3, max_jump=0.18)
|
||||
a = (400, 200, 80, 100)
|
||||
b = (800, 400, 80, 100)
|
||||
assert gate.update(a, 960, 540) is None
|
||||
assert gate.update(a, 960, 540) is None
|
||||
assert gate.update(b, 960, 540) is None # new target
|
||||
assert gate.update(b, 960, 540) is None
|
||||
assert gate.update(b, 960, 540) == b
|
||||
|
||||
|
||||
def test_face_gate_loss_clears():
|
||||
gate = FaceGate(need=2, max_jump=0.18)
|
||||
box = (400, 200, 80, 100)
|
||||
gate.update(box, 960, 540)
|
||||
assert gate.update(box, 960, 540) == box
|
||||
assert gate.update(None, 960, 540) is None
|
||||
assert gate.update(box, 960, 540) is None
|
||||
|
||||
|
||||
def test_profile_face_accepted():
|
||||
"""Head turned: eyes cluster, nose not between them — still a face."""
|
||||
row = _face(
|
||||
x=400, y=200, w=80, h=100, score=0.90,
|
||||
re=(418, 238), le=(426, 237),
|
||||
nose=(410, 255),
|
||||
rm=(416, 278), lm=(428, 277),
|
||||
)
|
||||
assert accept_yunet_face(row, FW, FH) is True
|
||||
|
||||
|
||||
def test_head_hold_keeps_box_on_roi_motion():
|
||||
from rally_follow import HeadHold
|
||||
hold = HeadHold(hold_s=1.0)
|
||||
box = (400, 200, 80, 100)
|
||||
t = 10.0
|
||||
assert hold.update(box, moving=False, now=t) == box
|
||||
assert hold.update(None, moving=True, now=t + 0.3) == box
|
||||
assert hold.update(None, moving=False, now=t + 0.4) == box # coast
|
||||
assert hold.update(None, moving=False, now=t + 2.0) is None
|
||||
|
||||
|
||||
def test_roi_motion_ignores_outside_last_box():
|
||||
from rally_follow import roi_motion
|
||||
prev = np.zeros((540, 960), dtype=np.uint8)
|
||||
now = prev.copy()
|
||||
now[10:40, 10:40] = 200 # motion far from the face
|
||||
assert roi_motion(prev, now, (400, 200, 80, 100)) is False
|
||||
now2 = prev.copy()
|
||||
now2[210:260, 420:460] = 200
|
||||
assert roi_motion(prev, now2, (400, 200, 80, 100)) is True
|
||||
|
||||
|
||||
def test_stale_hold_freezes_does_not_keep_panning():
|
||||
"""Last image-space box is not a pan target — that derails as the camera moves."""
|
||||
from rally_follow import lock_action, steer_speeds
|
||||
left = (80, 400, 160, 200)
|
||||
pan, tilt = steer_speeds(left, 1920, 1080, deadband=0.14)
|
||||
assert pan == -1
|
||||
assert lock_action(confirmed=left, held=left) == "steer"
|
||||
assert lock_action(confirmed=None, held=left) == "freeze"
|
||||
assert lock_action(confirmed=None, held=None) == "lost"
|
||||
cx = 1920 / 2 - 40
|
||||
mid = (int(cx - 80), 440, 160, 200)
|
||||
pan, tilt = steer_speeds(mid, 1920, 1080, deadband=0.14)
|
||||
assert pan == 0 and tilt == 0
|
||||
@@ -0,0 +1,11 @@
|
||||
from rally_follow import video_cmd
|
||||
|
||||
|
||||
def test_video_cmd_without_preview_has_no_kmssink():
|
||||
cmd = video_cmd("/dev/rally", 1920, 1080, 30, 8, preview=None)
|
||||
s = " ".join(cmd)
|
||||
assert "kmssink" not in s
|
||||
assert "tee" not in cmd
|
||||
assert "fdsink" in cmd
|
||||
assert "fd=8" in cmd
|
||||
assert "device=/dev/rally" in s
|
||||
@@ -0,0 +1,103 @@
|
||||
from config import Config
|
||||
from discovery import Camera
|
||||
from run import worker_args_for
|
||||
import json
|
||||
import jwt
|
||||
|
||||
|
||||
def _cfg(**kw) -> Config:
|
||||
return Config(url="ws://127.0.0.1:7880", api_key="k", api_secret="s", **kw)
|
||||
|
||||
|
||||
def _cam(identity: str = "cam-07") -> Camera:
|
||||
return Camera(index=0, identity=identity, video_device="/dev/cam07", audio_card=2)
|
||||
|
||||
|
||||
def test_worker_args_legacy_omits_optional_latency_flags():
|
||||
args = worker_args_for(_cfg(), _cam())
|
||||
assert "--audio-queue-ms" not in args
|
||||
assert "--video-hold-s" not in args
|
||||
assert "--split-executor" not in args
|
||||
assert args[args.index("--enhance-mode") + 1] == "auto"
|
||||
|
||||
|
||||
def test_worker_args_latency_flags_and_non_speaker_enhance_off():
|
||||
cfg = _cfg(
|
||||
audio_queue_ms=60,
|
||||
video_hold_s=0.0,
|
||||
worker_split_executor=True,
|
||||
enhance_speaker_only=True,
|
||||
display_speaker_camera="cam-01",
|
||||
enhance_mode="auto",
|
||||
)
|
||||
args = worker_args_for(cfg, _cam("cam-07"))
|
||||
assert args[args.index("--audio-queue-ms") + 1] == "60"
|
||||
assert args[args.index("--video-hold-s") + 1] == "0.0"
|
||||
assert "--split-executor" in args
|
||||
assert args[args.index("--enhance-mode") + 1] == "off"
|
||||
assert args[args.index("--beamform-mode") + 1] == "off"
|
||||
|
||||
|
||||
def test_worker_args_speaker_keeps_enhance_mode():
|
||||
cfg = _cfg(
|
||||
enhance_mode="force",
|
||||
enhance_speaker_only=True,
|
||||
display_speaker_camera="cam-07",
|
||||
)
|
||||
args = worker_args_for(cfg, _cam("cam-07"))
|
||||
assert args[args.index("--enhance-mode") + 1] == "force"
|
||||
assert args[args.index("--beamform-mode") + 1] == "auto"
|
||||
|
||||
|
||||
def test_worker_token_carries_uwh_tag():
|
||||
args = worker_args_for(_cfg(), _cam("cam-07"))
|
||||
token = args[args.index("--token") + 1]
|
||||
claims = jwt.decode(token, "s", algorithms=["HS256"])
|
||||
assert claims["attributes"]["tag"] == "uwh"
|
||||
assert claims["attributes"]["tags"] == "uwh"
|
||||
|
||||
|
||||
def test_worker_args_passes_enhance_socket():
|
||||
args = worker_args_for(_cfg(enhance_socket="/tmp/e.sock"), _cam())
|
||||
assert args[args.index("--enhance-socket") + 1] == "/tmp/e.sock"
|
||||
|
||||
|
||||
def test_worker_args_omits_enhance_socket_when_empty():
|
||||
args = worker_args_for(_cfg(enhance_socket=""), _cam())
|
||||
assert "--enhance-socket" not in args
|
||||
|
||||
|
||||
def test_worker_args_passes_video_shm():
|
||||
args = worker_args_for(_cfg(), _cam("cam-07"))
|
||||
assert args[args.index("--video-shm") + 1] == "/run/livekit-cameras/raw/cam-07.i420"
|
||||
|
||||
|
||||
def test_worker_args_omits_video_shm_when_capture_daemon_off():
|
||||
args = worker_args_for(_cfg(capture_daemon=False), _cam())
|
||||
assert "--video-shm" not in args
|
||||
|
||||
|
||||
def test_publisher_args_single_process_and_speaker_only():
|
||||
from run import publisher_args_for
|
||||
cfg = _cfg(enhance_speaker_only=True, display_speaker_camera="cam-01")
|
||||
args = publisher_args_for(cfg, [_cam("cam-07"), _cam("cam-01")])
|
||||
assert "--enhance-speaker-only" in args
|
||||
assert "--audio-gate" not in args
|
||||
payload = json.loads(args[args.index("--cameras-json") + 1])
|
||||
assert {p["identity"] for p in payload} == {"cam-07", "cam-01"}
|
||||
assert all("token" in p for p in payload)
|
||||
assert args[args.index("--video-shm-dir") + 1] == "/run/livekit-cameras/raw"
|
||||
assert args[args.index("--audio-shm-dir") + 1] == "/run/livekit-cameras/pcm"
|
||||
|
||||
|
||||
def test_publisher_args_rally_keeps_1080p_dims():
|
||||
from run import publisher_args_for
|
||||
rally = Camera(index=20, identity="rally", video_device="/dev/rally",
|
||||
audio_card=0, width=1920, height=1080)
|
||||
args = publisher_args_for(_cfg(), [_cam("cam-01"), rally])
|
||||
payload = json.loads(args[args.index("--cameras-json") + 1])
|
||||
by_id = {p["identity"]: p for p in payload}
|
||||
assert by_id["rally"]["width"] == 1920
|
||||
assert by_id["rally"]["height"] == 1080
|
||||
assert by_id["cam-01"]["width"] is None
|
||||
|
||||
@@ -0,0 +1,199 @@
|
||||
"""Green speaking chrome on the KMS grid (LiveKit active-speaker)."""
|
||||
import numpy as np
|
||||
|
||||
from display_grid import FRAME_RAISE, FRAME_SPEAK, CameraGrid, bgra_to_i420, bgra_to_yuv_pixel
|
||||
|
||||
|
||||
def test_set_speakers_paints_green_i420_border():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
src = np.zeros((360, 640, 4), dtype=np.uint8)
|
||||
src[:] = (40, 40, 40, 255)
|
||||
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
|
||||
before = np.frombuffer(g.render_i420(), dtype=np.uint8).copy()
|
||||
g.set_speakers(["cam-07"])
|
||||
assert g.speaking("cam-07") is True
|
||||
after = np.frombuffer(g.render_i420(), dtype=np.uint8)
|
||||
assert not np.array_equal(before, after)
|
||||
idx = g.identities.index("cam-07")
|
||||
ox, oy = g.tile_origin(idx)
|
||||
yplane = after[: g.width * g.height].reshape(g.height, g.width)
|
||||
yv, _, _ = bgra_to_yuv_pixel(FRAME_SPEAK)
|
||||
assert int(yplane[oy, ox + g.tile_w // 2]) == yv
|
||||
# Extra thickness lives in the gutter, not on the feed.
|
||||
assert int(yplane[oy - 2, ox + g.tile_w // 2]) == yv
|
||||
ix, iy, iw, _ih = g.inner_rect(idx)
|
||||
assert int(yplane[iy, ix + iw // 2]) != yv
|
||||
|
||||
|
||||
def test_speaking_overrides_yellow_then_yellow_returns():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
src = np.zeros((360, 640, 4), dtype=np.uint8)
|
||||
g.set_frame_i420("cam-07", bgra_to_i420(src), 640, 360)
|
||||
g.set_highlight("cam-07", True)
|
||||
g.set_speakers(["cam-07"])
|
||||
idx = g.identities.index("cam-07")
|
||||
ox, oy = g.tile_origin(idx)
|
||||
yplane = np.frombuffer(g.render_i420(), dtype=np.uint8)[: g.width * g.height].reshape(
|
||||
g.height, g.width)
|
||||
y_speak, _, _ = bgra_to_yuv_pixel(FRAME_SPEAK)
|
||||
y_raise, _, _ = bgra_to_yuv_pixel(FRAME_RAISE)
|
||||
assert int(yplane[oy, ox + g.tile_w // 2]) == y_speak
|
||||
g.set_speakers([])
|
||||
yplane = np.frombuffer(g.render_i420(), dtype=np.uint8)[: g.width * g.height].reshape(
|
||||
g.height, g.width)
|
||||
assert g.speaking("cam-07") is False
|
||||
assert g.highlighted("cam-07") is True
|
||||
assert int(yplane[oy, ox + g.tile_w // 2]) == y_raise
|
||||
|
||||
|
||||
def test_clear_drops_speaking():
|
||||
g = CameraGrid(width=1920, height=1080, cols=5, rows=4)
|
||||
src = np.zeros((360, 640, 4), dtype=np.uint8)
|
||||
g.set_frame_i420("cam-01", bgra_to_i420(src), 640, 360)
|
||||
g.set_speakers(["cam-01"])
|
||||
g.clear("cam-01")
|
||||
assert g.speaking("cam-01") is False
|
||||
|
||||
|
||||
class _P:
|
||||
def __init__(self, identity, keep=True):
|
||||
self.identity = identity
|
||||
self.keep = keep
|
||||
|
||||
|
||||
def test_active_speaker_ids_keeps_order_and_filter():
|
||||
from display_grid import active_speaker_ids, primary_grid_speaker
|
||||
speakers = [_P("cam-07"), _P("rally", False), _P("cam-03"), _P("")]
|
||||
assert active_speaker_ids(speakers, want=lambda p: p.keep) == ["cam-07", "cam-03"]
|
||||
ids = ["rally", "cam-03", "cam-07"]
|
||||
assert primary_grid_speaker(ids, ["cam-01", "cam-03", "cam-07"]) == ["cam-03"]
|
||||
assert primary_grid_speaker(["rally"], ["cam-01"]) == []
|
||||
|
||||
|
||||
def test_select_grid_speakers_drops_floor_noise():
|
||||
from display_grid import select_grid_speakers
|
||||
assert select_grid_speakers(
|
||||
[("cam-01", -34.0), ("cam-02", -36.0)], min_db=-24.0, margin_db=6.0,
|
||||
) == []
|
||||
|
||||
|
||||
def test_select_grid_speakers_bleed_keeps_louder_only():
|
||||
from display_grid import select_grid_speakers
|
||||
assert select_grid_speakers(
|
||||
[("cam-07", -12.0), ("cam-08", -22.0)], min_db=-24.0, margin_db=6.0,
|
||||
) == ["cam-07"]
|
||||
|
||||
|
||||
def test_select_grid_speakers_two_talkers_both_kept():
|
||||
from display_grid import select_grid_speakers
|
||||
assert select_grid_speakers(
|
||||
[("cam-07", -12.0), ("cam-08", -14.0)], min_db=-24.0, margin_db=6.0,
|
||||
) == ["cam-07", "cam-08"]
|
||||
|
||||
|
||||
def test_select_grid_speakers_normal_speech_above_mic_floor():
|
||||
"""Conversational C920 speech is often ~-24 dBFS; absolute -20 missed it."""
|
||||
from display_grid import select_grid_speakers
|
||||
floors = {"cam-07": -34.0, "cam-08": -34.0}
|
||||
assert select_grid_speakers(
|
||||
[("cam-07", -24.0), ("cam-08", -33.0)],
|
||||
min_db=-40.0, margin_db=6.0, floors=floors, rise_db=6.0,
|
||||
) == ["cam-07"]
|
||||
|
||||
|
||||
def test_select_grid_speakers_two_talkers_each_above_own_floor():
|
||||
from display_grid import select_grid_speakers
|
||||
floors = {"cam-07": -34.0, "cam-08": -36.0}
|
||||
assert select_grid_speakers(
|
||||
[("cam-07", -24.0), ("cam-08", -26.0)],
|
||||
min_db=-40.0, margin_db=6.0, floors=floors, rise_db=6.0,
|
||||
) == ["cam-07", "cam-08"]
|
||||
|
||||
|
||||
def test_select_quiet_mic_speech_not_cut_by_abs_min():
|
||||
"""Quiet C920 idle is ~-45; +3 dB speech is still below -40 dBFS."""
|
||||
from display_grid import select_grid_speakers
|
||||
floors = {"cam-10": -45.0, "cam-01": -37.0}
|
||||
assert select_grid_speakers(
|
||||
[("cam-10", -41.5), ("cam-01", -36.5)],
|
||||
min_db=-40.0, margin_db=6.0, floors=floors, rise_db=3.0,
|
||||
) == ["cam-10"]
|
||||
|
||||
|
||||
def test_update_noise_floor_tracks_idle_not_speech():
|
||||
from display_grid import update_noise_floor
|
||||
fl = update_noise_floor(-34.0, -33.5, speaking=False, rise_db=4.0)
|
||||
assert -34.0 < fl < -33.5
|
||||
stuck = update_noise_floor(-34.0, -20.0, speaking=True, rise_db=4.0)
|
||||
assert stuck == -34.0
|
||||
|
||||
|
||||
def test_confirm_speakers_needs_two_hops_new_talker():
|
||||
from display_grid import confirm_speakers
|
||||
first, counts = confirm_speakers({}, ["cam-07"], need=2)
|
||||
assert first == []
|
||||
second, counts = confirm_speakers(counts, ["cam-07"], need=2)
|
||||
assert second == ["cam-07"]
|
||||
# already on: one hop of continued speech keeps chrome
|
||||
stay, _ = confirm_speakers(counts, ["cam-07"], need=2, already=["cam-07"])
|
||||
assert stay == ["cam-07"]
|
||||
# a one-hop idle spike on another seat does not light
|
||||
spike, _ = confirm_speakers({}, ["cam-01"], need=2, already=["cam-07"])
|
||||
assert spike == []
|
||||
|
||||
|
||||
def test_waveform_similarity_high_for_delayed_copy():
|
||||
from display_grid import waveform_similarity
|
||||
t = np.linspace(0, 1, 1600, endpoint=False)
|
||||
a = np.sin(2 * np.pi * 220 * t).astype(np.float32)
|
||||
b = np.roll(a * 0.4, 80)
|
||||
assert waveform_similarity(a, b) > 0.85
|
||||
|
||||
|
||||
def test_waveform_similarity_low_for_independent_signals():
|
||||
from display_grid import waveform_similarity
|
||||
rng = np.random.default_rng(0)
|
||||
a = rng.normal(size=1600).astype(np.float32)
|
||||
b = rng.normal(size=1600).astype(np.float32)
|
||||
assert waveform_similarity(a, b) < 0.35
|
||||
|
||||
|
||||
def test_collapse_bleed_drops_similar_quieter_mic():
|
||||
from display_grid import collapse_bleed
|
||||
t = np.linspace(0, 1, 1600, endpoint=False)
|
||||
src = np.sin(2 * np.pi * 180 * t).astype(np.float32)
|
||||
waves = {"cam-07": src, "cam-08": np.roll(src * 0.3, 40)}
|
||||
levels = {"cam-07": -12.0, "cam-08": -18.0}
|
||||
assert collapse_bleed(["cam-07", "cam-08"], waves, levels, corr_min=0.4) == ["cam-07"]
|
||||
|
||||
|
||||
def test_collapse_bleed_keeps_two_unlike_talkers():
|
||||
from display_grid import collapse_bleed
|
||||
t = np.linspace(0, 1, 1600, endpoint=False)
|
||||
a = np.sin(2 * np.pi * 180 * t).astype(np.float32)
|
||||
b = np.sin(2 * np.pi * 440 * t + 1.3).astype(np.float32)
|
||||
waves = {"cam-07": a, "cam-08": b}
|
||||
levels = {"cam-07": -12.0, "cam-08": -13.0}
|
||||
assert collapse_bleed(["cam-07", "cam-08"], waves, levels, corr_min=0.6) == [
|
||||
"cam-07", "cam-08",
|
||||
]
|
||||
|
||||
|
||||
def test_hold_speakers_adds_second_talker_immediately():
|
||||
from display_grid import hold_speakers
|
||||
cur, until = hold_speakers(
|
||||
["cam-07"], ["cam-07", "cam-08"], prev_peak=-12.0, new_peak=-12.0,
|
||||
now=10.0, hold_until=10.5, hold_s=0.6, switch_db=3.0)
|
||||
assert cur == ["cam-07", "cam-08"]
|
||||
|
||||
|
||||
def test_hold_speakers_ignores_similar_bleed_flicker():
|
||||
from display_grid import hold_speakers
|
||||
cur, until = hold_speakers(
|
||||
["cam-07"], ["cam-08"], prev_peak=-12.0, new_peak=-12.5,
|
||||
now=10.2, hold_until=10.8, hold_s=0.6, switch_db=3.0)
|
||||
assert cur == ["cam-07"]
|
||||
cur, _ = hold_speakers(
|
||||
["cam-07"], ["cam-08"], prev_peak=-12.0, new_peak=-8.0,
|
||||
now=10.2, hold_until=10.8, hold_s=0.6, switch_db=3.0)
|
||||
assert cur == ["cam-08"]
|
||||
@@ -0,0 +1,24 @@
|
||||
import jwt
|
||||
|
||||
from tokens import make_token
|
||||
|
||||
|
||||
def test_make_token_includes_uwh_attributes():
|
||||
token = make_token(
|
||||
"ws://127.0.0.1:7880", "devkey", "secret", "cameras", "cam-01",
|
||||
attributes={"tag": "uwh", "tags": "uwh"},
|
||||
)
|
||||
claims = jwt.decode(token, "secret", algorithms=["HS256"])
|
||||
assert claims["sub"] == "cam-01"
|
||||
video = claims.get("video") or {}
|
||||
assert video.get("canPublish") is True
|
||||
assert video.get("canSubscribe") is False
|
||||
assert claims["attributes"]["tag"] == "uwh"
|
||||
assert claims["attributes"]["tags"] == "uwh"
|
||||
|
||||
|
||||
def test_make_token_omits_attributes_when_empty():
|
||||
token = make_token(
|
||||
"ws://127.0.0.1:7880", "devkey", "secret", "cameras", "cam-01")
|
||||
claims = jwt.decode(token, "secret", algorithms=["HS256"])
|
||||
assert "attributes" not in claims or not claims.get("attributes")
|
||||
@@ -0,0 +1,15 @@
|
||||
from usb_ids import (
|
||||
HUDDLY_VID,
|
||||
LOGITECH_VID,
|
||||
is_c920_pid,
|
||||
is_fleet_video,
|
||||
is_rally_pid,
|
||||
)
|
||||
|
||||
|
||||
def test_c920_is_fleet_huddly_and_rally_are_not():
|
||||
assert is_c920_pid(0x08E5)
|
||||
assert is_fleet_video(LOGITECH_VID, 0x08E5)
|
||||
assert not is_fleet_video(HUDDLY_VID, 0x0021)
|
||||
assert not is_fleet_video(LOGITECH_VID, 0x0881)
|
||||
assert is_rally_pid(0x0881)
|
||||
@@ -0,0 +1,16 @@
|
||||
from uvc_exposure import pin_cmd
|
||||
|
||||
|
||||
def test_pin_cmd_order_manual_then_time_then_dynfps():
|
||||
cmd = pin_cmd("/dev/cam07")
|
||||
assert cmd[:3] == ["v4l2-ctl", "-d", "/dev/cam07"]
|
||||
assert cmd[3] == "--set-ctrl=auto_exposure=1"
|
||||
assert cmd[4] == "--set-ctrl=exposure_time_absolute=333"
|
||||
assert cmd[5] == "--set-ctrl=exposure_dynamic_framerate=0"
|
||||
|
||||
|
||||
def test_pin_cmd_auto_skips_inactive_time():
|
||||
cmd = pin_cmd("/dev/cam07", auto_exposure=3, dynamic_framerate=1)
|
||||
assert "--set-ctrl=auto_exposure=3" in cmd
|
||||
assert "--set-ctrl=exposure_dynamic_framerate=1" in cmd
|
||||
assert not any(c.startswith("--set-ctrl=exposure_time_absolute=") for c in cmd)
|
||||
@@ -0,0 +1,38 @@
|
||||
from vaapi_pin import (
|
||||
classify_render_nodes,
|
||||
igpu_render,
|
||||
pin_encode_env,
|
||||
should_pin_vaapi,
|
||||
)
|
||||
|
||||
|
||||
def test_classify_by_pci_even_if_card_numbers_swap():
|
||||
def pci_id_of(node: str) -> str:
|
||||
return {
|
||||
"/dev/dri/renderD128": "0xe20b",
|
||||
"/dev/dri/renderD129": "0xa780",
|
||||
}[node]
|
||||
|
||||
c = classify_render_nodes(
|
||||
["/dev/dri/renderD128", "/dev/dri/renderD129"],
|
||||
pci_id_of=pci_id_of,
|
||||
)
|
||||
assert c["arc"] == "/dev/dri/renderD128"
|
||||
assert c["igpu"] == "/dev/dri/renderD129"
|
||||
|
||||
|
||||
def test_pin_encode_env_points_at_igpu_not_arc():
|
||||
env = pin_encode_env({}, igpu="/dev/dri/renderD129")
|
||||
assert env["LIBVA_DRIVER_NAME"] == "iHD"
|
||||
assert env["LIVEKIT_ENCODE_RENDER"] == "/dev/dri/renderD129"
|
||||
assert env.get("LIBVA_DRM_DEVICE") != "/dev/dri/renderD128"
|
||||
|
||||
|
||||
def test_should_not_pin_vaapi_when_preencoded():
|
||||
assert should_pin_vaapi(preencoded=True) is False
|
||||
assert should_pin_vaapi(preencoded=False) is True
|
||||
|
||||
|
||||
def test_igpu_render_matches_live_a780():
|
||||
node = igpu_render()
|
||||
assert node.startswith("/dev/dri/renderD")
|
||||
@@ -0,0 +1,69 @@
|
||||
"""VideoSource in-flight cap + VAAPI stale-frame drop (encoder behind)."""
|
||||
from latency import (
|
||||
should_drop_encoder_queue_depth,
|
||||
should_drop_stale_vaapi_frame,
|
||||
video_capture_gate_acquire,
|
||||
video_capture_gate_release,
|
||||
)
|
||||
|
||||
|
||||
def test_vaapi_keeps_fresh_delta_frame():
|
||||
# 30 fps, 33 ms old — still inside the 3-frame budget.
|
||||
assert should_drop_stale_vaapi_frame(
|
||||
now_us=1_000_000, capture_us=967_000, fps=30, max_frames_behind=3,
|
||||
is_keyframe=False,
|
||||
) is False
|
||||
|
||||
|
||||
def test_vaapi_drops_delta_older_than_three_frame_periods():
|
||||
# 30 fps × 3 = 100 ms. 150 ms on EncoderQueue → drop before ToI420.
|
||||
assert should_drop_stale_vaapi_frame(
|
||||
now_us=1_000_000, capture_us=850_000, fps=30, max_frames_behind=3,
|
||||
is_keyframe=False,
|
||||
) is True
|
||||
|
||||
|
||||
def test_vaapi_never_drops_keyframe():
|
||||
assert should_drop_stale_vaapi_frame(
|
||||
now_us=1_000_000, capture_us=1, fps=30, max_frames_behind=3,
|
||||
is_keyframe=True,
|
||||
) is False
|
||||
|
||||
|
||||
def test_vaapi_missing_timestamp_is_not_dropped():
|
||||
assert should_drop_stale_vaapi_frame(
|
||||
now_us=1_000_000, capture_us=0, fps=30, max_frames_behind=3,
|
||||
is_keyframe=False,
|
||||
) is False
|
||||
|
||||
|
||||
def test_gate_admits_until_max_in_flight():
|
||||
gate = {"in_flight": 0, "dropped": 0, "max_in_flight": 3}
|
||||
assert video_capture_gate_acquire(gate) is True
|
||||
assert video_capture_gate_acquire(gate) is True
|
||||
assert video_capture_gate_acquire(gate) is True
|
||||
assert gate["in_flight"] == 3
|
||||
assert video_capture_gate_acquire(gate) is False
|
||||
assert gate["dropped"] == 1
|
||||
assert gate["in_flight"] == 3
|
||||
|
||||
|
||||
def test_gate_release_allows_another_capture():
|
||||
gate = {"in_flight": 3, "dropped": 0, "max_in_flight": 3}
|
||||
video_capture_gate_release(gate)
|
||||
assert gate["in_flight"] == 2
|
||||
assert video_capture_gate_acquire(gate) is True
|
||||
assert gate["in_flight"] == 3
|
||||
|
||||
|
||||
def test_queue_depth_keeps_slot_when_empty():
|
||||
assert should_drop_encoder_queue_depth(queued=0, max_queued=1) is False
|
||||
|
||||
|
||||
def test_queue_depth_drops_when_at_max():
|
||||
# Latest-frame cap: one in-flight OnFrame/Encode, overwrite don't push.
|
||||
assert should_drop_encoder_queue_depth(queued=1, max_queued=1) is True
|
||||
|
||||
|
||||
def test_queue_depth_drops_when_over_max():
|
||||
assert should_drop_encoder_queue_depth(queued=3, max_queued=1) is True
|
||||
@@ -0,0 +1,52 @@
|
||||
"""Publisher must not catch up frames and must recycle native encoders on RSS."""
|
||||
from latency import (
|
||||
next_video_capture,
|
||||
should_recycle_encoders,
|
||||
)
|
||||
|
||||
|
||||
def test_next_video_capture_sleeps_until_due():
|
||||
capture, due = next_video_capture(now=10.0, due=10.05, period=1 / 30)
|
||||
assert capture is False
|
||||
assert due == 10.05
|
||||
|
||||
|
||||
def test_next_video_capture_fires_on_due_and_advances_one_period():
|
||||
capture, due = next_video_capture(now=10.0, due=10.0, period=0.05)
|
||||
assert capture is True
|
||||
assert due == 10.05
|
||||
|
||||
|
||||
def test_next_video_capture_skips_catchup_when_behind():
|
||||
# 80 ms late at 30 fps would be two extra ticks if we kept adding period.
|
||||
capture, due = next_video_capture(now=10.08, due=10.0, period=1 / 30)
|
||||
assert capture is True
|
||||
assert due == 10.08 + (1 / 30)
|
||||
|
||||
|
||||
def test_should_recycle_when_anon_rss_exceeds_limit_and_interval_elapsed():
|
||||
assert should_recycle_encoders(
|
||||
rss_anon_kb=4_000_000, limit_kb=3_000_000,
|
||||
now=100.0, last_recycle_at=0.0, min_interval_s=60.0,
|
||||
) is True
|
||||
|
||||
|
||||
def test_should_not_recycle_below_limit():
|
||||
assert should_recycle_encoders(
|
||||
rss_anon_kb=1_000_000, limit_kb=3_000_000,
|
||||
now=100.0, last_recycle_at=0.0, min_interval_s=60.0,
|
||||
) is False
|
||||
|
||||
|
||||
def test_should_not_recycle_inside_min_interval():
|
||||
assert should_recycle_encoders(
|
||||
rss_anon_kb=5_000_000, limit_kb=3_000_000,
|
||||
now=100.0, last_recycle_at=80.0, min_interval_s=60.0,
|
||||
) is False
|
||||
|
||||
|
||||
def test_should_not_recycle_when_limit_disabled():
|
||||
assert should_recycle_encoders(
|
||||
rss_anon_kb=9_000_000, limit_kb=0,
|
||||
now=100.0, last_recycle_at=0.0, min_interval_s=60.0,
|
||||
) is False
|
||||
@@ -0,0 +1,25 @@
|
||||
from visca_xu_bridge import parse_visca, visca_frames
|
||||
|
||||
|
||||
def test_visca_split_frames():
|
||||
frames, rest = visca_frames(bytes([0x81, 0x01, 0x06, 0x04, 0xFF, 0x81]))
|
||||
assert frames == [bytes([0x81, 0x01, 0x06, 0x04, 0xFF])]
|
||||
assert rest == bytes([0x81])
|
||||
|
||||
|
||||
def test_visca_stop_and_move():
|
||||
stop = parse_visca(bytes([0x81, 0x01, 0x06, 0x01, 0x15, 0x15, 0x03, 0x03, 0xFF]))
|
||||
assert stop["action"] == "stop"
|
||||
left = parse_visca(bytes([0x81, 0x01, 0x06, 0x01, 0x08, 0x08, 0x01, 0x03, 0xFF]))
|
||||
assert left == {"action": "move", "pan": -1.0, "tilt": 0.0, "zoom": 0.0}
|
||||
up = parse_visca(bytes([0x81, 0x01, 0x06, 0x01, 0x08, 0x08, 0x03, 0x01, 0xFF]))
|
||||
assert up["tilt"] == 1.0
|
||||
|
||||
|
||||
def test_visca_zoom_home_preset():
|
||||
assert parse_visca(bytes([0x81, 0x01, 0x04, 0x07, 0x00, 0xFF]))["action"] == "zoom_stop"
|
||||
tele = parse_visca(bytes([0x81, 0x01, 0x04, 0x07, 0x27, 0xFF]))
|
||||
assert tele == {"action": "zoom", "zoom": 1.0}
|
||||
assert parse_visca(bytes([0x81, 0x01, 0x06, 0x04, 0xFF]))["action"] == "home"
|
||||
rec = parse_visca(bytes([0x81, 0x01, 0x04, 0x3F, 0x02, 0x01, 0xFF]))
|
||||
assert rec == {"action": "preset_recall", "index": 1}
|
||||
@@ -4,7 +4,8 @@ from livekit.api import AccessToken, VideoGrants
|
||||
|
||||
|
||||
def make_token(url: str, api_key: str, api_secret: str,
|
||||
room: str, identity: str) -> str:
|
||||
room: str, identity: str,
|
||||
attributes: dict[str, str] | None = None) -> str:
|
||||
token = (
|
||||
AccessToken(api_key, api_secret)
|
||||
.with_identity(identity)
|
||||
@@ -13,10 +14,12 @@ def make_token(url: str, api_key: str, api_secret: str,
|
||||
room_join=True,
|
||||
room=room,
|
||||
can_publish=True,
|
||||
can_subscribe=True,
|
||||
can_subscribe=False,
|
||||
can_publish_data=True,
|
||||
))
|
||||
)
|
||||
if attributes:
|
||||
token = token.with_attributes(attributes)
|
||||
return token.to_jwt()
|
||||
|
||||
|
||||
|
||||
+62
@@ -0,0 +1,62 @@
|
||||
"""USB identities for the fleet vs the (future) speaker Rally."""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
LOGITECH_VID = 0x046D
|
||||
# HD Pro C920 (this fleet) and the older 082d id.
|
||||
C920_PIDS = frozenset({0x08E5, 0x082D})
|
||||
# Logi Rally Camera 0881; Rally Bar / Mini / Huddle variants.
|
||||
RALLY_PIDS = frozenset({0x0881, 0x087C, 0x089B, 0x08D3})
|
||||
RALLY_IDENTITY = "rally"
|
||||
# Huddly IQ on the PCH/Dante hub — not a cam-NN slot.
|
||||
HUDDLY_VID = 0x2BD9
|
||||
HUDDLY_PIDS = frozenset({0x0021})
|
||||
|
||||
|
||||
def usb_vid_pid_from_sysfs(sysfs_device: str) -> tuple[int, int] | None:
|
||||
p = Path(sysfs_device)
|
||||
for _ in range(12):
|
||||
vend = p / "idVendor"
|
||||
prod = p / "idProduct"
|
||||
if vend.is_file() and prod.is_file():
|
||||
try:
|
||||
return int(vend.read_text().strip(), 16), int(prod.read_text().strip(), 16)
|
||||
except ValueError:
|
||||
return None
|
||||
if str(p).startswith("/sys/devices"):
|
||||
p = p.parent
|
||||
else:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def usb_vid_pid_of_video_node(video_node: str) -> tuple[int, int] | None:
|
||||
link = Path(f"/sys/class/video4linux/{Path(video_node).name}/device")
|
||||
try:
|
||||
real = str(link.resolve())
|
||||
except OSError:
|
||||
return None
|
||||
return usb_vid_pid_from_sysfs(real)
|
||||
|
||||
|
||||
def is_c920_pid(pid: int) -> bool:
|
||||
return int(pid) in C920_PIDS
|
||||
|
||||
|
||||
def is_huddly_iq(vid: int, pid: int) -> bool:
|
||||
return int(vid) == HUDDLY_VID and int(pid) in HUDDLY_PIDS
|
||||
|
||||
|
||||
def is_fleet_video(vid: int, pid: int) -> bool:
|
||||
"""True only for the 20 C920s. Rally, Huddly IQ, etc. stay out of cam-NN."""
|
||||
return int(vid) == LOGITECH_VID and int(pid) in C920_PIDS
|
||||
|
||||
|
||||
def is_rally_pid(pid: int) -> bool:
|
||||
return int(pid) in RALLY_PIDS
|
||||
|
||||
|
||||
def looks_like_rally_name(name: str) -> bool:
|
||||
n = (name or "").lower()
|
||||
return "rally" in n
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Pin Logitech C920 exposure so 360p stays 30 fps.
|
||||
|
||||
Aperture-priority + exposure_dynamic_framerate=1 lets the sensor sit at
|
||||
~66 ms / gain 255 and drop to 15 fps. Manual 333 (0.1 ms units = 1/30 s)
|
||||
with dyn-fps off holds the gst 30 fps request.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
log = logging.getLogger("cameras.uvc-exposure")
|
||||
|
||||
# UVC: 1 = Manual Mode, 3 = Aperture Priority. Time unit is 0.1 ms.
|
||||
AUTO_EXPOSURE_MANUAL = 1
|
||||
EXPOSURE_TIME_30FPS = 333
|
||||
|
||||
|
||||
def pin_cmd(
|
||||
device: str,
|
||||
*,
|
||||
auto_exposure: int = AUTO_EXPOSURE_MANUAL,
|
||||
exposure_time: int = EXPOSURE_TIME_30FPS,
|
||||
dynamic_framerate: int = 0,
|
||||
) -> list[str]:
|
||||
cmd = [
|
||||
"v4l2-ctl",
|
||||
"-d",
|
||||
str(device),
|
||||
f"--set-ctrl=auto_exposure={int(auto_exposure)}",
|
||||
f"--set-ctrl=exposure_dynamic_framerate={int(dynamic_framerate)}",
|
||||
]
|
||||
# Time is inactive under aperture-priority (3); setting it fails the whole pin.
|
||||
if int(auto_exposure) == AUTO_EXPOSURE_MANUAL:
|
||||
cmd.insert(4, f"--set-ctrl=exposure_time_absolute={int(exposure_time)}")
|
||||
return cmd
|
||||
|
||||
|
||||
def pin_from_env() -> tuple[int, int, int]:
|
||||
ae = int(os.environ.get("VIDEO_AUTO_EXPOSURE", str(AUTO_EXPOSURE_MANUAL)))
|
||||
exp = int(os.environ.get("VIDEO_EXPOSURE_TIME", str(EXPOSURE_TIME_30FPS)))
|
||||
dyn = int(os.environ.get("VIDEO_EXPOSURE_DYNAMIC", "0"))
|
||||
return ae, exp, dyn
|
||||
|
||||
|
||||
def pin_c920_exposure(device: str) -> None:
|
||||
ae, exp, dyn = pin_from_env()
|
||||
cmd = pin_cmd(device, auto_exposure=ae, exposure_time=exp, dynamic_framerate=dyn)
|
||||
try:
|
||||
r = subprocess.run(cmd, capture_output=True, text=True, timeout=5)
|
||||
except OSError as exc:
|
||||
log.warning("pin %s failed: %s", device, exc)
|
||||
return
|
||||
if r.returncode:
|
||||
log.warning("pin %s rc=%s %s", device, r.returncode, (r.stderr or "").strip())
|
||||
return
|
||||
log.info("pinned %s ae=%s time=%s dynfps=%s", device, ae, exp, dyn)
|
||||
|
||||
|
||||
def pin_devices(devices: list[str]) -> None:
|
||||
for dev in devices:
|
||||
pin_c920_exposure(dev)
|
||||
+104
@@ -0,0 +1,104 @@
|
||||
"""DRM render-node identity for this host.
|
||||
|
||||
Arc B580 PCI id 0xe20b (today often renderD128). iGPU UHD 770 PCI id 0xa780
|
||||
(today often renderD129). Card indexes swap; never hardcode which is Arc.
|
||||
Encode belongs on the iGPU. The Arc is kmssink wall only.
|
||||
|
||||
LiveKit PreEncoded publish does not need a VA bind-mount.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
log = logging.getLogger("cameras.vaapi")
|
||||
|
||||
ARC_PCI = "0xe20b"
|
||||
IGPU_PCI = "0xa780"
|
||||
|
||||
|
||||
def _pci_id_of_render(node: str) -> str:
|
||||
name = Path(node).name
|
||||
p = Path("/sys/class/drm") / name / "device" / "device"
|
||||
try:
|
||||
return p.read_text().strip().lower()
|
||||
except OSError:
|
||||
return ""
|
||||
|
||||
|
||||
def classify_render_nodes(
|
||||
nodes: list[str] | None = None,
|
||||
pci_id_of=None,
|
||||
) -> dict:
|
||||
reader = pci_id_of or _pci_id_of_render
|
||||
if nodes is None:
|
||||
dri = Path("/dev/dri")
|
||||
nodes = sorted(str(p) for p in dri.glob("renderD*")) if dri.is_dir() else []
|
||||
arc = igpu = None
|
||||
ids: dict[str, str] = {}
|
||||
for n in nodes:
|
||||
pid = (reader(n) or "").strip().lower()
|
||||
if not pid.startswith("0x"):
|
||||
try:
|
||||
pid = hex(int(pid, 16))
|
||||
except ValueError:
|
||||
pass
|
||||
ids[n] = pid
|
||||
if pid == ARC_PCI:
|
||||
arc = n
|
||||
elif pid == IGPU_PCI:
|
||||
igpu = n
|
||||
return {"arc": arc, "igpu": igpu, "ids": ids}
|
||||
|
||||
|
||||
def igpu_render() -> str:
|
||||
found = classify_render_nodes()["igpu"]
|
||||
if found:
|
||||
return found
|
||||
fallback = "/dev/dri/renderD129"
|
||||
log.warning("iGPU PCI %s not found; falling back to %s", IGPU_PCI, fallback)
|
||||
return fallback
|
||||
|
||||
|
||||
def arc_render() -> str:
|
||||
found = classify_render_nodes()["arc"]
|
||||
if found:
|
||||
return found
|
||||
fallback = "/dev/dri/renderD128"
|
||||
log.warning("Arc PCI %s not found; falling back to %s", ARC_PCI, fallback)
|
||||
return fallback
|
||||
|
||||
|
||||
def pin_encode_env(env: dict | None = None, igpu: str | None = None) -> dict:
|
||||
out = dict(os.environ if env is None else env)
|
||||
node = igpu or igpu_render()
|
||||
out["LIBVA_DRIVER_NAME"] = "iHD"
|
||||
out["GST_VA_ALL_DRIVERS"] = "1"
|
||||
out["LIVEKIT_ENCODE_RENDER"] = node
|
||||
out.pop("LIBVA_DRM_DEVICE", None)
|
||||
return out
|
||||
|
||||
|
||||
def should_pin_vaapi(preencoded: bool) -> bool:
|
||||
return not bool(preencoded)
|
||||
|
||||
|
||||
def pin_arc_vaapi_env(env: dict | None = None) -> dict:
|
||||
"""Deprecated: PreEncoded does not encode in libwebrtc."""
|
||||
out = dict(os.environ if env is None else env)
|
||||
out["LIBVA_DRIVER_NAME"] = "iHD"
|
||||
out["GST_VA_ALL_DRIVERS"] = "1"
|
||||
return out
|
||||
|
||||
|
||||
def apply_env(env: dict) -> None:
|
||||
for k, v in env.items():
|
||||
os.environ[k] = v
|
||||
|
||||
|
||||
def bind_arc_over_igpu_render(arc: str | None = None, igpu: str | None = None) -> bool:
|
||||
"""Deprecated leftover. Do not call from capture or publisher."""
|
||||
del arc, igpu
|
||||
log.warning("bind_arc_over_igpu_render is deprecated; skipped")
|
||||
return False
|
||||
@@ -0,0 +1,222 @@
|
||||
"""VISCA-IP façade for Logitech Rally (UVC + XU), not native VISCA.
|
||||
|
||||
Listens on TCP (default 127.0.0.1:5678), accepts raw VISCA frames ending in
|
||||
0xFF, and drives pan/tilt/zoom through v4l2. AutoPTZ should use backend
|
||||
visca_ip, address 127.0.0.1:5678, raw framing. visca_usb is pyserial — do
|
||||
not point it at this camera.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import select
|
||||
import socket
|
||||
import subprocess
|
||||
import threading
|
||||
from typing import Callable
|
||||
|
||||
log = logging.getLogger("cameras.visca_xu")
|
||||
|
||||
# Sony/PTZOptics ACK + completion for a command addressed as 81.
|
||||
_ACK = bytes([0x90, 0x41, 0xFF])
|
||||
_COMP = bytes([0x90, 0x51, 0xFF])
|
||||
_SYNTAX = bytes([0x90, 0x60, 0x02, 0xFF])
|
||||
|
||||
|
||||
def visca_frames(buf: bytes) -> tuple[list[bytes], bytes]:
|
||||
"""Split a byte stream into 0xFF-terminated frames; return (frames, rest)."""
|
||||
frames: list[bytes] = []
|
||||
start = 0
|
||||
for i, b in enumerate(buf):
|
||||
if b == 0xFF:
|
||||
frames.append(buf[start:i + 1])
|
||||
start = i + 1
|
||||
return frames, buf[start:]
|
||||
|
||||
|
||||
def parse_visca(frame: bytes) -> dict:
|
||||
"""Parse one VISCA command. Unknown frames get action='unknown'."""
|
||||
f = bytes(frame)
|
||||
if len(f) < 5 or f[-1] != 0xFF:
|
||||
return {"action": "invalid"}
|
||||
# 81 01 06 01 VV WW PP QQ FF pan/tilt drive
|
||||
if len(f) >= 9 and f[1:4] == bytes([0x01, 0x06, 0x01]):
|
||||
pan_dir, tilt_dir = f[6], f[7]
|
||||
pan = 0.0
|
||||
tilt = 0.0
|
||||
if pan_dir == 0x01:
|
||||
pan = -1.0
|
||||
elif pan_dir == 0x02:
|
||||
pan = 1.0
|
||||
if tilt_dir == 0x01:
|
||||
tilt = 1.0
|
||||
elif tilt_dir == 0x02:
|
||||
tilt = -1.0
|
||||
if pan_dir == 0x03 and tilt_dir == 0x03:
|
||||
return {"action": "stop"}
|
||||
return {"action": "move", "pan": pan, "tilt": tilt, "zoom": 0.0}
|
||||
# 81 01 04 07 ZZ FF zoom
|
||||
if len(f) >= 6 and f[1:4] == bytes([0x01, 0x04, 0x07]):
|
||||
z = f[4]
|
||||
if z == 0x00:
|
||||
return {"action": "zoom_stop"}
|
||||
if (z & 0xF0) == 0x20:
|
||||
return {"action": "zoom", "zoom": 1.0}
|
||||
if (z & 0xF0) == 0x30:
|
||||
return {"action": "zoom", "zoom": -1.0}
|
||||
return {"action": "zoom_stop"}
|
||||
# 81 01 06 04 FF home
|
||||
if f[1:4] == bytes([0x01, 0x06, 0x04]):
|
||||
return {"action": "home"}
|
||||
# 81 01 04 3F 01/02 MM FF preset set/recall
|
||||
if len(f) >= 7 and f[1:4] == bytes([0x01, 0x04, 0x3F]):
|
||||
if f[4] == 0x02:
|
||||
return {"action": "preset_recall", "index": int(f[5])}
|
||||
if f[4] == 0x01:
|
||||
return {"action": "preset_set", "index": int(f[5])}
|
||||
# inquiries: answer none (get_position returns None on AutoPTZ)
|
||||
if len(f) >= 5 and f[1] == 0x09:
|
||||
return {"action": "inquiry"}
|
||||
return {"action": "unknown"}
|
||||
|
||||
|
||||
class RallyV4L2:
|
||||
"""Thin v4l2-ctl wrapper. Rally exposes generic pan/tilt CIDs on this host."""
|
||||
|
||||
def __init__(self, device: str = "/dev/rally"):
|
||||
self.device = device
|
||||
|
||||
def _ctl(self, *args: str) -> str:
|
||||
r = subprocess.run(
|
||||
["v4l2-ctl", "-d", self.device, *args],
|
||||
capture_output=True, text=True, check=False)
|
||||
if r.returncode != 0:
|
||||
log.warning("v4l2-ctl rc=%s %s", r.returncode, r.stderr.strip())
|
||||
return r.stdout
|
||||
|
||||
def get(self) -> dict[str, int]:
|
||||
out = self._ctl("--get-ctrl=pan_absolute,tilt_absolute,zoom_absolute")
|
||||
vals: dict[str, int] = {}
|
||||
for line in out.splitlines():
|
||||
if ":" not in line:
|
||||
continue
|
||||
k, v = line.split(":", 1)
|
||||
try:
|
||||
vals[k.strip()] = int(v.strip())
|
||||
except ValueError:
|
||||
pass
|
||||
return vals
|
||||
|
||||
def set_speed(self, pan: int, tilt: int) -> None:
|
||||
pan = max(-1, min(1, int(pan)))
|
||||
tilt = max(-1, min(1, int(tilt)))
|
||||
self._ctl(f"--set-ctrl=pan_speed={pan},tilt_speed={tilt}")
|
||||
|
||||
def stop(self) -> None:
|
||||
self.set_speed(0, 0)
|
||||
|
||||
def nudge_zoom(self, direction: int, step: int = 20) -> None:
|
||||
cur = self.get().get("zoom_absolute", 100)
|
||||
nxt = max(100, min(1500, cur + (step if direction > 0 else -step)))
|
||||
self._ctl(f"--set-ctrl=zoom_absolute={nxt}")
|
||||
|
||||
def home(self) -> None:
|
||||
self.stop()
|
||||
self._ctl("--set-ctrl=pan_absolute=0,tilt_absolute=0,zoom_absolute=100")
|
||||
|
||||
def apply(self, cmd: dict) -> None:
|
||||
action = cmd.get("action")
|
||||
if action in ("stop", "zoom_stop"):
|
||||
self.stop()
|
||||
elif action == "move":
|
||||
pan = cmd.get("pan") or 0.0
|
||||
tilt = cmd.get("tilt") or 0.0
|
||||
self.set_speed(
|
||||
-1 if pan < -0.01 else (1 if pan > 0.01 else 0),
|
||||
-1 if tilt < -0.01 else (1 if tilt > 0.01 else 0),
|
||||
)
|
||||
elif action == "zoom":
|
||||
z = cmd.get("zoom") or 0.0
|
||||
if abs(z) < 0.01:
|
||||
return
|
||||
self.nudge_zoom(1 if z > 0 else -1)
|
||||
elif action == "home":
|
||||
self.home()
|
||||
|
||||
|
||||
def serve(
|
||||
host: str,
|
||||
port: int,
|
||||
apply: Callable[[dict], None],
|
||||
stop: threading.Event | None = None,
|
||||
) -> None:
|
||||
srv = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
srv.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
||||
srv.bind((host, port))
|
||||
srv.listen(4)
|
||||
srv.settimeout(0.5)
|
||||
log.info("VISCA-IP façade %s:%d (raw) -> Rally v4l2", host, port)
|
||||
clients: dict[socket.socket, bytearray] = {}
|
||||
try:
|
||||
while stop is None or not stop.is_set():
|
||||
rlist = [srv, *clients]
|
||||
try:
|
||||
ready, _, _ = select.select(rlist, [], [], 0.5)
|
||||
except InterruptedError:
|
||||
continue
|
||||
for s in ready:
|
||||
if s is srv:
|
||||
conn, addr = srv.accept()
|
||||
conn.setblocking(False)
|
||||
clients[conn] = bytearray()
|
||||
log.info("visca client %s", addr)
|
||||
continue
|
||||
try:
|
||||
data = s.recv(4096)
|
||||
except BlockingIOError:
|
||||
continue
|
||||
except OSError:
|
||||
data = b""
|
||||
if not data:
|
||||
clients.pop(s, None)
|
||||
s.close()
|
||||
continue
|
||||
clients[s].extend(data)
|
||||
frames, rest = visca_frames(bytes(clients[s]))
|
||||
clients[s] = bytearray(rest)
|
||||
for fr in frames:
|
||||
cmd = parse_visca(fr)
|
||||
if cmd.get("action") in ("invalid", "inquiry", "unknown"):
|
||||
if cmd.get("action") == "invalid":
|
||||
s.sendall(_SYNTAX)
|
||||
continue
|
||||
try:
|
||||
apply(cmd)
|
||||
except Exception: # noqa: BLE001
|
||||
log.exception("apply failed %s", cmd)
|
||||
try:
|
||||
s.sendall(_ACK + _COMP)
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
for c in list(clients):
|
||||
c.close()
|
||||
srv.close()
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("--device", default="/dev/rally")
|
||||
ap.add_argument("--host", default="127.0.0.1")
|
||||
ap.add_argument("--port", type=int, default=5678)
|
||||
args = ap.parse_args(argv)
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [visca-xu] %(levelname)s %(message)s")
|
||||
rally = RallyV4L2(args.device)
|
||||
serve(args.host, args.port, rally.apply)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user