Publish and subscribe over wss://livekit.uni-wh.de:7800 and refuse cleartext ws://. Conference room is uwh-telhai. Includes the uncommitted encoded H.264 publish path, Rally hairpin, and KMS wall overlay.
321 lines
15 KiB
Python
321 lines
15 KiB
Python
"""Load configuration from environment / .env file."""
|
|
|
|
import os
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
|
|
def _load_dotenv(path: Path) -> None:
|
|
if not path.is_file():
|
|
return
|
|
for line in path.read_text().splitlines():
|
|
line = line.strip()
|
|
if not line or line.startswith("#") or "=" not in line:
|
|
continue
|
|
key, _, value = line.partition("=")
|
|
os.environ.setdefault(key.strip(), value.strip())
|
|
|
|
|
|
@dataclass
|
|
class Config:
|
|
# connection
|
|
url: str = ""
|
|
api_key: str = ""
|
|
api_secret: str = ""
|
|
room: str = "cameras"
|
|
participant_prefix: str = "cam"
|
|
# video
|
|
width: int = 320
|
|
height: int = 180
|
|
fps: int = 15
|
|
min_fps: int = 5
|
|
video_bitrate: int = 400_000
|
|
rally_video_bitrate: int = 20_000_000
|
|
video_codec: str = "h264"
|
|
video_encoder: str = "vaapi" # auto|software|hardware|nvenc|vaapi
|
|
vaapi_device: str = "/dev/dri/renderD129"
|
|
# audio
|
|
audio_rate: int = 16000
|
|
audio_channels: int = 1
|
|
# speechbrain cleanup
|
|
enhance_mode: str = "auto" # auto | force | off
|
|
enhance_model: str = "speechbrain/metricgan-plus-voicebank"
|
|
enhance_chunk_s: float = 1.0
|
|
enhance_hop_s: float = 0.1
|
|
enhance_infer: str = "auto" # auto | openvino | torch
|
|
# SpeechBrain Delay-and-Sum (GCC-PHAT) on C920 stereo before enhance
|
|
beamform_mode: str = "auto" # auto | force | off
|
|
torch_num_threads: int = 1
|
|
beamform_tdoa_every: int = 5
|
|
# latency knobs (unset / False = legacy behavior)
|
|
audio_queue_ms: Optional[int] = None # AUDIO_QUEUE_MS; None = max(150, hop_ms+50)
|
|
video_hold_s: Optional[float] = None # VIDEO_HOLD_S; None = match hop
|
|
worker_split_executor: bool = False
|
|
enhance_speaker_only: bool = False
|
|
audio_gate: bool = False
|
|
audio_gate_open_db: float = -28.0
|
|
audio_gate_close_db: float = -34.0
|
|
audio_gate_hold_s: float = 0.3
|
|
enhance_daemon: bool = True
|
|
enhance_socket: str = "/tmp/livekit-enhance.sock"
|
|
capture_daemon: bool = True
|
|
capture_shm_dir: str = "/run/livekit-cameras/raw"
|
|
audio_daemon: bool = True
|
|
audio_shm_dir: str = "/run/livekit-cameras/pcm"
|
|
encode_daemon: bool = True
|
|
encode_shm_dir: str = "/run/livekit-cameras/h264"
|
|
single_publisher: bool = True
|
|
display_video_capacity: int = 0 # 0 = unbounded VideoStream
|
|
display_video_format: str = "" # "" = SDK default, "bgra" | "i420"
|
|
display_gst_queue_buffers: int = 2
|
|
display_speaker_push: str = "timer" # timer | arrival
|
|
display_pump_workers: int = 4
|
|
# misc
|
|
publish_timeout_s: float = 60.0
|
|
log_level: str = "INFO"
|
|
cameras: list[str] = field(default_factory=lambda: ["all"])
|
|
# headless displays (LiveKit subscribe -> GStreamer kmssink)
|
|
display_count: int = 3
|
|
display_identities: list[str] = field(
|
|
default_factory=lambda: ["cam-01", "cam-02", "cam-03"])
|
|
display_connectors: list[str] = field(default_factory=list)
|
|
display_identity: str = "display-wall"
|
|
display_roles: list[str] = field(
|
|
default_factory=lambda: ["grid", "speaker", "screenshare"])
|
|
display_speaker_camera: str = "rally" # LiveKit identity; not a C920
|
|
display_speaker_participant: str = "speaker"
|
|
display_grid_cols: int = 5
|
|
display_grid_rows: int = 4
|
|
display_grid_width: int = 1920
|
|
display_grid_height: int = 1080
|
|
display_grid_fps: int = 30
|
|
livekit_public_url: str = ""
|
|
# Site tag published on local cameras (LiveKit participant attributes).
|
|
participant_tags: list[str] = field(default_factory=lambda: ["uwh"])
|
|
# Display wall: skip participants carrying these tags. Empty = off
|
|
# (show local cameras). Set to ["uwh"] to show remotes instead.
|
|
display_hide_tags: list[str] = field(default_factory=list)
|
|
# Yellow tile chrome when MoveNet sees a raised hand (display process only).
|
|
display_hand_raise: bool = True
|
|
display_hand_raise_hz: float = 4.0
|
|
display_hand_raise_hold_s: float = 1.0
|
|
display_hand_raise_model: str = "thunder"
|
|
display_speak_min_db: float = -40.0
|
|
display_speak_margin_db: float = 6.0
|
|
display_speak_hold_s: float = 0.6
|
|
display_speak_rise_db: float = 3.0
|
|
display_speak_corr: float = 0.4
|
|
display_speak_confirm: int = 2
|
|
# Person-aware background blur (C920s only; OpenVINO CPU). Rollback: PORTRAIT_BLUR=0.
|
|
portrait_blur: bool = True
|
|
portrait_shm_dir: str = "/run/livekit-cameras/portrait"
|
|
portrait_hz: float = 8.0
|
|
portrait_blur_px: int = 7
|
|
portrait_hold_s: float = 0.8
|
|
|
|
|
|
def _parse_cameras(spec: str) -> list[str]:
|
|
"""Expand 'all' / '1,3,5' / '0-19' into a list of indices ('' = all)."""
|
|
spec = (spec or "all").strip().lower()
|
|
if spec in ("all", ""):
|
|
return ["all"]
|
|
out = []
|
|
for part in spec.split(","):
|
|
part = part.strip()
|
|
if not part:
|
|
continue
|
|
if "-" in part:
|
|
lo, hi = part.split("-", 1)
|
|
out.extend(str(i) for i in range(int(lo), int(hi) + 1))
|
|
else:
|
|
out.append(part)
|
|
return out
|
|
|
|
|
|
def _parse_list(spec: str) -> list[str]:
|
|
"""Comma-separated names, preserving case (HDMI-A-7, cam-01)."""
|
|
spec = (spec or "").strip()
|
|
if not spec:
|
|
return []
|
|
return [p.strip() for p in spec.split(",") if p.strip()]
|
|
|
|
|
|
def _parse_bool(spec: str, default: bool = False) -> bool:
|
|
raw = (spec or "").strip().lower()
|
|
if not raw:
|
|
return default
|
|
if raw in ("1", "true", "yes", "on"):
|
|
return True
|
|
if raw in ("0", "false", "no", "off"):
|
|
return False
|
|
raise ValueError(f"invalid bool {spec!r}")
|
|
|
|
|
|
def _opt_int(spec: str) -> Optional[int]:
|
|
raw = (spec or "").strip()
|
|
if raw == "":
|
|
return None
|
|
return int(raw)
|
|
|
|
|
|
def _opt_float(spec: str) -> Optional[float]:
|
|
raw = (spec or "").strip()
|
|
if raw == "":
|
|
return None
|
|
return float(raw)
|
|
|
|
|
|
def load_config(env_path: Optional[Path] = None) -> Config:
|
|
_load_dotenv(Path(env_path or Path(__file__).parent / ".env"))
|
|
|
|
cfg = Config(
|
|
url=os.environ.get("LIVEKIT_URL", ""),
|
|
api_key=os.environ.get("LIVEKIT_API_KEY", ""),
|
|
api_secret=os.environ.get("LIVEKIT_API_SECRET", ""),
|
|
room=os.environ.get("LIVEKIT_ROOM", "uwh-telhai"),
|
|
participant_prefix=os.environ.get("PARTICIPANT_PREFIX", "cam"),
|
|
width=int(os.environ.get("VIDEO_WIDTH", "320")),
|
|
height=int(os.environ.get("VIDEO_HEIGHT", "180")),
|
|
fps=int(os.environ.get("VIDEO_FPS", "15")),
|
|
min_fps=int(os.environ.get("VIDEO_MIN_FPS", "5")),
|
|
video_bitrate=int(os.environ.get("VIDEO_BITRATE", "400000")),
|
|
rally_video_bitrate=int(os.environ.get("VIDEO_RALLY_BITRATE", "20000000")),
|
|
video_codec=os.environ.get("VIDEO_CODEC", "h264").lower(),
|
|
video_encoder=os.environ.get("VIDEO_ENCODER", "vaapi").lower(),
|
|
vaapi_device=os.environ.get("LIBVA_DRM_DEVICE", "/dev/dri/renderD129"),
|
|
audio_rate=int(os.environ.get("AUDIO_SAMPLE_RATE", "16000")),
|
|
audio_channels=int(os.environ.get("AUDIO_CHANNELS", "1")),
|
|
enhance_mode=os.environ.get("ENHANCE_MODE", "auto").lower(),
|
|
enhance_model=os.environ.get(
|
|
"ENHANCE_MODEL", "speechbrain/metricgan-plus-voicebank"
|
|
),
|
|
enhance_chunk_s=float(os.environ.get("ENHANCE_CHUNK_S", "1.0")),
|
|
enhance_hop_s=float(os.environ.get("ENHANCE_HOP_S", "0.1")),
|
|
enhance_infer=os.environ.get("ENHANCE_INFER", "auto").lower(),
|
|
beamform_mode=os.environ.get("BEAMFORM_MODE", "auto").lower(),
|
|
torch_num_threads=int(os.environ.get("TORCH_NUM_THREADS", "1")),
|
|
beamform_tdoa_every=int(os.environ.get("BEAMFORM_TDOA_EVERY", "5")),
|
|
audio_queue_ms=_opt_int(os.environ.get("AUDIO_QUEUE_MS", "")),
|
|
video_hold_s=_opt_float(os.environ.get("VIDEO_HOLD_S", "")),
|
|
worker_split_executor=_parse_bool(os.environ.get("WORKER_SPLIT_EXECUTOR", "")),
|
|
enhance_speaker_only=_parse_bool(os.environ.get("ENHANCE_SPEAKER_ONLY", "")),
|
|
audio_gate=_parse_bool(os.environ.get("AUDIO_GATE", ""), default=False),
|
|
audio_gate_open_db=float(os.environ.get("AUDIO_GATE_OPEN_DB", "-28")),
|
|
audio_gate_close_db=float(os.environ.get("AUDIO_GATE_CLOSE_DB", "-34")),
|
|
audio_gate_hold_s=float(os.environ.get("AUDIO_GATE_HOLD_S", "0.3")),
|
|
enhance_daemon=_parse_bool(os.environ.get("ENHANCE_DAEMON", ""), default=True),
|
|
enhance_socket=os.environ.get("ENHANCE_SOCKET", "/tmp/livekit-enhance.sock").strip()
|
|
or "/tmp/livekit-enhance.sock",
|
|
capture_daemon=_parse_bool(os.environ.get("CAPTURE_DAEMON", ""), default=True),
|
|
capture_shm_dir=os.environ.get("CAPTURE_SHM_DIR", "/run/livekit-cameras/raw").strip()
|
|
or "/run/livekit-cameras/raw",
|
|
audio_daemon=_parse_bool(os.environ.get("AUDIO_DAEMON", ""), default=True),
|
|
audio_shm_dir=os.environ.get("AUDIO_SHM_DIR", "/run/livekit-cameras/pcm").strip()
|
|
or "/run/livekit-cameras/pcm",
|
|
encode_daemon=_parse_bool(os.environ.get("ENCODE_DAEMON", ""), default=True),
|
|
encode_shm_dir=os.environ.get("ENCODE_SHM_DIR", "/run/livekit-cameras/h264").strip()
|
|
or "/run/livekit-cameras/h264",
|
|
single_publisher=_parse_bool(os.environ.get("SINGLE_PUBLISHER", ""), default=True),
|
|
display_video_capacity=int(os.environ.get("DISPLAY_VIDEO_CAPACITY", "0")),
|
|
display_video_format=os.environ.get("DISPLAY_VIDEO_FORMAT", "").strip().lower(),
|
|
display_gst_queue_buffers=int(os.environ.get("DISPLAY_GST_QUEUE_BUFFERS", "2")),
|
|
display_speaker_push=os.environ.get("DISPLAY_SPEAKER_PUSH", "timer").strip().lower(),
|
|
display_pump_workers=int(os.environ.get("DISPLAY_PUMP_WORKERS", "4")),
|
|
publish_timeout_s=float(os.environ.get("PUBLISH_TIMEOUT_S", "60")),
|
|
log_level=os.environ.get("LOG_LEVEL", "INFO").upper(),
|
|
cameras=_parse_cameras(os.environ.get("CAMERAS", "all")),
|
|
display_count=int(os.environ.get("DISPLAY_COUNT", "3")),
|
|
display_identities=_parse_list(
|
|
os.environ.get("DISPLAY_IDENTITIES", "cam-01,cam-02,cam-03")),
|
|
display_connectors=_parse_list(os.environ.get("DISPLAY_CONNECTORS", "")),
|
|
display_identity=os.environ.get("DISPLAY_IDENTITY", "display-wall"),
|
|
display_roles=_parse_list(
|
|
os.environ.get("DISPLAY_ROLES", "grid,speaker,screenshare"))
|
|
or ["grid", "speaker", "screenshare"],
|
|
display_speaker_camera=os.environ.get("DISPLAY_SPEAKER_CAMERA", "rally").strip()
|
|
or "rally",
|
|
display_speaker_participant=os.environ.get(
|
|
"DISPLAY_SPEAKER_PARTICIPANT", "speaker"),
|
|
display_grid_cols=int(os.environ.get("DISPLAY_GRID_COLS", "5")),
|
|
display_grid_rows=int(os.environ.get("DISPLAY_GRID_ROWS", "4")),
|
|
display_grid_width=int(os.environ.get("DISPLAY_GRID_WIDTH", "1920")),
|
|
display_grid_height=int(os.environ.get("DISPLAY_GRID_HEIGHT", "1080")),
|
|
display_grid_fps=int(os.environ.get("DISPLAY_GRID_FPS", "30")),
|
|
livekit_public_url=os.environ.get("LIVEKIT_PUBLIC_URL", ""),
|
|
participant_tags=_parse_list(os.environ.get("PARTICIPANT_TAGS", "uwh")),
|
|
display_hide_tags=_parse_list(os.environ.get("DISPLAY_HIDE_TAGS", "")),
|
|
display_hand_raise=_parse_bool(os.environ.get("DISPLAY_HAND_RAISE", ""), default=True),
|
|
display_hand_raise_hz=float(os.environ.get("DISPLAY_HAND_RAISE_HZ", "4")),
|
|
display_hand_raise_hold_s=float(os.environ.get("DISPLAY_HAND_RAISE_HOLD_S", "1.0")),
|
|
display_hand_raise_model=os.environ.get("DISPLAY_HAND_RAISE_MODEL", "thunder").strip().lower(),
|
|
display_speak_min_db=float(os.environ.get("DISPLAY_SPEAK_MIN_DB", "-40")),
|
|
display_speak_margin_db=float(os.environ.get("DISPLAY_SPEAK_MARGIN_DB", "6")),
|
|
display_speak_hold_s=float(os.environ.get("DISPLAY_SPEAK_HOLD_S", "0.6")),
|
|
display_speak_rise_db=float(os.environ.get("DISPLAY_SPEAK_RISE_DB", "3")),
|
|
display_speak_corr=float(os.environ.get("DISPLAY_SPEAK_CORR", "0.4")),
|
|
display_speak_confirm=int(os.environ.get("DISPLAY_SPEAK_CONFIRM", "2")),
|
|
portrait_blur=_parse_bool(os.environ.get("PORTRAIT_BLUR", ""), default=True),
|
|
portrait_shm_dir=os.environ.get(
|
|
"PORTRAIT_SHM_DIR", "/run/livekit-cameras/portrait"
|
|
).strip() or "/run/livekit-cameras/portrait",
|
|
portrait_hz=float(os.environ.get("PORTRAIT_HZ", "8")),
|
|
portrait_blur_px=int(os.environ.get("PORTRAIT_BLUR_PX", "7")),
|
|
portrait_hold_s=float(os.environ.get("PORTRAIT_HOLD_S", "0.8")),
|
|
)
|
|
|
|
if cfg.enhance_mode not in ("auto", "force", "off"):
|
|
raise ValueError(f"ENHANCE_MODE must be auto|force|off, got {cfg.enhance_mode}")
|
|
if cfg.enhance_infer not in ("auto", "openvino", "torch"):
|
|
raise ValueError(
|
|
f"ENHANCE_INFER must be auto|openvino|torch, got {cfg.enhance_infer}")
|
|
if cfg.beamform_mode not in ("auto", "force", "off"):
|
|
raise ValueError(
|
|
f"BEAMFORM_MODE must be auto|force|off, got {cfg.beamform_mode}")
|
|
if cfg.torch_num_threads < 1:
|
|
raise ValueError("TORCH_NUM_THREADS must be >= 1")
|
|
if cfg.beamform_tdoa_every < 1:
|
|
raise ValueError("BEAMFORM_TDOA_EVERY must be >= 1")
|
|
if cfg.audio_queue_ms is not None and cfg.audio_queue_ms < 1:
|
|
raise ValueError("AUDIO_QUEUE_MS must be >= 1")
|
|
if cfg.audio_gate_hold_s < 0:
|
|
raise ValueError("AUDIO_GATE_HOLD_S must be >= 0")
|
|
if cfg.video_hold_s is not None and cfg.video_hold_s < 0:
|
|
raise ValueError("VIDEO_HOLD_S must be >= 0")
|
|
if cfg.display_video_capacity < 0:
|
|
raise ValueError("DISPLAY_VIDEO_CAPACITY must be >= 0")
|
|
if cfg.display_video_format not in ("", "bgra", "i420", "yuv420p"):
|
|
raise ValueError(
|
|
f"DISPLAY_VIDEO_FORMAT must be empty, bgra or i420, got {cfg.display_video_format}")
|
|
if cfg.display_gst_queue_buffers < 1:
|
|
raise ValueError("DISPLAY_GST_QUEUE_BUFFERS must be >= 1")
|
|
if cfg.display_speaker_push not in ("timer", "arrival"):
|
|
raise ValueError(
|
|
f"DISPLAY_SPEAKER_PUSH must be timer|arrival, got {cfg.display_speaker_push}")
|
|
if cfg.display_pump_workers < 1:
|
|
raise ValueError("DISPLAY_PUMP_WORKERS must be >= 1")
|
|
if cfg.display_hand_raise_hz <= 0:
|
|
raise ValueError("DISPLAY_HAND_RAISE_HZ must be > 0")
|
|
if cfg.display_hand_raise_hold_s < 0:
|
|
raise ValueError("DISPLAY_HAND_RAISE_HOLD_S must be >= 0")
|
|
if cfg.display_hand_raise_model not in ("lightning", "thunder"):
|
|
raise ValueError(
|
|
f"DISPLAY_HAND_RAISE_MODEL must be lightning|thunder, got {cfg.display_hand_raise_model}")
|
|
if cfg.display_speak_margin_db < 0:
|
|
raise ValueError("DISPLAY_SPEAK_MARGIN_DB must be >= 0")
|
|
if cfg.display_speak_hold_s < 0:
|
|
raise ValueError("DISPLAY_SPEAK_HOLD_S must be >= 0")
|
|
if cfg.display_speak_rise_db < 0:
|
|
raise ValueError("DISPLAY_SPEAK_RISE_DB must be >= 0")
|
|
if not 0.0 <= cfg.display_speak_corr <= 1.0:
|
|
raise ValueError("DISPLAY_SPEAK_CORR must be in 0..1")
|
|
if cfg.display_speak_confirm < 1:
|
|
raise ValueError("DISPLAY_SPEAK_CONFIRM must be >= 1")
|
|
if cfg.portrait_hz <= 0:
|
|
raise ValueError("PORTRAIT_HZ must be > 0")
|
|
if cfg.portrait_blur_px < 1:
|
|
raise ValueError("PORTRAIT_BLUR_PX must be >= 1")
|
|
if cfg.portrait_hold_s < 0:
|
|
raise ValueError("PORTRAIT_HOLD_S must be >= 0")
|
|
return cfg
|