feat: call UX overhaul — deafen sync, share dialog, fullscreen redesign

Speaking ring:
- Switch useActiveSpeakers from LiveKit's smoothed isSpeaking / server-
  batched ActiveSpeakersChanged to Web Audio API AnalyserNode on each
  participant's raw audio MediaStreamTrack. Poll 50ms, RMS threshold 0.03,
  250ms hold. Feels real-time vs the old ~500ms lag
- Defensive syncProbe on every tick so probes catch up if TrackPublished
  missed (local mic publish race on join)
- Universal speaking overlay on tile (3px emerald border + inset glow,
  z-10) so video mode shows the ring too, not just audio mode

Screen sharing:
- Separate "screen" tile per sharer so the sharer's avatar tile stays
  intact with its speaking ring. Tile.id is kind-prefixed (user:xxx /
  screen:xxx) so focus tracking distinguishes them
- New ScreenShareDialog (quality preset + fps override + displaySurface
  hint) opens on the share button. startScreenShare / stopScreenShare
  actions in CallContext replace the one-shot toggle
- ScreenShareViewer: plain CSS-only fullscreen overlay (Tauri WKWebView
  doesn't implement requestFullscreen), always `h-full w-full
  object-contain`, Esc exits

Camera:
- toggleCamera action in CallContext tracks isCameraEnabled
- VideoStub renders real <video> srcObject for the participant's camera
  MediaStreamTrack; local preview is mirrored
- Tile video track resolves to Track.Source.Camera publications of the
  LocalParticipant / each RemoteParticipant
- Room listens for TrackMuted / TrackUnmuted and re-publishes remote
  state so peers switch to avatar placeholder when a camera is disabled

Deafen:
- New isDeafened state + toggleDeafen action. Sets `muted = true` on all
  attached `<audio[data-livekit-track]>` plus mutes fresh ones on attach
  via module-level flag
- Broadcast state over the LiveKit data channel
  ({type:'presence', deafened}) so peers can render the headphones-off
  badge. Attributes API not used because the self-hosted server may run
  older LiveKit versions
- remoteDeafen: Record<identity, bool> exposed via context, bumped on
  DataReceived and re-broadcast on ParticipantConnected

Incoming video call:
- acceptIncoming takes an optional CallKind override so the receiver can
  answer a video invite with audio only or promote an audio invite to
  video on accept
- IncomingCallPanel shows two accept buttons (audio + video) when the
  invite is a video call

Audio devices:
- audioSettings adds inputDeviceId + outputDeviceId, persisted
- CallContext uses them on setMicrophoneEnabled, plus new
  setAudioInputDevice / setAudioOutputDevice hot-swap actions.
  Output swap applies setSinkId to every attached remote-audio element
  since LiveKit's own switchActiveDevice only tracks elements it
  attached itself
- SettingsPage "Mikrofon" + "Ausgabegerät" selects with devicechange
  listener and a permission-probe button

Fullscreen mode:
- Replaced absolute-positioned speaker + floating thumbnails with a real
  flex layout. Default = even grid of all tiles. Clicking a tile flips
  to big-speaker + horizontal thumbnail strip. Click focused tile =
  back to grid
- Controls overlay pinned bottom; content wrapper has pb-24 so tiles
  never sit behind the toolbar
- Grid now uses explicit grid-rows-* so cells get a defined 1fr height
  (without it, video intrinsic dimensions blew tiles past the container
  bounds on Windows)

UI chips:
- Mic-off badge combines isMuted flag AND
  localParticipant.isMicrophoneEnabled, so a user with no mic / denied
  permission sees the badge + the toolbar button red even though they
  never pressed mute
- Deafen badge on tile chips for local + remote (remote driven by the
  data-channel broadcast)
This commit is contained in:
2026-04-21 00:02:43 +02:00
parent a4c9b959a9
commit da85f0ba54
10 changed files with 1129 additions and 241 deletions
+151 -14
View File
@@ -1,10 +1,83 @@
import type { Participant, Room } from 'livekit-client';
import { RoomEvent } from 'livekit-client';
import type { AudioTrack, Participant, Room } from 'livekit-client';
import { ParticipantEvent, RoomEvent, Track } from 'livekit-client';
import { useEffect, useState } from 'react';
// Subscribe to LiveKit ActiveSpeakersChanged to drive the speaking-ring pulse
// on participant tiles. Returns the set of currently speaking identities
// (includes local participant when they're active).
// Real-time speaking ring driven by Web Audio API AnalyserNodes directly on
// each participant's audio MediaStreamTrack. LiveKit's own `audioLevel` +
// `isSpeaking` are updated by a background monitor (default ~1s) — too
// laggy. AnalyserNodes give us raw PCM at browser frame rate, so the ring
// lights up within one animation frame of actual speech.
//
// Hold time of 250ms prevents flicker between words / short pauses.
const POLL_MS = 50;
const HOLD_MS = 250;
const THRESHOLD = 0.03; // RMS on 0..1 — tuned against soft speech
const FFT_SIZE = 256;
interface Probe {
ctx: AudioContext;
analyser: AnalyserNode;
source: MediaStreamAudioSourceNode;
buf: Uint8Array;
// Cached MediaStreamTrack reference — if a publication swaps tracks
// (mute/unmute, device switch) we rebuild the node chain.
trackId: string;
}
function makeProbe(track: MediaStreamTrack): Probe | null {
try {
const ctx = new (window.AudioContext || (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext)();
const stream = new MediaStream([track]);
const source = ctx.createMediaStreamSource(stream);
const analyser = ctx.createAnalyser();
analyser.fftSize = FFT_SIZE;
analyser.smoothingTimeConstant = 0.2;
source.connect(analyser);
return {
ctx,
analyser,
source,
buf: new Uint8Array(analyser.frequencyBinCount),
trackId: track.id,
};
} catch {
return null;
}
}
function destroyProbe(probe: Probe): void {
try {
probe.source.disconnect();
} catch {
/* ignore */
}
try {
void probe.ctx.close();
} catch {
/* ignore */
}
}
function sampleRms(probe: Probe): number {
// `getByteFrequencyData` types expect `Uint8Array<ArrayBuffer>` (no
// SharedArrayBuffer). Our `probe.buf` satisfies that at runtime; the cast
// sidesteps the TS lib signature quirk.
probe.analyser.getByteFrequencyData(probe.buf as Uint8Array<ArrayBuffer>);
let sum = 0;
for (let i = 0; i < probe.buf.length; i++) sum += probe.buf[i]!;
return sum / (probe.buf.length * 255);
}
function firstAudioTrack(p: Participant): AudioTrack | null {
const pubs = p.audioTrackPublications;
for (const pub of pubs.values()) {
if (pub.kind === Track.Kind.Audio && pub.track) {
return pub.track as AudioTrack;
}
}
return null;
}
export function useActiveSpeakers(room: Room | null): Set<string> {
const [ids, setIds] = useState<Set<string>>(() => new Set());
@@ -14,20 +87,84 @@ export function useActiveSpeakers(room: Room | null): Set<string> {
return;
}
const update = (speakers: Participant[]) => {
const next = new Set<string>();
for (const s of speakers) {
if (s.identity) next.add(s.identity);
const probes = new Map<string, Probe>();
const lastActive = new Map<string, number>();
const syncProbe = (p: Participant): void => {
if (!p.identity) return;
const track = firstAudioTrack(p);
const mst = track?.mediaStreamTrack ?? null;
const existing = probes.get(p.identity);
if (!mst) {
if (existing) {
destroyProbe(existing);
probes.delete(p.identity);
}
return;
}
setIds(next);
if (existing && existing.trackId === mst.id) return;
if (existing) destroyProbe(existing);
const probe = makeProbe(mst);
if (probe) probes.set(p.identity, probe);
};
// Seed from current state.
update(room.activeSpeakers ?? []);
const bind = (p: Participant): void => {
p.on(ParticipantEvent.TrackPublished, () => syncProbe(p));
p.on(ParticipantEvent.TrackUnpublished, () => syncProbe(p));
p.on(ParticipantEvent.TrackSubscribed, () => syncProbe(p));
p.on(ParticipantEvent.TrackUnsubscribed, () => syncProbe(p));
p.on(ParticipantEvent.TrackMuted, () => syncProbe(p));
p.on(ParticipantEvent.TrackUnmuted, () => syncProbe(p));
syncProbe(p);
};
bind(room.localParticipant);
room.remoteParticipants.forEach(bind);
const onConnected = (p: Participant) => bind(p);
const onDisconnected = (p: Participant) => {
if (!p.identity) return;
const probe = probes.get(p.identity);
if (probe) {
destroyProbe(probe);
probes.delete(p.identity);
}
lastActive.delete(p.identity);
};
room.on(RoomEvent.ParticipantConnected, onConnected);
room.on(RoomEvent.ParticipantDisconnected, onDisconnected);
const tick = () => {
// Defensively resync every tick — covers the gap where local mic
// finishes publishing between effect mount and the first
// TrackPublished event, and handles event misses on Tauri WebView.
syncProbe(room.localParticipant);
room.remoteParticipants.forEach(syncProbe);
const now = Date.now();
for (const [id, probe] of probes) {
if (sampleRms(probe) > THRESHOLD) lastActive.set(id, now);
}
const next = new Set<string>();
for (const [id, t] of lastActive) {
if (now - t <= HOLD_MS) next.add(id);
}
setIds((prev) => {
if (prev.size === next.size && Array.from(prev).every((v) => next.has(v))) {
return prev;
}
return next;
});
};
const interval = window.setInterval(tick, POLL_MS);
room.on(RoomEvent.ActiveSpeakersChanged, update);
return () => {
room.off(RoomEvent.ActiveSpeakersChanged, update);
window.clearInterval(interval);
room.off(RoomEvent.ParticipantConnected, onConnected);
room.off(RoomEvent.ParticipantDisconnected, onDisconnected);
for (const probe of probes.values()) destroyProbe(probe);
probes.clear();
};
}, [room]);