diff --git a/apps/desktop/src/components/CallCaptionsOverlay.tsx b/apps/desktop/src/components/CallCaptionsOverlay.tsx
deleted file mode 100644
index 944a8df..0000000
--- a/apps/desktop/src/components/CallCaptionsOverlay.tsx
+++ /dev/null
@@ -1,55 +0,0 @@
-import { useEffect, useState } from 'react';
-
-import type { ConversationSummary } from '@chat-app/shared/chat';
-import { useCall } from '../context/CallContext';
-
-interface Props {
- conversation: ConversationSummary;
-}
-
-const STALE_AFTER_MS = 5000;
-
-/** Discord-style live-captions overlay. Pinned to the bottom-center of the
- * call surface; renders the most recent caption per participant, fading
- * entries out after `STALE_AFTER_MS` of silence. Self-captions are shown
- * too so the speaker can sanity-check what's being broadcast. */
-export function CallCaptionsOverlay({ conversation }: Props) {
- const { captions } = useCall();
- // Re-render every second so stale entries fade without needing the data
- // channel to fire — captions module just stores timestamps.
- const [, setNow] = useState(Date.now());
- useEffect(() => {
- const id = window.setInterval(() => setNow(Date.now()), 1000);
- return () => window.clearInterval(id);
- }, []);
-
- const now = Date.now();
- const visible = Object.entries(captions)
- .filter(([, v]) => now - v.timestamp < STALE_AFTER_MS)
- .sort(([, a], [, b]) => a.timestamp - b.timestamp);
-
- if (visible.length === 0) return null;
-
- return (
-
- {visible.map(([identity, c]) => {
- const member = conversation.members.find((m) => m.userId === identity);
- const name = member?.profile?.displayName ?? '?';
- const age = now - c.timestamp;
- const opacity = age > 3500 ? 1 - (age - 3500) / (STALE_AFTER_MS - 3500) : 1;
- return (
-
-
- {name}
-
- {c.text}
-
- );
- })}
-
- );
-}
diff --git a/apps/desktop/src/components/CallControls.tsx b/apps/desktop/src/components/CallControls.tsx
index 0656776..0250fa9 100644
--- a/apps/desktop/src/components/CallControls.tsx
+++ b/apps/desktop/src/components/CallControls.tsx
@@ -1,7 +1,6 @@
import { useTranslation } from 'react-i18next';
import {
- CaptionsIcon,
HeadphonesIcon,
HeadphonesOffIcon,
MicIcon,
@@ -32,10 +31,6 @@ interface Props {
/** Toggle the in-call soundboard popover. Active = panel currently open. */
onToggleSoundboard?: () => void;
soundboardOpen?: boolean;
- /** Discord-style live-captions toggle. Optional — pages that don't support
- * SpeechRecognition (Firefox) skip the prop and the button is hidden. */
- onToggleCaptions?: () => void;
- captionsOn?: boolean;
participantsOpen?: boolean;
/** Compact variant used inside the docked call (36px buttons). */
compact?: boolean;
@@ -59,8 +54,6 @@ export function CallControls({
onOpenParticipants,
onToggleSoundboard,
soundboardOpen = false,
- onToggleCaptions,
- captionsOn = false,
participantsOpen = false,
compact = false,
glass = false,
@@ -148,22 +141,6 @@ export function CallControls({
)}
- {onToggleCaptions && (
-
-
-
- )}
{onOpenParticipants && (
(
- () => getLiveCaptionsSettings().enabled,
- );
- useEffect(
- () => subscribeLiveCaptionsSettings((s) => setCaptionsEnabled(s.enabled)),
- [],
- );
useEffect(() => {
const onKey = (e: KeyboardEvent) => {
// Match the modifier exactly to avoid clobbering other Ctrl+Shift combos.
@@ -362,15 +344,6 @@ export function InCallPanel({ conversation }: Props) {
soundboardOpen,
}
: {})}
- // Live-Captions only when SpeechRecognition is available in the
- // runtime — Firefox lacks it, would just show a dead button.
- {...(isLiveCaptionsSupported()
- ? {
- onToggleCaptions: () =>
- updateLiveCaptionsSettings({ enabled: !captionsEnabled }),
- captionsOn: captionsEnabled,
- }
- : {})}
onHangup={() => void hangup()}
compact={callMode !== 'fullscreen'}
glass={callMode === 'fullscreen'}
@@ -499,7 +472,6 @@ export function InCallPanel({ conversation }: Props) {
onClose={() => setStatsOverlayOpen(false)}
/>
)}
- {captionsEnabled && }
>
);
}
@@ -644,8 +616,6 @@ export function InCallPanel({ conversation }: Props) {
onClose={() => setStatsOverlayOpen(false)}
/>
)}
-
- {captionsEnabled && }
);
}
diff --git a/apps/desktop/src/components/icons.tsx b/apps/desktop/src/components/icons.tsx
index ff456f0..32857c8 100644
--- a/apps/desktop/src/components/icons.tsx
+++ b/apps/desktop/src/components/icons.tsx
@@ -160,17 +160,6 @@ function EyeIconInner(props: IconProps) {
}
export const EyeIcon = memo(EyeIconInner);
-function CaptionsIconInner(props: IconProps) {
- return (
-
-
-
-
-
- );
-}
-export const CaptionsIcon = memo(CaptionsIconInner);
-
function PinOffIconInner(props: IconProps) {
return (
diff --git a/apps/desktop/src/context/CallContext.tsx b/apps/desktop/src/context/CallContext.tsx
index 7773771..ad97c05 100644
--- a/apps/desktop/src/context/CallContext.tsx
+++ b/apps/desktop/src/context/CallContext.tsx
@@ -39,7 +39,6 @@ import {
playUndeafenBeep,
playUnmuteBeep,
} from '../lib/callSounds';
-import { useLiveCaptions } from '../lib/useLiveCaptions';
import { setCallWakeLock } from '../lib/wakeLock';
import { notify } from '../lib/osNotify';
import {
@@ -209,14 +208,6 @@ interface CallContextValue {
* invite (fromUserId). Cleared on disconnect. Drives the crown badge,
* but only in group calls. Null while idle or in 1:1 contexts. */
callHostId: string | null;
- /** identity -> latest live-caption fragment received via data channel.
- * Includes own captions for self-overlay. Receivers prune entries whose
- * timestamp is older than ~5s so stale lines fade out. */
- captions: Record;
- /** Surface a caption for the local user — the live-captions hook calls
- * this on every interim/final SpeechRecognition result so the overlay
- * shows our own line without going through the SFU round-trip. */
- pushLocalCaption: (text: string, final: boolean) => void;
/** identity -> mute state. Broadcast from peer whenever mic-gain flips.
* Needed because we can't rely on LiveKit's native isMicrophoneEnabled —
* the mic pipeline keeps the track published with sound flowing even
@@ -347,9 +338,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
// useEffect) so peers don't hear themselves echoed back when the OS-level
// process-tree exclusion isn't watertight.
const [isCapturingSystemAudio, setIsCapturingSystemAudio] = useState(false);
- const [captions, setCaptions] = useState<
- Record
- >({});
const [lastCallConversationId, setLastCallConversationId] = useState(null);
const [callMode, setCallModeState] = useState('grid');
const [focusedId, setFocusedIdState] = useState(null);
@@ -828,7 +816,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
setRemoteScreenShares([]);
setConnectionQualities({});
setCallHostId(null);
- setCaptions({});
setIsScreenSharing(false);
setIsE2EEActive(false);
}
@@ -970,8 +957,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
type?: string;
deafened?: boolean;
muted?: boolean;
- captionText?: string;
- captionFinal?: boolean;
};
const id: string = participant.identity;
if (msg.type === 'presence') {
@@ -991,15 +976,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
}
return;
}
- if (msg.type === 'caption' && typeof msg.captionText === 'string') {
- const text2 = msg.captionText;
- const final = msg.captionFinal === true;
- setCaptions((prev) => ({
- ...prev,
- [id]: { text: text2, final, timestamp: Date.now() },
- }));
- return;
- }
} catch {
/* ignore malformed */
}
@@ -1754,17 +1730,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
});
}, []);
- const pushLocalCaption = useCallback(
- (text: string, final: boolean) => {
- if (!myId) return;
- setCaptions((prev) => ({
- ...prev,
- [myId]: { text, final, timestamp: Date.now() },
- }));
- },
- [myId],
- );
-
const toggleCamera = useCallback(async () => {
const r = roomRef.current;
if (!r) return;
@@ -2603,15 +2568,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
return () => window.removeEventListener('keydown', onKey);
}, [callMode, state.kind]);
- // Discord-style live-captions broadcaster — runs on the local mic while
- // we're connected, and ships interim/final transcripts on the LiveKit
- // DataChannel so peers can render them.
- useLiveCaptions({
- room,
- active: state.kind === 'connected' || state.kind === 'reconnecting',
- onLocalCaption: pushLocalCaption,
- });
-
const value = useMemo(
() => ({
state,
@@ -2626,8 +2582,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
remoteMute,
connectionQualities,
callHostId,
- captions,
- pushLocalCaption,
remoteScreenShares,
lastCallConversationId,
callMode,
@@ -2683,8 +2637,6 @@ export function CallProvider({ children }: { children: ReactNode }) {
remoteMute,
connectionQualities,
callHostId,
- captions,
- pushLocalCaption,
remoteScreenShares,
lastCallConversationId,
callMode,
diff --git a/apps/desktop/src/lib/liveCaptions.ts b/apps/desktop/src/lib/liveCaptions.ts
deleted file mode 100644
index 5063bfb..0000000
--- a/apps/desktop/src/lib/liveCaptions.ts
+++ /dev/null
@@ -1,119 +0,0 @@
-// Discord-style live captions. Uses the browser's SpeechRecognition API to
-// transcribe the LOCAL user's mic, then broadcasts each interim/final result
-// to peers via the LiveKit DataChannel. Receivers store and display them.
-//
-// Privacy note: speech recognition runs in the browser. On Chromium-based
-// runtimes (incl. Tauri's WebView2 on Windows) this calls into the browser's
-// own engine, which today reaches Google's cloud — same trade-off as Discord.
-// We ship a hard off switch and require an explicit user toggle.
-
-const STORAGE_KEY = 'chatapp.liveCaptions.v1';
-
-export interface LiveCaptionsSettings {
- enabled: boolean;
- /** BCP-47 language tag, e.g. "de-DE" or "en-US". Auto-detect uses navigator.language. */
- lang: string | null;
-}
-
-const DEFAULTS: LiveCaptionsSettings = {
- enabled: false,
- lang: null,
-};
-
-type Listener = (s: LiveCaptionsSettings) => void;
-const listeners = new Set();
-let cached: LiveCaptionsSettings | null = null;
-
-function read(): LiveCaptionsSettings {
- if (cached) return cached;
- try {
- const raw = window.localStorage.getItem(STORAGE_KEY);
- if (!raw) {
- cached = DEFAULTS;
- return cached;
- }
- const parsed = JSON.parse(raw) as Partial;
- cached = {
- enabled: typeof parsed.enabled === 'boolean' ? parsed.enabled : DEFAULTS.enabled,
- lang:
- typeof parsed.lang === 'string' && parsed.lang.length > 0
- ? parsed.lang
- : DEFAULTS.lang,
- };
- return cached;
- } catch {
- cached = DEFAULTS;
- return cached;
- }
-}
-
-function write(s: LiveCaptionsSettings): void {
- cached = s;
- try {
- window.localStorage.setItem(STORAGE_KEY, JSON.stringify(s));
- } catch {
- /* quota / private mode */
- }
- for (const l of listeners) l(s);
-}
-
-export function getLiveCaptionsSettings(): LiveCaptionsSettings {
- return read();
-}
-
-export function updateLiveCaptionsSettings(
- patch: Partial,
-): LiveCaptionsSettings {
- const next = { ...read(), ...patch };
- write(next);
- return next;
-}
-
-export function subscribeLiveCaptionsSettings(listener: Listener): () => void {
- listeners.add(listener);
- return () => listeners.delete(listener);
-}
-
-// Browser feature-detect. Chromium ships under both names; Firefox lacks it
-// outright. Returns the constructor or null.
-type SpeechRecognitionCtor = new () => SpeechRecognitionLike;
-interface SpeechRecognitionLike extends EventTarget {
- continuous: boolean;
- interimResults: boolean;
- lang: string;
- start: () => void;
- stop: () => void;
- abort: () => void;
- onresult: ((e: SpeechRecognitionEventLike) => void) | null;
- onerror: ((e: SpeechRecognitionErrorLike) => void) | null;
- onend: (() => void) | null;
-}
-interface SpeechRecognitionEventLike {
- resultIndex: number;
- results: ArrayLike<{
- isFinal: boolean;
- [index: number]: { transcript: string };
- length: number;
- }>;
-}
-interface SpeechRecognitionErrorLike {
- error: string;
-}
-
-export function getSpeechRecognitionCtor(): SpeechRecognitionCtor | null {
- const w = window as unknown as {
- SpeechRecognition?: SpeechRecognitionCtor;
- webkitSpeechRecognition?: SpeechRecognitionCtor;
- };
- return w.SpeechRecognition ?? w.webkitSpeechRecognition ?? null;
-}
-
-export function isLiveCaptionsSupported(): boolean {
- return getSpeechRecognitionCtor() !== null;
-}
-
-export type {
- SpeechRecognitionLike,
- SpeechRecognitionEventLike,
- SpeechRecognitionErrorLike,
-};
diff --git a/apps/desktop/src/lib/useLiveCaptions.ts b/apps/desktop/src/lib/useLiveCaptions.ts
deleted file mode 100644
index 09ae07a..0000000
--- a/apps/desktop/src/lib/useLiveCaptions.ts
+++ /dev/null
@@ -1,142 +0,0 @@
-// Hook that runs SpeechRecognition on the local mic when live-captions are
-// enabled and a Room is connected. Each interim/final result is broadcast as
-// a `caption`-typed message via the LiveKit DataChannel so peers can render
-// it. Recognition stops cleanly when the call ends or the toggle flips off.
-
-import type { Room } from 'livekit-client';
-import { useEffect, useRef } from 'react';
-
-import {
- type LiveCaptionsSettings,
- getLiveCaptionsSettings,
- getSpeechRecognitionCtor,
- type SpeechRecognitionEventLike,
- type SpeechRecognitionLike,
- subscribeLiveCaptionsSettings,
-} from './liveCaptions';
-
-interface Args {
- room: Room | null;
- /** True while we're connected and want captions to flow. */
- active: boolean;
- /** Callback fired locally for our own captions so the overlay can show
- * them without going through the SFU round-trip. */
- onLocalCaption: (text: string, final: boolean) => void;
-}
-
-export function useLiveCaptions({ room, active, onLocalCaption }: Args): void {
- const recognitionRef = useRef(null);
- const settingsRef = useRef(getLiveCaptionsSettings());
-
- useEffect(() => {
- return subscribeLiveCaptionsSettings((s) => {
- settingsRef.current = s;
- });
- }, []);
-
- useEffect(() => {
- const Ctor = getSpeechRecognitionCtor();
- if (!Ctor) return; // unsupported runtime
- if (!active || !room) return;
- if (!getLiveCaptionsSettings().enabled) return;
-
- const send = (text: string, final: boolean) => {
- onLocalCaption(text, final);
- try {
- const payload = new TextEncoder().encode(
- JSON.stringify({ type: 'caption', captionText: text, captionFinal: final }),
- );
- // Reliable channel — captions are infrequent enough to afford it,
- // and dropping interims looks worse than slight lag.
- void room.localParticipant.publishData(payload, { reliable: true });
- } catch {
- /* ignore — best-effort */
- }
- };
-
- const start = () => {
- const r = new Ctor();
- r.continuous = true;
- r.interimResults = true;
- const lang = settingsRef.current.lang ?? navigator.language ?? 'de-DE';
- r.lang = lang;
- r.onresult = (e: SpeechRecognitionEventLike) => {
- // Pull whichever results arrived since last fire. Interim fires
- // many times per second; the final one is sticky and persists.
- for (let i = e.resultIndex; i < e.results.length; i++) {
- const result = e.results[i];
- if (!result || result.length === 0) continue;
- const alt = result[0];
- if (!alt) continue;
- const transcript = alt.transcript.trim();
- if (!transcript) continue;
- send(transcript, result.isFinal);
- }
- };
- r.onerror = () => {
- // Recoverable: stop + retry on next effect cycle. `not-allowed` and
- // `service-not-allowed` are permission-permanent — bail.
- try {
- r.stop();
- } catch {
- /* ignore */
- }
- };
- r.onend = () => {
- // SpeechRecognition tends to auto-stop after silence — if we still
- // want captions, restart it. Guard against tear-down race.
- if (recognitionRef.current === r && getLiveCaptionsSettings().enabled) {
- try {
- r.start();
- } catch {
- /* already running or browser refused */
- }
- }
- };
- try {
- r.start();
- recognitionRef.current = r;
- } catch {
- // Some browsers throw when start() is called too soon after a
- // previous abort — wait a tick and retry.
- window.setTimeout(() => {
- try {
- r.start();
- recognitionRef.current = r;
- } catch {
- /* give up */
- }
- }, 250);
- }
- };
-
- start();
-
- const unsub = subscribeLiveCaptionsSettings((s) => {
- const cur = recognitionRef.current;
- if (!s.enabled && cur) {
- recognitionRef.current = null;
- try {
- cur.abort();
- } catch {
- /* ignore */
- }
- } else if (s.enabled && !cur) {
- start();
- }
- });
-
- return () => {
- unsub();
- const cur = recognitionRef.current;
- recognitionRef.current = null;
- if (cur) {
- try {
- cur.abort();
- } catch {
- /* ignore */
- }
- }
- };
- }, [active, room, onLocalCaption]);
-}