- Injection is gated on localParticipant.isMicrophoneEnabled and replies
{ played:false, reason:"muted" } (host UI follow-up in cinny) (#13).
- One lazily created module-level AudioContext/destination for all
clips, ref-counted and closed on last teardown; per clip only a
BufferSource + Gain. The replace-mode race handling is preserved and
three latent dangling-placeholder paths are closed (#14).
Fixes #13
Fixes #14
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PPmy3tPq869XDW4njjVaKA
321 lines
12 KiB
TypeScript
321 lines
12 KiB
TypeScript
/*
|
|
Copyright 2026 Lotus Guild
|
|
|
|
SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-Element-Commercial
|
|
Please see LICENSE in the repository root for full details.
|
|
*/
|
|
|
|
import { type Room as LivekitRoom, Track } from "livekit-client";
|
|
import { logger } from "matrix-js-sdk/lib/logger";
|
|
import { type IWidgetApiRequest } from "matrix-widget-api";
|
|
|
|
import { type CallViewModel } from "../state/CallViewModel/CallViewModel";
|
|
import { widget } from "../widget";
|
|
import { LotusWidgetActions } from "./lotusActions";
|
|
import { lotusFlag } from "./lotusWidget";
|
|
|
|
/** Hard cap so a malformed/huge clip can't hold a published track open forever. */
|
|
const MAX_CLIP_MS = 30_000;
|
|
|
|
// [lotus] One shared module-level AudioContext + MediaStreamAudioDestinationNode
|
|
// for ALL injected clips (#14), instead of `new AudioContext()` per clip. An
|
|
// in-call page already sits close to Chrome's per-document AudioContext limit
|
|
// (~6) between useAudioContext, MatrixAudioRenderer, LiveKit's own Room
|
|
// context and LotusDenoiseProcessor; rapid clip replacement (replace-mode
|
|
// closing the previous clip's context in the background) could transiently
|
|
// exceed the cap and make `new AudioContext()` throw. Lazily created on first
|
|
// use, ref-counted by the number of active `startLotusAudioInject` instances,
|
|
// and closed only when the last one tears down.
|
|
let sharedCtx: AudioContext | undefined;
|
|
let sharedDest: MediaStreamAudioDestinationNode | undefined;
|
|
let handlerCount = 0;
|
|
|
|
function acquireSharedAudio(): {
|
|
ctx: AudioContext;
|
|
dest: MediaStreamAudioDestinationNode;
|
|
} {
|
|
if (!sharedCtx || sharedCtx.state === "closed") {
|
|
sharedCtx = new AudioContext();
|
|
sharedDest = sharedCtx.createMediaStreamDestination();
|
|
}
|
|
return { ctx: sharedCtx, dest: sharedDest! };
|
|
}
|
|
|
|
/**
|
|
* Handle the host's `io.lotus.inject_audio` toWidget action (#3): mix a
|
|
* soundboard clip into the call so other participants hear it.
|
|
*
|
|
* Rather than splice into the local mic track (which would fight the denoise
|
|
* pipeline), we publish the clip as a separate `Unknown`-source LiveKit audio
|
|
* track — which `MatrixAudioRenderer` already renders for valid call members —
|
|
* and unpublish it when the clip ends. This is the real call-audio injection
|
|
* that was impossible against the prebuilt EC bundle (LiveKit's
|
|
* LocalParticipant lived in EC's module scope).
|
|
*
|
|
* Action data: `{ url: string, volume?: number }`. `url` must be an https/blob
|
|
* URL (the host resolves mxc → media URL).
|
|
*
|
|
* No effect unless the host sends the action. Returns a teardown function that
|
|
* also aborts any clip still playing.
|
|
*/
|
|
export function startLotusAudioInject(vm: CallViewModel): () => void {
|
|
const w = widget;
|
|
if (!w) return () => undefined;
|
|
|
|
// [lotus] Count this instance toward the shared AudioContext's lifetime
|
|
// (#14) — closed only once the last active instance tears down.
|
|
handlerCount++;
|
|
|
|
// Track the set of connected LiveKit rooms to publish into. Drive off the
|
|
// LOCAL participant's connection(s), not `livekitRoomItems$` — that stream
|
|
// omits rooms with no remote members, so inject would no-op while you're
|
|
// alone. Map the connections to their livekit rooms like `lotusDenoise.ts`.
|
|
let rooms: LivekitRoom[] = [];
|
|
const sub = vm.allConnections$.subscribe((data) => {
|
|
rooms = data.getConnections().map((c) => c.livekitRoom);
|
|
});
|
|
|
|
// In-flight clips, so we can abort them on teardown (unmount / vm change /
|
|
// call leave) instead of leaving audio blasting to peers.
|
|
const activeClips = new Set<() => void>();
|
|
|
|
const handler = (ev: CustomEvent<IWidgetApiRequest>): void => {
|
|
if (!lotusFlag("lotusAudioInject")) {
|
|
// Always ack so the transport doesn't hang, but only act when the host
|
|
// has explicitly opted in: audio-inject publishes under the local
|
|
// user's identity, so it must not be silently armed for every call.
|
|
w.api.transport.reply(ev.detail, {});
|
|
return;
|
|
}
|
|
const data = ev.detail.data as
|
|
| { url?: unknown; volume?: unknown }
|
|
| undefined;
|
|
const url = typeof data?.url === "string" ? safeMediaUrl(data.url) : null;
|
|
if (!url) {
|
|
w.api.transport.reply(ev.detail, {});
|
|
logger.warn("[lotus] inject_audio: missing/invalid url");
|
|
return;
|
|
}
|
|
const volume =
|
|
typeof data?.volume === "number" && data.volume >= 0 && data.volume <= 1
|
|
? data.volume
|
|
: 1;
|
|
|
|
// [lotus] Gate on the local mic being enabled (#13): the clip is
|
|
// published as an independent track, fully decoupled from the mic
|
|
// publication's mute state, so without this a muted (or push-to-talk
|
|
// idle) user could still transmit soundboard audio under their own
|
|
// identity — breaking the "I am muted, nothing I do makes noise" mental
|
|
// model. Reply with a machine-readable reason so cinny's soundboard UI
|
|
// can surface a hint instead of the click silently doing nothing.
|
|
const micEnabled = rooms[0]?.localParticipant.isMicrophoneEnabled ?? true;
|
|
if (!micEnabled) {
|
|
w.api.transport.reply(ev.detail, { played: false, reason: "muted" });
|
|
return;
|
|
}
|
|
|
|
w.api.transport.reply(ev.detail, {});
|
|
void playInjectedClip(url, volume, rooms, activeClips).catch((e) =>
|
|
logger.warn("[lotus] inject_audio failed", e),
|
|
);
|
|
};
|
|
|
|
w.lazyActions.on(LotusWidgetActions.InjectAudio, handler);
|
|
return () => {
|
|
sub.unsubscribe();
|
|
w.lazyActions.off(LotusWidgetActions.InjectAudio, handler);
|
|
// Abort anything still playing.
|
|
// oxlint-disable-next-line unicorn/no-useless-spread -- the spread is a
|
|
// required defensive copy: abort() deletes from activeClips while we
|
|
// iterate it.
|
|
// The spread is a required defensive copy: abort() deletes from
|
|
// activeClips while we iterate it.
|
|
// eslint-disable-next-line unicorn/no-useless-spread
|
|
for (const abort of [...activeClips]) abort();
|
|
// [lotus] Close the shared AudioContext only when the last active
|
|
// instance tears down (#14).
|
|
handlerCount--;
|
|
if (handlerCount === 0 && sharedCtx) {
|
|
const ctx = sharedCtx;
|
|
sharedCtx = undefined;
|
|
sharedDest = undefined;
|
|
void ctx.close().catch(() => undefined);
|
|
}
|
|
};
|
|
}
|
|
|
|
/** Only allow fetchable media URLs; never same-origin credentialed GETs etc. */
|
|
function safeMediaUrl(raw: string): string | null {
|
|
try {
|
|
const u = new URL(raw, window.location.href);
|
|
return u.protocol === "https:" || u.protocol === "blob:" ? u.href : null;
|
|
} catch {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
async function playInjectedClip(
|
|
url: string,
|
|
volume: number,
|
|
rooms: LivekitRoom[],
|
|
activeClips: Set<() => void>,
|
|
): Promise<void> {
|
|
if (rooms.length === 0) {
|
|
logger.warn("[lotus] inject_audio: no connected rooms");
|
|
return;
|
|
}
|
|
|
|
// Max ONE clip at a time (replace mode): stop any in-flight or playing clip
|
|
// before starting a new one, so clips can't overlap or be spammed.
|
|
// The spread is a required defensive copy: abort() deletes from
|
|
// activeClips while we iterate it.
|
|
// eslint-disable-next-line unicorn/no-useless-spread
|
|
for (const abort of [...activeClips]) abort();
|
|
|
|
// A second inject action can arrive while THIS one is still awaiting its
|
|
// fetch/decode/publish — before its real cleanup() exists. Register a
|
|
// synchronous placeholder abort NOW, BEFORE the first await, so the
|
|
// replace-mode loop above (run by that later action) cancels this one;
|
|
// otherwise both clips would sail past their awaits and DOUBLE-PUBLISH. The
|
|
// placeholder aborts the in-flight fetch and flips `aborted`, which we check
|
|
// after every await; the real cleanup() replaces it once the track is live.
|
|
let aborted = false;
|
|
const controller = new AbortController();
|
|
const placeholder = (): void => {
|
|
aborted = true;
|
|
controller.abort();
|
|
activeClips.delete(placeholder);
|
|
};
|
|
activeClips.add(placeholder);
|
|
|
|
let resp: Response;
|
|
try {
|
|
resp = await fetch(url, {
|
|
credentials: "omit",
|
|
mode: "cors",
|
|
signal: controller.signal,
|
|
});
|
|
} catch (e) {
|
|
// Superseded by a newer clip mid-fetch — expected, not a failure.
|
|
if (aborted) return;
|
|
throw e;
|
|
}
|
|
if (aborted) return;
|
|
if (!resp.ok) {
|
|
activeClips.delete(placeholder);
|
|
throw new Error(`fetch ${url} -> ${resp.status}`);
|
|
}
|
|
const arrayBuffer = await resp.arrayBuffer();
|
|
if (aborted) return;
|
|
|
|
// [lotus] Reuse the shared module-level context/destination (#14) rather
|
|
// than `new AudioContext()` per clip — see the declaration above.
|
|
const { ctx, dest } = acquireSharedAudio();
|
|
// The action arrives via host postMessage, not a gesture in this iframe, so
|
|
// the context may start suspended — resume it or the clip is silent and
|
|
// `onended` never fires. A no-op if an earlier clip already resumed it.
|
|
try {
|
|
await ctx.resume();
|
|
} catch {
|
|
/* best effort */
|
|
}
|
|
if (aborted) return;
|
|
if (ctx.state !== "running")
|
|
logger.warn(`[lotus] inject_audio: AudioContext is ${ctx.state}`);
|
|
|
|
let buffer: AudioBuffer;
|
|
try {
|
|
buffer = await ctx.decodeAudioData(arrayBuffer);
|
|
} catch (e) {
|
|
activeClips.delete(placeholder);
|
|
throw e;
|
|
}
|
|
if (aborted) return;
|
|
|
|
// Per clip, only the BufferSource/GainNode are created (#14) — the shared
|
|
// context/destination are reused across every clip.
|
|
const gain = ctx.createGain();
|
|
gain.gain.value = volume;
|
|
const source = ctx.createBufferSource();
|
|
source.buffer = buffer;
|
|
source.connect(gain).connect(dest);
|
|
|
|
const mst = dest.stream.getAudioTracks()[0];
|
|
if (!mst) {
|
|
source.disconnect();
|
|
gain.disconnect();
|
|
activeClips.delete(placeholder);
|
|
throw new Error("no audio track from destination");
|
|
}
|
|
|
|
// Publish (a clone of) the clip track to every connected room.
|
|
const publications = await Promise.all(
|
|
rooms.map(async (room) => {
|
|
const clone = mst.clone();
|
|
try {
|
|
const pub = await room.localParticipant.publishTrack(clone, {
|
|
source: Track.Source.Unknown,
|
|
name: "lotus-soundboard",
|
|
dtx: false,
|
|
red: false,
|
|
});
|
|
return { room, pub };
|
|
} catch (e) {
|
|
clone.stop(); // don't leak the clone if publish failed
|
|
logger.warn("[lotus] inject_audio: publish failed", e);
|
|
return null;
|
|
}
|
|
}),
|
|
);
|
|
|
|
let cleanedUp = false;
|
|
const cleanup = (): void => {
|
|
if (cleanedUp) return;
|
|
cleanedUp = true;
|
|
activeClips.delete(cleanup);
|
|
try {
|
|
source.stop();
|
|
} catch {
|
|
/* already stopped */
|
|
}
|
|
// [lotus] Dispose only this clip's own nodes (#14) — the shared
|
|
// AudioContext/destination outlive it and are closed separately, only
|
|
// when the last startLotusAudioInject instance tears down.
|
|
source.disconnect();
|
|
gain.disconnect();
|
|
for (const entry of publications) {
|
|
if (entry?.pub.track)
|
|
void entry.room.localParticipant
|
|
.unpublishTrack(entry.pub.track, true)
|
|
.catch(() => undefined);
|
|
}
|
|
};
|
|
// Swap the synchronous placeholder for the real cleanup: from here an abort
|
|
// (teardown or a newer clip) must unpublish the LIVE track, not just cancel a
|
|
// fetch. This delete+add is synchronous (no await), so a newer clip's
|
|
// replace-mode loop always sees exactly one of {placeholder, cleanup}.
|
|
activeClips.delete(placeholder);
|
|
activeClips.add(cleanup);
|
|
|
|
// If a newer clip aborted us WHILE we were publishing, tear down now so we
|
|
// don't leave an orphan track published after it ran its replace-mode loop.
|
|
if (aborted) {
|
|
cleanup();
|
|
return;
|
|
}
|
|
|
|
source.onended = cleanup;
|
|
// Safety net: clip metadata can lie (NaN/huge duration), so force teardown
|
|
// after a sane, capped delay.
|
|
const durationMs = Number.isFinite(buffer.duration)
|
|
? buffer.duration * 1000 + 500
|
|
: MAX_CLIP_MS;
|
|
const guard = setTimeout(
|
|
cleanup,
|
|
Math.min(MAX_CLIP_MS, Math.max(0, durationMs)),
|
|
);
|
|
source.addEventListener("ended", () => clearTimeout(guard));
|
|
|
|
source.start();
|
|
}
|