fix(lotus): denoise — prefetch assets, RNNoise fallback + host notify, suspend while muted, mono/delay graph, typed modules

- Assets (context, worklets, wasm, DFN core) are prepared as soon as the
  flag is seen, so init() under LiveKit's trackChangeLock only wires
  already-loaded pieces; resume timeout 3 s -> 500 ms (#7).
- init failure retries once with rnnoise; success/failure is reported to
  the host as io.lotus.denoise_state so the UI can reflect reality (#8).
- Mic TrackMuted/TrackUnmuted suspend/resume the processor's context so
  no inference runs on silence (#9).
- Every node is explicit mono; the dry path gets a per-model DelayNode so
  the floor mix no longer comb-filters (#24, #25).
- DTLN/DFN dynamic imports are typed and their exports asserted at load,
  feeding the #8 fallback instead of failing silently (#26).
Unit-tested (13 tests across the two files).

Fixes #7
Fixes #8
Fixes #9
Fixes #24
Fixes #25
Fixes #26

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PPmy3tPq869XDW4njjVaKA
This commit is contained in:
Lotus CI
2026-09-13 01:22:20 -04:00
co-authored by Claude Opus 5
parent 872b877248
commit e504a31efd
4 changed files with 740 additions and 87 deletions
+312 -58
View File
@@ -95,7 +95,9 @@ async function fetchWasm(url: string): Promise<ArrayBuffer> {
* lands, and a still-suspended context degrades to (temporary) silence that the
* watcher heals, not a hang.
*/
async function resumeCtx(ctx: AudioContext, timeoutMs = 3_000): Promise<void> {
// [lotus] 500 ms, not 3 s: this still runs under LiveKit's trackChangeLock (#7),
// and the statechange watcher heals a still-suspended context later anyway.
async function resumeCtx(ctx: AudioContext, timeoutMs = 500): Promise<void> {
await Promise.race([
ctx.resume().catch(() => undefined),
new Promise<void>((resolve) => setTimeout(resolve, timeoutMs)),
@@ -127,6 +129,226 @@ interface Graph {
track: MediaStreamTrack;
}
// [lotus #26] Minimal local contracts for the two dynamically-imported ESM
// helpers (not bundled here — see the CONTRACT note above). Their exports are
// asserted at runtime so an asset bump that renames/removes one fails loudly
// (and flows into the rnnoise fallback in lotusDenoise.ts) instead of as a
// vague TypeError deep inside `init()`.
interface DtlnModule {
createNoiseSuppressionAudioWorklet: (
ctx: AudioContext,
opts: { bypassUntilReady: boolean },
) => Promise<MlNode>;
}
interface DfnCore {
initialize: () => Promise<void>;
createAudioWorkletNode: (ctx: AudioContext) => Promise<AudioNode>;
destroy: () => void;
}
interface DfnModule {
DeepFilterNet3Core: new (opts: {
sampleRate: number;
noiseReductionLevel: number;
assetConfig: { cdnUrl: string };
}) => DfnCore;
}
/** Throw a clear error if a dynamic-import module lacks an expected export. */
export function assertModuleExport<T>(
mod: unknown,
name: string,
url: string,
): T {
const exp = (mod as Record<string, unknown> | null | undefined)?.[name];
if (typeof exp !== "function")
throw new Error(
`denoise: ${url} does not export ${name} (got ${typeof exp}) — asset/version mismatch`,
);
return mod as T;
}
async function loadDfnCore(config: LotusDenoiseConfig): Promise<DfnCore> {
const base = config.assetBase;
const url = `${base}deepfilternet/index.esm.js`;
const dfnBase = new URL(`${base}deepfilternet`, window.location.href).href;
const mod = assertModuleExport<DfnModule>(
await import(/* @vite-ignore */ url),
"DeepFilterNet3Core",
url,
);
const core = new mod.DeepFilterNet3Core({
sampleRate: 48_000,
// 60, not 80: full-strength suppression is the main source of the
// "over-processed" character; a lower level keeps voice natural while
// the dry/wet floor handles the noise tail.
noiseReductionLevel: 60,
assetConfig: { cdnUrl: dfnBase },
});
await core.initialize();
return core;
}
async function loadDtlnModule(config: LotusDenoiseConfig): Promise<DtlnModule> {
const url = `${config.assetBase}workadventure/audio-worklet.js`;
return assertModuleExport<DtlnModule>(
await import(/* @vite-ignore */ url),
"createNoiseSuppressionAudioWorklet",
url,
);
}
/** Which wasm file a flat model uses (SIMD build when supported). */
function flatWasmFiles(model: "rnnoise" | "speex"): {
primary: string;
fallback?: string;
} {
const flat = FLAT[model];
const useSimd = model === "rnnoise" && !!flat.simdWasm && supportsSimd();
return useSimd
? { primary: flat.simdWasm!, fallback: flat.wasm }
: { primary: flat.wasm };
}
// [lotus #24] Force every node in the graph to a single, explicitly-downmixed
// channel. Without `channelCountMode: "explicit"` the default ("max") IGNORES
// `channelCount`, so a stereo capture device would feed 2 channels into a
// worklet configured with `maxChannels: 1` and sum a stereo dry copy against a
// mono wet one at the destination.
const MONO: AudioNodeOptions = {
channelCount: 1,
channelCountMode: "explicit",
channelInterpretation: "speakers",
};
// [lotus #25] Algorithmic latency of each model in samples at its native rate,
// used to delay the DRY copy of the floor mix so it lines up with the wet path
// (otherwise the sum comb-filters — a hollow/phasey colouration on voice).
// - rnnoise: 480-sample (10 ms @ 48 kHz) frames; the sapphi worklet buffers
// 128-sample quanta up to one frame, so the wet path lags by one frame.
// - speex: the sapphi speex worklet uses the same 480-sample framing.
// - dtln: 512-sample block / 128 hop @ 16 kHz (~32 ms) per the DTLN paper —
// best-known, unmeasured (the floor is not mixed for dtln, see buildGraph).
// - deepfilternet: 480-sample hop + 2-frame lookahead @ 48 kHz (~30 ms) per
// DeepFilterNet3 — best-known, unmeasured (floor not mixed for dfn either).
const DRY_DELAY_SAMPLES: Record<LotusDenoiseModel, number> = {
rnnoise: 480,
speex: 480,
dtln: 512,
deepfilternet: 1440,
};
/**
* Create the model-rate context and register the flat/gate worklet modules.
* Closes the context (and rethrows) on any failure so nothing half-built leaks.
*/
async function createModelContext(
config: LotusDenoiseConfig,
): Promise<AudioContext> {
const rate = sampleRateFor(config.model);
const ctx = new AudioContext({ sampleRate: rate });
try {
if (ctx.sampleRate !== rate)
throw new Error(`denoise: got ${ctx.sampleRate}Hz, need ${rate}Hz`);
// Flat models register via addModule here; DTLN/DeepFilterNet bring their
// own processor via the dynamic-imported helper (see buildMlNode).
if (config.model === "rnnoise" || config.model === "speex")
await ctx.audioWorklet.addModule(
config.assetBase + FLAT[config.model].script,
);
if (config.gate)
await ctx.audioWorklet.addModule(config.assetBase + GATE.script);
return ctx;
} catch (e) {
await ctx.close().catch(() => undefined);
throw e;
}
}
// [lotus #7] Everything heavy that `init()` needs but that does NOT depend on
// the mic track: the AudioContext + worklet modules, the flat wasm binary, and
// (DFN) the fully-initialised model core. `LocalAudioTrack.setProcessor()`
// holds LiveKit's `trackChangeLock` while awaiting `init()`, so every
// mute/unmute/device-switch queues behind it — prepare these as soon as the
// flag is seen (before any track exists) so `init()` only wires them up.
interface PreparedAssets {
ctx: AudioContext;
dfnCore?: DfnCore;
}
const preparedAssets = new Map<string, Promise<PreparedAssets>>();
const preparedKey = (c: LotusDenoiseConfig): string =>
`${c.model}|${c.gate ? 1 : 0}|${c.assetBase}`;
async function prepareUncached(
config: LotusDenoiseConfig,
): Promise<PreparedAssets> {
const ctx = await createModelContext(config);
try {
let dfnCore: DfnCore | undefined;
if (config.model === "rnnoise" || config.model === "speex") {
const { primary, fallback } = flatWasmFiles(config.model);
// Warm the wasm cache; a SIMD miss is fine — buildMlNode falls back.
await fetchWasm(config.assetBase + primary).catch(async () =>
fallback ? fetchWasm(config.assetBase + fallback) : undefined,
);
} else if (config.model === "dtln") {
await loadDtlnModule(config); // warms the browser's module map
} else {
dfnCore = await loadDfnCore(config);
}
return { ctx, dfnCore };
} catch (e) {
await ctx.close().catch(() => undefined);
throw e;
}
}
/**
* Prefetch/prepare the assets for `config` (idempotent per model). Never
* rejects: a failed prepare is evicted so `init()` simply loads inline and
* surfaces the real error there.
*/
export async function prepareDenoiseAssets(
config: LotusDenoiseConfig,
): Promise<void> {
const key = preparedKey(config);
let p = preparedAssets.get(key);
if (!p) {
p = prepareUncached(config);
void p.catch((e) => {
if (preparedAssets.get(key) === p) preparedAssets.delete(key);
logger.warn(`[lotus] denoise prepare failed (${config.model})`, e);
});
preparedAssets.set(key, p);
}
await p.catch(() => undefined);
}
/** Take (one-shot) the prepared assets for `config`, if any were prepared. */
function claimPreparedAssets(
config: LotusDenoiseConfig,
): Promise<PreparedAssets> | undefined {
const key = preparedKey(config);
const p = preparedAssets.get(key);
if (p) preparedAssets.delete(key);
return p;
}
/** Close any prepared-but-unclaimed contexts (call on feature teardown). */
export async function releasePreparedDenoiseAssets(): Promise<void> {
const all = [...preparedAssets.values()];
preparedAssets.clear();
await Promise.all(
all.map(async (p) =>
p
.then(async (a) => {
safeCall(() => a.dfnCore?.destroy());
if (a.ctx.state !== "closed") await a.ctx.close();
})
.catch(() => undefined),
),
);
}
/**
* A LiveKit audio TrackProcessor that runs Lotus ML noise suppression
* (RNNoise / Speex / DTLN / DeepFilterNet) on the local microphone track, as a
@@ -151,9 +373,31 @@ export class LotusDenoiseProcessor implements TrackProcessor<
private ctx?: AudioContext;
private graph?: Graph;
private ctxStateHandler?: () => void;
private preparedDfnCore?: DfnCore;
// [lotus #9] True while the mic is muted: we suspend our own context so the
// worklet stops running inference on silence, and the statechange watcher
// must not "heal" that intentional suspension.
private micMuted = false;
public constructor(private readonly config: LotusDenoiseConfig) {}
/**
* [lotus #9] Mirror the mic's mute state onto the owned context. EC uses
* `stopMicTrackOnMute: false`, so a muted mic keeps producing (silent) frames
* and the ML worklet would otherwise keep running full inference for the
* whole time the user is muted.
*/
public setMicMuted(muted: boolean): void {
this.micMuted = muted;
const ctx = this.ctx;
if (!ctx || ctx.state === "closed") return;
if (muted) {
if (ctx.state === "running") void ctx.suspend().catch(() => undefined);
} else if (ctx.state === "suspended" && this.graph) {
void ctx.resume().catch(() => undefined);
}
}
public async init(_opts: AudioProcessorOptions): Promise<void> {
try {
await this.ensureContext();
@@ -163,6 +407,9 @@ export class LotusDenoiseProcessor implements TrackProcessor<
// Don't orphan the owned context if graph construction fails (browsers
// cap live AudioContexts, so repeated failed inits could exhaust them).
// The caller degrades to the raw mic; we just release our resources.
const core = this.preparedDfnCore;
this.preparedDfnCore = undefined;
if (core) safeCall(() => core.destroy());
await this.closeContext();
throw e;
}
@@ -194,6 +441,9 @@ export class LotusDenoiseProcessor implements TrackProcessor<
this.disposeGraph(this.graph);
this.graph = undefined;
this.processedTrack = undefined;
const core = this.preparedDfnCore;
this.preparedDfnCore = undefined;
if (core) safeCall(() => core.destroy());
await this.closeContext();
}
@@ -209,7 +459,7 @@ export class LotusDenoiseProcessor implements TrackProcessor<
if (ctx.state !== "closed") await ctx.close().catch(() => undefined);
}
/** Create (once) the model-rate context + register the flat worklet modules. */
/** Adopt the prepared context (or create one) + install the state watcher. */
private async ensureContext(): Promise<void> {
const rate = sampleRateFor(this.config.model);
if (
@@ -217,36 +467,43 @@ export class LotusDenoiseProcessor implements TrackProcessor<
this.ctx.state !== "closed" &&
this.ctx.sampleRate === rate
) {
if (this.ctx.state === "suspended") await resumeCtx(this.ctx);
if (this.ctx.state === "suspended" && !this.micMuted)
await resumeCtx(this.ctx);
return;
}
await this.closeContext();
const ctx = new AudioContext({ sampleRate: rate });
// [lotus #7] Prefer the context/modules/model prepared before
// setProcessor() was called; only load inline if nothing was prepared
// (e.g. the rnnoise fallback path, or a second processor after a
// republish).
const claimed = await claimPreparedAssets(this.config)?.catch(
() => undefined,
);
let ctx: AudioContext;
if (claimed && claimed.ctx.state !== "closed") {
ctx = claimed.ctx;
this.preparedDfnCore = claimed.dfnCore;
} else {
ctx = await createModelContext(this.config);
}
try {
if (ctx.sampleRate !== rate)
throw new Error(`denoise: got ${ctx.sampleRate}Hz, need ${rate}Hz`);
// Auto-resume if the OS/browser suspends the context mid-call (mobile
// backgrounding, audio interruption): the dest node otherwise emits
// silence with no recovery. Only resume while a graph is live.
// silence with no recovery. Only resume while a graph is live and the
// suspension isn't our own mute suspension (#9).
const onStateChange = (): void => {
if (ctx.state === "suspended" && this.graph)
if (ctx.state === "suspended" && this.graph && !this.micMuted)
void ctx.resume().catch(() => undefined);
};
ctx.addEventListener("statechange", onStateChange);
// Flat models register via addModule here; DTLN/DeepFilterNet bring their
// own processor via the dynamic-imported helper (see buildMlNode).
if (this.config.model === "rnnoise" || this.config.model === "speex")
await ctx.audioWorklet.addModule(
this.config.assetBase + FLAT[this.config.model].script,
);
if (this.config.gate)
await ctx.audioWorklet.addModule(this.config.assetBase + GATE.script);
// The action can arrive via host postMessage, not a gesture in this
// iframe, so the context can start suspended — resume without hanging.
if (ctx.state === "suspended") await resumeCtx(ctx);
if (ctx.state === "suspended" && !this.micMuted) await resumeCtx(ctx);
// Attached while already muted (#9): don't let a prepared, running
// context burn inference until the first unmute.
else if (ctx.state === "running" && this.micMuted)
await ctx.suspend().catch(() => undefined);
this.ctx = ctx;
this.ctxStateHandler = onStateChange;
@@ -260,7 +517,7 @@ export class LotusDenoiseProcessor implements TrackProcessor<
private async buildGraph(track: MediaStreamTrack): Promise<Graph> {
const ctx = this.ctx!;
const source = ctx.createMediaStreamSource(new MediaStream([track]));
const dest = ctx.createMediaStreamDestination();
const dest = new MediaStreamAudioDestinationNode(ctx, MONO);
const nodes: AudioNode[] = [];
const disposes: (() => void)[] = [];
@@ -277,6 +534,7 @@ export class LotusDenoiseProcessor implements TrackProcessor<
// made the threshold operate on pre-denoise levels. Gate the residual.
if (this.config.gate) {
const gate = new AudioWorkletNode(ctx, GATE.name, {
...MONO,
processorOptions: {
openThreshold: this.config.gateThreshold,
closeThreshold: this.config.gateThreshold - 5,
@@ -289,11 +547,11 @@ export class LotusDenoiseProcessor implements TrackProcessor<
nodes.push(gate);
}
// Only mix a dry floor for the LOW-LATENCY flat models (RNNoise/Speex).
// DTLN/DeepFilterNet add tens of ms of algorithmic latency, so summing an
// undelayed dry copy would comb-filter the voice — for those we rely on
// the model's own level (e.g. DFN noiseReductionLevel) instead. RNNoise is
// also where the "robotic/underwater" reports come from, so this targets it.
// Only mix a dry floor for the flat models (RNNoise/Speex), whose
// framing latency is known exactly (DRY_DELAY_SAMPLES); the DTLN/DFN
// figures are best-known estimates, so for those we rely on the model's
// own level (e.g. DFN noiseReductionLevel) instead. RNNoise is also where
// the "robotic/underwater" reports come from, so this targets it.
const lowLatency =
this.config.model === "rnnoise" || this.config.model === "speex";
const floor = lowLatency
@@ -305,17 +563,25 @@ export class LotusDenoiseProcessor implements TrackProcessor<
// (kills the "underwater"/pumping artifact). During speech (denoised ≈
// original) the two sum back to ~unity; in noise-only gaps the output
// floors at `floor` × original instead of digital silence.
const wetGain = ctx.createGain();
wetGain.gain.value = 1 - floor;
const wetGain = new GainNode(ctx, { ...MONO, gain: 1 - floor });
wetHead.connect(wetGain);
wetGain.connect(dest);
nodes.push(wetGain);
const dryGain = ctx.createGain();
dryGain.gain.value = floor;
source.connect(dryGain);
// [lotus #25] Delay the dry copy by the model's algorithmic latency so
// it sums in phase with the (framed, hence delayed) wet path instead
// of comb-filtering against it.
const delaySec = DRY_DELAY_SAMPLES[this.config.model] / ctx.sampleRate;
const dryDelay = new DelayNode(ctx, {
...MONO,
maxDelayTime: Math.max(delaySec, 1 / ctx.sampleRate),
delayTime: delaySec,
});
const dryGain = new GainNode(ctx, { ...MONO, gain: floor });
source.connect(dryDelay);
dryDelay.connect(dryGain);
dryGain.connect(dest);
nodes.push(dryGain);
nodes.push(dryDelay, dryGain);
} else {
wetHead.connect(dest);
}
@@ -350,48 +616,36 @@ export class LotusDenoiseProcessor implements TrackProcessor<
if (model === "dtln") {
// Self-contained ESM that resolves its own processor + LiteRT wasm +
// TFLite models. bypassUntilReady passes raw audio until the model loads.
const mod = await import(
/* @vite-ignore */ `${base}workadventure/audio-worklet.js`
);
return (await mod.createNoiseSuppressionAudioWorklet(ctx, {
const mod = await loadDtlnModule(this.config);
return await mod.createNoiseSuppressionAudioWorklet(ctx, {
bypassUntilReady: true,
})) as MlNode;
});
}
if (model === "deepfilternet") {
const dfnBase = new URL(`${base}deepfilternet`, window.location.href)
.href;
const mod = await import(
/* @vite-ignore */ `${base}deepfilternet/index.esm.js`
);
const core = new mod.DeepFilterNet3Core({
sampleRate: 48_000,
// 60, not 80: full-strength suppression is the main source of the
// "over-processed" character; a lower level keeps voice natural while
// the dry/wet floor handles the noise tail.
noiseReductionLevel: 60,
assetConfig: { cdnUrl: dfnBase },
});
await core.initialize();
const node = (await core.createAudioWorkletNode(ctx)) as AudioNode;
// [lotus #7] Use the core initialised by prepareDenoiseAssets() if we
// have one (first graph); later rebuilds (restart) load a fresh core.
const prepared = this.preparedDfnCore;
this.preparedDfnCore = undefined;
const core = prepared ?? (await loadDfnCore(this.config));
const node = await core.createAudioWorkletNode(ctx);
return { node, dispose: () => safeCall(() => core.destroy()) };
}
// Flat sapphi worklet (rnnoise/speex).
const flat = FLAT[model];
const useSimd = model === "rnnoise" && !!flat.simdWasm && supportsSimd();
const wasmFile = useSimd ? flat.simdWasm! : flat.wasm;
const { primary, fallback } = flatWasmFiles(model);
let wasmBinary: ArrayBuffer;
try {
wasmBinary = await fetchWasm(base + wasmFile);
wasmBinary = await fetchWasm(base + primary);
} catch (e) {
if (useSimd) {
wasmCache.delete(base + wasmFile);
wasmBinary = await fetchWasm(base + flat.wasm); // fall back to non-SIMD
if (fallback) {
wasmCache.delete(base + primary);
wasmBinary = await fetchWasm(base + fallback); // fall back to non-SIMD
} else throw e;
}
const node = new AudioWorkletNode(ctx, flat.name, {
channelCount: 1,
...MONO,
numberOfInputs: 1,
numberOfOutputs: 1,
processorOptions: { maxChannels: 1, wasmBinary },