import assert from 'node:assert/strict';
import test, { after, type TestContext } from 'node:test';
import { createRequire } from 'node:module';
import { ReadableStream, type ReadableStreamDefaultController } from 'node:stream/web';
import { asLanguageCode, initializeLogger, llm, stt, telemetry } from '@livekit/agents';
import { BasicTracerProvider } from '@opentelemetry/sdk-trace-base';
import { AudioRecognition, type EndOfTurnInfo, type RecognitionHooks } from '../node_modules/@livekit/agents/dist/voice/audio_recognition.js';
import { VAD, VADEventType, type VADEvent, type VADStream } from '../node_modules/@livekit/agents/dist/vad.js';
// Dependency regression only: internal event entry points control the ordering.
// The package itself is unmodified. No provider, network, credentials or app code.
initializeLogger({ pretty: false, level: 'silent' });
const require = createRequire(import.meta.url);
const cjs = require('@livekit/agents') as typeof import('@livekit/agents');
cjs.initializeLogger({ pretty: false, level: 'silent' });
const CjsRecognition = require('../node_modules/@livekit/agents/dist/voice/audio_recognition.cjs').AudioRecognition as typeof AudioRecognition;
const processors = new telemetry.FanoutSpanProcessor();
const provider = new BasicTracerProvider({ spanProcessors: [processors] });
for (const sdk of [{ telemetry }, cjs]) {
sdk.telemetry.genAI.setCaptureContent(false);
sdk.telemetry.setTracerProvider(provider, {
registerSpanProcessor: processor => processors.add(processor), allowPii: false,
});
}
after(() => provider.shutdown());
class ScriptedVAD extends VAD {
label = 'synthetic-onset-reproducer';
private controller!: ReadableStreamDefaultController<VADEvent>;
private events = new ReadableStream<VADEvent>({ start: c => { this.controller = c; } });
constructor() { super({ updateInterval: 1 }); }
push(event: VADEvent) { this.controller.enqueue(event); }
finish() { this.controller.close(); }
stream(): VADStream {
return {
// Model a VAD event already queued when STT completes the prior turn.
updateInputStream() {}, detachInputStream() {}, close() {}, flush() {},
[Symbol.asyncIterator]: () => this.events[Symbol.asyncIterator](),
} as unknown as VADStream;
}
}
function fixture(t: TestContext, Runtime: typeof AudioRecognition) {
let now = 1000;
t.mock.method(Date, 'now', () => now);
const turns: EndOfTurnInfo[] = [];
const noop = () => {};
const hooks: RecognitionHooks = {
onInterruption: noop, onBackchannelConfirmed: noop, onStartOfSpeech: noop,
onVADInferenceDone: noop, onEndOfSpeech: noop, onInterimTranscript: noop,
onFinalTranscript: noop, onPreemptiveGeneration: noop, onAgentBackchannelOpportunity: noop,
onUserTurnExceeded: noop, onTranscriptionTimeout: noop, onEotPrediction: noop,
retrieveChatCtx: () => llm.ChatContext.empty(),
onEndOfTurn: async info => { turns.push(info); return true; },
};
const vad = new ScriptedVAD();
const recognition = new Runtime({
recognitionHooks: hooks, stt: async () => new ReadableStream(), vad,
turnDetectionMode: 'stt', minEndpointingDelay: 0, maxEndpointingDelay: 0,
});
const events = recognition as unknown as {
createVadTask(vad: VAD, signal: AbortSignal): Promise<void>;
onSTTEvent(event: stt.SpeechEvent): Promise<void>;
};
const loop = events.createVadTask(vad, new AbortController().signal);
t.after(async () => { vad.finish(); await loop; });
return {
async vad(type: VADEventType, at: number) {
now = at;
vad.push({ type, samplesIndex: 0, timestamp: at, speechDuration: 100,
silenceDuration: 0, frames: [], probability: 1, inferenceDuration: 5,
speaking: type === VADEventType.START_OF_SPEECH,
rawAccumulatedSilence: 0, rawAccumulatedSpeech: 0 });
await new Promise<void>(resolve => setImmediate(resolve));
},
async complete(at: number) {
now = at;
const before = turns.length;
const alternatives: [stt.SpeechData] = [{ language: asLanguageCode('en'),
text: 'Example utterance.', startTime: 0, endTime: 0, confidence: 1 }];
await events.onSTTEvent({ type: stt.SpeechEventType.FINAL_TRANSCRIPT, alternatives });
await events.onSTTEvent({ type: stt.SpeechEventType.END_OF_SPEECH, alternatives });
await recognition.waitForEndOfTurnTask();
assert.equal(turns.length, before + 1, 'one final is committed once');
return turns.at(-1)!;
},
};
}
for (const [name, Runtime] of [['ESM', AudioRecognition], ['CommonJS', CjsRecognition]] as const) {
test(`${name}: late VAD end cannot become next turn onset`, async t => {
const f = fixture(t, Runtime);
await f.vad(VADEventType.START_OF_SPEECH, 1000);
await f.complete(2000);
await f.vad(VADEventType.END_OF_SPEECH, 2010);
await f.vad(VADEventType.START_OF_SPEECH, 10000);
const turn = await f.complete(11000);
assert.equal(turn.startedSpeakingAt, 9895,
'expected genuine onset 10000 - 100 - 5, not prior late-end trace start');
});
test(`${name}: end without start cannot manufacture onset`, async t => {
const f = fixture(t, Runtime);
await f.vad(VADEventType.END_OF_SPEECH, 10000);
const turn = await f.complete(11000);
assert.equal(turn.startedSpeakingAt, undefined, 'no actual start event was observed');
});
test(`${name}: multiple segments preserve first genuine onset`, async t => {
const f = fixture(t, Runtime);
await f.vad(VADEventType.START_OF_SPEECH, 7000);
await f.vad(VADEventType.END_OF_SPEECH, 7600);
await f.vad(VADEventType.START_OF_SPEECH, 8500);
assert.equal((await f.complete(9000)).startedSpeakingAt, 6895);
});
test(`${name}: ordinary fresh onset is available`, async t => {
const f = fixture(t, Runtime);
await f.vad(VADEventType.START_OF_SPEECH, 10000);
assert.equal((await f.complete(11000)).startedSpeakingAt, 9895);
});
}
Summary
On the unmodified npm release
@livekit/agents@1.8.0, a delayed VADEND_OF_SPEECHafter a committed STT turn can initialize the next turn'sstartedSpeakingAtfrom a telemetry-span timestamp. A subsequent genuine VAD start does not replace it.A related case creates a nonempty onset when only a VAD end was observed. This matters to consumers of speech timing and ordering; a tracing-span start is not necessarily speech onset.
Reproduction and actual result
The self-contained dependency regression below uses the actual packaged
AudioRecognitionand tracing implementation, with scripted VAD/STT event ordering. It does not use a network, model, provider credentials, application code, or recordings. Internal recognition entry points are used only to reproduce the dependency issue.Eight tests run against ESM and CommonJS: 4 pass, 4 fail on clean 1.8.0:
The scripted VAD's flush deliberately does not discard already-queued events. This isolates late-event handling; it is not a claim that every bundled VAD flush produces this ordering.
Environment and command
@livekit/agents: exactly 1.8.0 from npm; no patch-package or runtime modifications.@opentelemetry/sdk-trace-base: 2.11.0;tsx: 4.23.13.In an empty directory, install these dependencies, save the code below as
repro/speech-onset.test.mts, and run:Complete runnable reproducer
Source observation / requested behavior
In the 1.8.0
audio_recognition.ts,ensureUserTurnSpan()initializesuserTurnStartwhen it creates a span and the value is undefined. The method is called both for genuine starts and for later end/turn processing. Therefore span creation can provide an onset, and the next real start retains it.Please keep first genuine speech onset distinct from telemetry span creation, while retaining the passing behavior that multiple speech segments within one uncommitted turn preserve their first onset. If this event sequence is intentionally unsupported, please clarify the supported timing/ownership contract.
This is separate from #2382's speech-end/transcription-delay anchor change. The report does not propose altering recognition thresholds or any downstream consent logic.