Skip to content

Commit 8317107

Browse files
RulaKhaledclaude
andauthored
test(server-utils): Cover the Flue instrumentation (#24266)
Stacked on #24265 — review that first, this is tests only. Unit coverage for the instrumentation, plus an integration suite that drives an agent through a tool call and asserts the `invoke_agent` → `chat` / `execute_tool` hierarchy — including that tool spans are siblings of `chat` rather than children, matching how Flue's own OpenTelemetry adapter projects them. The scenario uses `pi-ai`'s `faux` provider so responses are scripted in-process and no provider key is needed. It is ESM only (`@flue/runtime` has no `require` export condition) and installed per-suite, since `engines.node >= 22.19` would break `yarn install` on the Node 20 lane. Guarded by `conditionalTest({ min: 22 })`, so it skips on the repo's default Node 20. Each case was mutation-tested rather than just run green, which is what caught the tests that passed vacuously and an unreachable branch in `trackSpan` that has since been removed. --------- Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
1 parent 2b3c3a6 commit 8317107

4 files changed

Lines changed: 805 additions & 0 deletions

File tree

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,11 @@
1+
import * as Sentry from '@sentry/node';
2+
import { loggingTransport } from '@sentry-internal/node-integration-tests';
3+
4+
Sentry.init({
5+
traceLifecycle: 'stream',
6+
dsn: 'https://public@dsn.ingest.sentry.io/1337',
7+
release: '1.0',
8+
tracesSampleRate: 1.0,
9+
dataCollection: { genAI: { inputs: false, outputs: false } },
10+
transport: loggingTransport,
11+
});
Lines changed: 42 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
1+
import * as Sentry from '@sentry/node';
2+
import { __flueBindAgentModule, init, instrument, useModel, useTool } from '@flue/runtime';
3+
import { start } from '@flue/runtime/node';
4+
import { fauxAssistantMessage, fauxProvider, fauxToolCall } from '@earendil-works/pi-ai/providers/faux';
5+
import * as v from 'valibot';
6+
7+
// `pi-ai`'s faux provider scripts model responses in-process, so the run is deterministic and needs
8+
// no provider key or mock server. Two steps: a tool call, then the final answer.
9+
instrument(Sentry.createFlueInstrumentation());
10+
11+
const faux = fauxProvider({
12+
provider: 'faux',
13+
models: [{ id: 'faux-model', cost: { input: 1, output: 2, cacheRead: 0, cacheWrite: 0 } }],
14+
});
15+
faux.setResponses([
16+
fauxAssistantMessage(fauxToolCall('get_weather', { city: 'Berlin' }, { id: 'call_1' }), { stopReason: 'toolUse' }),
17+
fauxAssistantMessage('It is 21 degrees and sunny in Berlin.'),
18+
]);
19+
20+
function Hello() {
21+
useModel('faux/faux-model');
22+
useTool({
23+
name: 'get_weather',
24+
description: 'Get the current weather for a city.',
25+
// Without an `input` schema Flue validates the call against an empty one and rejects the
26+
// model's arguments, so the tool never runs and its span settles as an error.
27+
input: v.object({ city: v.string() }),
28+
run: ({ data }) => `It is 21 degrees and sunny in ${data.city}.`,
29+
});
30+
return 'You are a helpful assistant.';
31+
}
32+
__flueBindAgentModule(Hello, { identity: 'Hello' });
33+
34+
await Sentry.startSpan({ name: 'flue-test', op: 'function' }, async () => {
35+
const flue = await start({ agents: [Hello], providers: [faux.provider] });
36+
const agent = init(Hello, { id: 'e2e' });
37+
const receipt = await agent.dispatch('What is the weather in Berlin?');
38+
await agent.read(receipt);
39+
await flue[Symbol.asyncDispose]?.();
40+
});
41+
42+
await Sentry.flush(2000);
Lines changed: 114 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,114 @@
1+
import {
2+
GEN_AI_AGENT_NAME,
3+
GEN_AI_CONVERSATION_ID,
4+
GEN_AI_COST_TOTAL_TOKENS,
5+
GEN_AI_OPERATION_NAME,
6+
GEN_AI_RESPONSE_FINISH_REASONS,
7+
GEN_AI_TOOL_NAME,
8+
GEN_AI_USAGE_INPUT_TOKENS,
9+
GEN_AI_USAGE_OUTPUT_TOKENS,
10+
GEN_AI_USAGE_TOTAL_TOKENS,
11+
} from '@sentry/conventions/attributes';
12+
import { afterAll, expect } from 'vitest';
13+
import { conditionalTest } from '../../../utils';
14+
import { cleanupChildProcesses, createEsmAndCjsTests } from '../../../utils/runner';
15+
16+
// `@flue/runtime` declares `engines.node >= 22.19`, so it can't live in the package's root
17+
// `devDependencies` (that would break `yarn install` on the 20.19 CI matrix). Install it per-suite
18+
// instead, guarded by the `min: 22` skip below.
19+
const FLUE_DEPENDENCIES = {
20+
additionalDependencies: {
21+
'@flue/runtime': '2.0.3',
22+
'@earendil-works/pi-ai': '0.85.1',
23+
valibot: '1.1.0',
24+
},
25+
};
26+
27+
conditionalTest({ min: 22 })('Flue integration', () => {
28+
afterAll(() => {
29+
cleanupChildProcesses();
30+
});
31+
32+
createEsmAndCjsTests(
33+
__dirname,
34+
'scenario.mjs',
35+
'instrument.mjs',
36+
(createRunner, test, mode) => {
37+
// `@flue/runtime` is ESM-only — its `exports` map has no `require` condition, so there is no
38+
// CJS variant of this scenario to run.
39+
if (mode === 'cjs') {
40+
return;
41+
}
42+
43+
test('creates the invoke_agent / chat / execute_tool hierarchy', async () => {
44+
await createRunner()
45+
.expect({
46+
span: container => {
47+
const spans = container.items;
48+
49+
const root = spans.find(span => span.name === 'flue-test')!;
50+
expect(root.is_segment).toBe(true);
51+
52+
// Counted rather than looked up: the interceptor skips the submission wrapper
53+
// operation, so one dispatch opens exactly one agent span, and each turn and tool call
54+
// is spanned once. `find` passes just as happily on a duplicate.
55+
expect(spans.filter(span => span.attributes['sentry.origin']?.value === 'auto.ai.flue')).toHaveLength(4);
56+
57+
const agents = spans.filter(span => span.name === 'invoke_agent Hello');
58+
const chats = spans.filter(span => span.name === 'chat faux-model');
59+
const tools = spans.filter(span => span.name === 'execute_tool get_weather');
60+
expect(agents).toHaveLength(1);
61+
expect(tools).toHaveLength(1);
62+
// One turn asks for the tool, the second answers with its result.
63+
expect(chats).toHaveLength(2);
64+
65+
const agent = agents[0]!;
66+
expect(agent.attributes['sentry.op']?.value).toBe('gen_ai.invoke_agent');
67+
expect(agent.attributes[GEN_AI_OPERATION_NAME]?.value).toBe('invoke_agent');
68+
expect(agent.attributes[GEN_AI_AGENT_NAME]?.value).toBe('Hello');
69+
expect(agent.parent_span_id).toBe(root.span_id);
70+
71+
const conversationId = agent.attributes[GEN_AI_CONVERSATION_ID]?.value;
72+
expect(conversationId).toEqual(expect.any(String));
73+
74+
// Both turns, not just the first: they leave the provider by different paths (a tool
75+
// call, then a final answer) and resolve their parent through separate tracker lookups.
76+
for (const chat of chats) {
77+
expect(chat.attributes['sentry.op']?.value).toBe('gen_ai.chat');
78+
expect(chat.attributes['sentry.origin']?.value).toBe('auto.ai.flue');
79+
expect(chat.attributes[GEN_AI_OPERATION_NAME]?.value).toBe('chat');
80+
expect(chat.attributes[GEN_AI_CONVERSATION_ID]?.value).toBe(conversationId);
81+
expect(chat.parent_span_id).toBe(agent.span_id);
82+
expect(chat.attributes[GEN_AI_USAGE_INPUT_TOKENS]?.value).toBeGreaterThan(0);
83+
expect(chat.attributes[GEN_AI_USAGE_OUTPUT_TOKENS]?.value).toBeGreaterThan(0);
84+
expect(chat.attributes[GEN_AI_USAGE_TOTAL_TOKENS]?.value).toBeGreaterThan(0);
85+
// Flue computes cost itself; no provider SDK reports it. The faux provider prices
86+
// every model at zero, so this only proves the attribute is mapped.
87+
expect(chat.attributes[GEN_AI_COST_TOTAL_TOKENS]?.value).toEqual(expect.any(Number));
88+
}
89+
90+
expect(chats.map(chat => chat.attributes[GEN_AI_RESPONSE_FINISH_REASONS]?.value).sort()).toEqual([
91+
'["stop"]',
92+
'["toolUse"]',
93+
]);
94+
95+
const tool = tools[0]!;
96+
// The tool has to actually run: a schema mismatch still produces a correctly named and
97+
// parented span, so only the status separates a real call from a rejected one.
98+
expect(tool.status).toBe('ok');
99+
expect(tool.attributes['sentry.op']?.value).toBe('gen_ai.execute_tool');
100+
expect(tool.attributes['sentry.origin']?.value).toBe('auto.ai.flue');
101+
expect(tool.attributes[GEN_AI_TOOL_NAME]?.value).toBe('get_weather');
102+
103+
// Tool spans are siblings of `chat` under the agent invocation, matching how Flue's
104+
// own OpenTelemetry adapter projects them.
105+
expect(tool.parent_span_id).toBe(agent.span_id);
106+
},
107+
})
108+
.start()
109+
.completed();
110+
});
111+
},
112+
FLUE_DEPENDENCIES,
113+
);
114+
});

0 commit comments

Comments
 (0)