diff --git a/dev-packages/e2e-tests/test-applications/node-eve/agent/instructions.md b/dev-packages/e2e-tests/test-applications/node-eve/agent/instructions.md index e67cae081e40..ecdf9ca459ad 100644 --- a/dev-packages/e2e-tests/test-applications/node-eve/agent/instructions.md +++ b/dev-packages/e2e-tests/test-applications/node-eve/agent/instructions.md @@ -4,5 +4,6 @@ You are a concise assistant used by an automated end-to-end test. for that place and answer in one short sentence using its result. - When the user asks to count items, call the `count_items` tool with the item names. - When the user asks you to trigger a failure, call the `fail_now` tool. +- When the user asks to classify a ticket, call the `classify_ticket` tool with the ticket text. Do not ask follow-up questions. diff --git a/dev-packages/e2e-tests/test-applications/node-eve/agent/tools/classify_ticket.ts b/dev-packages/e2e-tests/test-applications/node-eve/agent/tools/classify_ticket.ts new file mode 100644 index 000000000000..9dd65a11fd90 --- /dev/null +++ b/dev-packages/e2e-tests/test-applications/node-eve/agent/tools/classify_ticket.ts @@ -0,0 +1,43 @@ +import { experimental_evaluate } from 'ai'; +import { Experimental_EvaluationMockModelV4 } from 'ai/test'; +import { defineTool } from 'eve/tools'; +import { z } from 'zod'; + +// Runs a TypeSafe Jev evaluation so the e2e test can assert its `gen_ai.evaluate` span. The model is +// mocked because the e2e environment has no TypeSafe credentials. +const jev = new Experimental_EvaluationMockModelV4({ + provider: 'gateway', + modelId: 'typesafe-ai/jev', + doEvaluate: async () => ({ + answers: { + wantsRefund: { type: 'boolean', probability: 0.99 }, + department: { + type: 'choice', + choice: 'billing', + probabilities: { billing: 0.64, technical: 0.36 }, + }, + }, + usage: { inputTokens: 275, outputTokens: 20 }, + warnings: [], + }), +}); + +export default defineTool({ + description: 'Classify a support ticket. Call this when asked to classify a ticket.', + inputSchema: z.object({ ticket: z.string().min(1) }), + async execute({ ticket }) { + const { answers } = await experimental_evaluate({ + model: jev, + state: ticket, + questions: { + wantsRefund: { type: 'boolean', instructions: 'Is a refund requested?' }, + department: { + type: 'choice', + instructions: 'Which team should handle this?', + criteria: { billing: 'Charges and refunds', technical: 'Bugs and outages' }, + }, + }, + }); + return answers; + }, +}); diff --git a/dev-packages/e2e-tests/test-applications/node-eve/tests/eve.test.ts b/dev-packages/e2e-tests/test-applications/node-eve/tests/eve.test.ts index 9099824bd7be..04ad1e5151f4 100644 --- a/dev-packages/e2e-tests/test-applications/node-eve/tests/eve.test.ts +++ b/dev-packages/e2e-tests/test-applications/node-eve/tests/eve.test.ts @@ -134,3 +134,34 @@ test('captures errors thrown inside an eve tool', async ({ baseURL }) => { transaction: expect.stringMatching(EVE_AGENT_PATH), }); }); + +test('captures a gen_ai.evaluate span for a Jev call inside an eve tool', async ({ baseURL }) => { + const traceSpansPromise = collectStreamedSpans(APP, spansOfTrace => + ['gen_ai.execute_tool', 'gen_ai.evaluate'].every(op => spansOfTrace.some(span => getSpanOp(span) === op)), + ); + + const sessionId = await runAgentTurn( + baseURL!, + 'Classify this ticket: I cannot log in, and I also want a refund for last month.', + ); + + const traceSpans = await traceSpansPromise; + + const executeTool = traceSpans.find(span => getSpanOp(span) === 'gen_ai.execute_tool'); + const evaluate = traceSpans.find(span => getSpanOp(span) === 'gen_ai.evaluate'); + + expect(executeTool?.attributes?.['gen_ai.tool.name']?.value).toBe('classify_ticket'); + + expect(evaluate?.name).toBe('evaluate typesafe-ai/jev'); + expect(evaluate?.status).toBe('ok'); + expect(evaluate?.attributes?.['sentry.origin']?.value).toBe('auto.vercelai.channel'); + expect(evaluate?.attributes?.['gen_ai.operation.name']?.value).toBe('evaluate'); + expect(evaluate?.attributes?.['gen_ai.request.model']?.value).toBe('typesafe-ai/jev'); + expect(evaluate?.attributes?.['gen_ai.usage.input_tokens']?.value).toBe(275); + expect(evaluate?.attributes?.['gen_ai.usage.output_tokens']?.value).toBe(20); + expect(evaluate?.attributes?.['gen_ai.input.messages']?.value).toContain('refund for last month'); + expect(evaluate?.attributes?.['gen_ai.conversation.id']?.value).toBe(sessionId); + + expect(evaluate?.trace_id).toBe(executeTool?.trace_id); + expect(evaluate?.parent_span_id).toBe(executeTool?.span_id); +});