diff --git a/.changeset/fix-eve-migrate-provider-to-trace-policy.md b/.changeset/fix-eve-migrate-provider-to-trace-policy.md new file mode 100644 index 000000000..3bcb1b9ea --- /dev/null +++ b/.changeset/fix-eve-migrate-provider-to-trace-policy.md @@ -0,0 +1,5 @@ +--- +"braintrust": patch +--- + +fix(eve): migrate provider to tracePolicy diff --git a/e2e/config/pr-comment-scenarios.json b/e2e/config/pr-comment-scenarios.json index 3b2ac8370..adead5482 100644 --- a/e2e/config/pr-comment-scenarios.json +++ b/e2e/config/pr-comment-scenarios.json @@ -656,6 +656,7 @@ "variants": [ { "variantKey": "eve-v0", "label": "v0 pinned" }, { "variantKey": "eve-v0-provider", "label": "v0 provider minimum" }, + { "variantKey": "eve-v0-latest-pinned", "label": "v0 latest pinned" }, { "variantKey": "eve-v0-latest", "label": "v0 latest" } ] }, diff --git a/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest-pinned.span-tree.json b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest-pinned.span-tree.json new file mode 100644 index 000000000..6be45b535 --- /dev/null +++ b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest-pinned.span-tree.json @@ -0,0 +1,902 @@ +{ + "span_tree": [ + { + "name": "eve.turn", + "type": "task", + "children": [ + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario", + "role": "user" + } + ], + "output": [ + { + "finish_reason": "tool_calls", + "index": 0, + "message": { + "content": null, + "reasoning": [ + { + "content": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n" + } + ], + "role": "assistant", + "tool_calls": [ + { + "function": { + "arguments": "{\"message\":\"Run the Braintrust Eve instrumentation e2e scenario\"}", + "name": "researcher" + }, + "id": "", + "type": "function" + }, + { + "function": { + "arguments": "{\"url\":\"https://eve.dev/docs/guides/instrumentation\"}", + "name": "read" + }, + "id": "", + "type": "function" + } + ] + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 514, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 3670, + "tokens": 4184 + } + }, + { + "name": "researcher", + "type": "tool", + "children": [ + { + "name": "eve.turn", + "type": "task", + "children": [ + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", + "role": "user" + } + ], + "output": [ + { + "finish_reason": "tool_calls", + "index": 0, + "message": { + "content": null, + "reasoning": [ + { + "content": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that.\n\nThe task is to search for \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that exact query. Once I get the results, I need to provide a concise answer with the title and URL from the search. Let me make sure I follow the instructions: one tool call, then a single sentence with the result. Alright, let's execute the search.\n" + } + ], + "role": "assistant", + "tool_calls": [ + { + "function": { + "arguments": "{\"query\":\"Braintrust Eve instrumentation e2e scenario\"}", + "name": "search" + }, + "id": "", + "type": "function" + } + ] + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 169, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 3057, + "tokens": 3226 + } + }, + { + "name": "search", + "type": "tool", + "children": [], + "input": { + "query": "Braintrust Eve instrumentation e2e scenario" + }, + "output": { + "query": "Braintrust Eve instrumentation e2e scenario", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "metadata": { + "eve.action_kind": "tool-call", + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + }, + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that.\n\nThe task is to search for \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that exact query. Once I get the results, I need to provide a concise answer with the title and URL from the search. Let me make sure I follow the instructions: one tool call, then a single sentence with the result. Alright, let's execute the search.\n", + "type": "reasoning" + }, + { + "input": { + "query": "Braintrust Eve instrumentation e2e scenario" + }, + "toolCallId": "", + "toolName": "search", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "query": "Braintrust Eve instrumentation e2e scenario", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "search", + "type": "tool-result" + } + ], + "role": "tool" + } + ], + "output": [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)", + "reasoning": [ + { + "content": "Okay, the user asked me to run the Braintrust Eve instrumentation e2e scenario. As the researcher subagent, my first step was to search the Eve documentation for relevant information. The search tool returned a result with the title \"Eve instrumentation\" and a URL. Now, I need to present this information clearly.\n\nThe instructions say to answer with a single sentence starting with \"Researcher result:\" including the title and URL. I should make sure the link is properly formatted. Let me check the parameters again to confirm the title and URL from the search response. The title is \"Eve instrumentation\" and the URL is \"https://eve.dev/docs/guides/instrumentation\". \n\nI need to structure the response correctly. The sentence should be concise and include both the title and the link. No other information is needed since the user's task is to research before the parent reads it. Alright, that's all. Time to put it together.\n" + } + ], + "role": "assistant" + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 219, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 3269, + "tokens": 3488 + } + } + ], + "metadata": { + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + } + ], + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario" + }, + "output": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)", + "metadata": { + "eve.action_kind": "subagent-call", + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + }, + { + "name": "read", + "type": "tool", + "children": [], + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "output": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "metadata": { + "eve.action_kind": "tool-call", + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + }, + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n", + "type": "reasoning" + }, + { + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-call" + }, + { + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "output": { + "type": "text", + "value": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-result" + } + ], + "role": "tool" + } + ], + "output": [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", + "reasoning": [ + { + "content": "Okay, let me process what's needed here. The user asked to run the Braintrust Eve instrumentation e2e scenario. According to the instructions, I first made the two required tool calls: researcher with the user's message and read with the documentation URL. Now, I have the results from those calls.\n\nThe researcher tool was used to send the message \"Run the Braintrust Eve instrumentation e2e scenario\" to the subagent. The read tool fetched the documentation page, which has a title \"Eve instrumentation\", a URL, and an excerpt mentioning that Eve hooks expose runtime stream events mapped into a nested turn trace.\n\nNow, I need to construct the final answer. The instructions specify that it should start with \"Final answer from read:\", include the researcher result, the title, URL, and excerpt. The researcher result here is the link to the documentation. The read response provided the title, URL, and excerpt. \n\nI should verify that all required elements are included. The researcher result is the link, the title is \"Eve instrumentation\", the URL is the one provided, and the excerpt is the text about runtime context. Putting it all together in a single sentence as specified. No markdown, just plain text. Let me check for any missing parts or formatting issues. Looks good. Time to compose the final answer.\n" + } + ], + "role": "assistant" + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 339, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 4264, + "tokens": 4603 + } + } + ], + "metadata": { + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + }, + { + "name": "eve.turn", + "type": "task", + "children": [ + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n", + "type": "reasoning" + }, + { + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-call" + }, + { + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "output": { + "type": "text", + "value": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "text": "Okay, let me process what's needed here. The user asked to run the Braintrust Eve instrumentation e2e scenario. According to the instructions, I first made the two required tool calls: researcher with the user's message and read with the documentation URL. Now, I have the results from those calls.\n\nThe researcher tool was used to send the message \"Run the Braintrust Eve instrumentation e2e scenario\" to the subagent. The read tool fetched the documentation page, which has a title \"Eve instrumentation\", a URL, and an excerpt mentioning that Eve hooks expose runtime stream events mapped into a nested turn trace.\n\nNow, I need to construct the final answer. The instructions specify that it should start with \"Final answer from read:\", include the researcher result, the title, URL, and excerpt. The researcher result here is the link to the documentation. The read response provided the title, URL, and excerpt. \n\nI should verify that all required elements are included. The researcher result is the link, the title is \"Eve instrumentation\", the URL is the one provided, and the excerpt is the text about runtime context. Putting it all together in a single sentence as specified. No markdown, just plain text. Let me check for any missing parts or formatting issues. Looks good. Time to compose the final answer.\n", + "type": "reasoning" + }, + { + "text": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", + "type": "text" + } + ], + "role": "assistant" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario again", + "role": "user" + } + ], + "output": [ + { + "finish_reason": "tool_calls", + "index": 0, + "message": { + "content": null, + "reasoning": [ + { + "content": "Okay, the user wants me to run the Braintrust Eve instrumentation e2e scenario again. Let me check the previous interaction to make sure I follow the correct steps.\n\nIn the first run, the user said \"Run the Braintrust Eve instrumentation e2e scenario,\" and I called the researcher and read tools. The researcher was given the message \"Run the Braintrust Eve instrumentation e2e scenario,\" and the read tool accessed the specified URL. Then, after getting the response, I provided the final answer combining the researcher result and the read excerpt.\n\nNow the user is asking to do it again. The instructions say that for every user task, I must first make the two tool calls: researcher with the full user message and read with the URL. I need to ensure that I don't reuse any previous messages and that both calls are made in the same response. Then, once both results are available, I answer with the single sentence starting with \"Final answer from read:\" including all the required details.\n\nSo, the next step is to generate the two tool calls again. The user's message is \"Run the Braintrust Eve instrumentation e2e scenario again,\" so the researcher argument should be exactly that. The read tool will use the same URL as before. I need to make sure the JSON is correctly formatted and that both tool calls are in the same response. After that, when the tool responses come back, I'll structure the final answer as instructed.\n" + } + ], + "role": "assistant", + "tool_calls": [ + { + "function": { + "arguments": "{\"message\":\"Run the Braintrust Eve instrumentation e2e scenario again\"}", + "name": "researcher" + }, + "id": "", + "type": "function" + }, + { + "function": { + "arguments": "{\"url\":\"https://eve.dev/docs/guides/instrumentation\"}", + "name": "read" + }, + "id": "", + "type": "function" + } + ] + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 355, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 3895, + "tokens": 4250 + } + }, + { + "name": "researcher", + "type": "tool", + "children": [ + { + "name": "eve.turn", + "type": "task", + "children": [ + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + "role": "user" + } + ], + "output": [ + { + "finish_reason": "tool_calls", + "index": 0, + "message": { + "content": null, + "reasoning": [ + { + "content": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario again. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that. The query should be the exact task name they mentioned: \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that query. Once I get the results, I'll provide the title and URL from the search outcome. Need to make sure I only do one tool call as per instructions.\n" + } + ], + "role": "assistant", + "tool_calls": [ + { + "function": { + "arguments": "{\"query\":\"Braintrust Eve instrumentation e2e scenario\"}", + "name": "search" + }, + "id": "", + "type": "function" + } + ] + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 155, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 3058, + "tokens": 3213 + } + }, + { + "name": "search", + "type": "tool", + "children": [], + "input": { + "query": "Braintrust Eve instrumentation e2e scenario" + }, + "output": { + "query": "Braintrust Eve instrumentation e2e scenario", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "metadata": { + "eve.action_kind": "tool-call", + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + }, + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario again. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that. The query should be the exact task name they mentioned: \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that query. Once I get the results, I'll provide the title and URL from the search outcome. Need to make sure I only do one tool call as per instructions.\n", + "type": "reasoning" + }, + { + "input": { + "query": "Braintrust Eve instrumentation e2e scenario" + }, + "toolCallId": "", + "toolName": "search", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "query": "Braintrust Eve instrumentation e2e scenario", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "search", + "type": "tool-result" + } + ], + "role": "tool" + } + ], + "output": [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation", + "reasoning": [ + { + "content": "Okay, the user asked me to run the Braintrust Eve instrumentation end-to-end scenario again. As the researcher subagent, my first step is to look up the relevant documentation. I used the search tool with the query \"Braintrust Eve instrumentation e2e scenario\" and found a result titled \"Eve instrumentation\" with the URL provided. Now, I need to confirm that this documentation covers the e2e scenario. The title seems relevant, but I should check if the content includes steps or guides for running such a scenario. Since I can't browse the link, I'll rely on the title and the assumption that the documentation is accurate. The next step is to inform the user of the research result, citing the title and URL. I should keep the response concise and directly address the task. No further tool calls are needed here since the search result provides the necessary information. I'll format the answer as instructed, starting with \"Researcher result:\" followed by the title and URL.\n" + } + ], + "role": "assistant" + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 221, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 3256, + "tokens": 3477 + } + } + ], + "metadata": { + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + } + ], + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario again" + }, + "output": "Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation", + "metadata": { + "eve.action_kind": "subagent-call", + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + }, + { + "name": "read", + "type": "tool", + "children": [], + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "output": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "metadata": { + "eve.action_kind": "tool-call", + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + }, + { + "name": "eve.step", + "type": "llm", + "children": [], + "input": [ + { + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n", + "type": "reasoning" + }, + { + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-call" + }, + { + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "output": { + "type": "text", + "value": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "text": "Okay, let me process what's needed here. The user asked to run the Braintrust Eve instrumentation e2e scenario. According to the instructions, I first made the two required tool calls: researcher with the user's message and read with the documentation URL. Now, I have the results from those calls.\n\nThe researcher tool was used to send the message \"Run the Braintrust Eve instrumentation e2e scenario\" to the subagent. The read tool fetched the documentation page, which has a title \"Eve instrumentation\", a URL, and an excerpt mentioning that Eve hooks expose runtime stream events mapped into a nested turn trace.\n\nNow, I need to construct the final answer. The instructions specify that it should start with \"Final answer from read:\", include the researcher result, the title, URL, and excerpt. The researcher result here is the link to the documentation. The read response provided the title, URL, and excerpt. \n\nI should verify that all required elements are included. The researcher result is the link, the title is \"Eve instrumentation\", the URL is the one provided, and the excerpt is the text about runtime context. Putting it all together in a single sentence as specified. No markdown, just plain text. Let me check for any missing parts or formatting issues. Looks good. Time to compose the final answer.\n", + "type": "reasoning" + }, + { + "text": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", + "type": "text" + } + ], + "role": "assistant" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario again", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation e2e scenario again. Let me check the previous interaction to make sure I follow the correct steps.\n\nIn the first run, the user said \"Run the Braintrust Eve instrumentation e2e scenario,\" and I called the researcher and read tools. The researcher was given the message \"Run the Braintrust Eve instrumentation e2e scenario,\" and the read tool accessed the specified URL. Then, after getting the response, I provided the final answer combining the researcher result and the read excerpt.\n\nNow the user is asking to do it again. The instructions say that for every user task, I must first make the two tool calls: researcher with the full user message and read with the URL. I need to ensure that I don't reuse any previous messages and that both calls are made in the same response. Then, once both results are available, I answer with the single sentence starting with \"Final answer from read:\" including all the required details.\n\nSo, the next step is to generate the two tool calls again. The user's message is \"Run the Braintrust Eve instrumentation e2e scenario again,\" so the researcher argument should be exactly that. The read tool will use the same URL as before. I need to make sure the JSON is correctly formatted and that both tool calls are in the same response. After that, when the tool responses come back, I'll structure the final answer as instructed.\n", + "type": "reasoning" + }, + { + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario again" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-call" + }, + { + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "output": { + "type": "text", + "value": "Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-result" + } + ], + "role": "tool" + } + ], + "output": [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", + "reasoning": [ + { + "content": "Okay, the user asked to run the Braintrust Eve instrumentation e2e scenario again. I need to follow the same steps as before. First, call the researcher with the exact user message, which is \"Run the Braintrust Eve instrumentation e2e scenario again\". Then, use the read tool to fetch the documentation from the same URL. Once both responses are received, compile the final answer with the researcher result, title, URL, and excerpt. Make sure the final answer is a single sentence starting with \"Final answer from read:\" and includes all the required details. Check that the researcher result is correctly formatted as a link and that the excerpt matches the provided data. Alright, everything seems in order.\n" + } + ], + "role": "assistant" + } + } + ], + "metadata": { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + }, + "metrics": { + "completion_tokens": 213, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 4329, + "tokens": 4542 + } + } + ], + "metadata": { + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + } + ] +} diff --git a/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest-pinned.span-tree.txt b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest-pinned.span-tree.txt new file mode 100644 index 000000000..dc430f2c0 --- /dev/null +++ b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest-pinned.span-tree.txt @@ -0,0 +1,821 @@ +span_tree: +├── eve.turn [task] +│ metadata: { +│ "eve.session_id": "", +│ "scenario": "eve-instrumentation", +│ "testRunId": "" +│ } +│ ├── eve.step [llm] +│ │ input: [ +│ │ { +│ │ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", +│ │ "role": "system" +│ │ }, +│ │ { +│ │ "content": "Run the Braintrust Eve instrumentation e2e scenario", +│ │ "role": "user" +│ │ } +│ │ ] +│ │ output: [ +│ │ { +│ │ "finish_reason": "tool_calls", +│ │ "index": 0, +│ │ "message": { +│ │ "content": null, +│ │ "reasoning": [ +│ │ { +│ │ "content": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n" +│ │ } +│ │ ], +│ │ "role": "assistant", +│ │ "tool_calls": [ +│ │ { +│ │ "function": { +│ │ "arguments": "{\"message\":\"Run the Braintrust Eve instrumentation e2e scenario\"}", +│ │ "name": "researcher" +│ │ }, +│ │ "id": "", +│ │ "type": "function" +│ │ }, +│ │ { +│ │ "function": { +│ │ "arguments": "{\"url\":\"https://eve.dev/docs/guides/instrumentation\"}", +│ │ "name": "read" +│ │ }, +│ │ "id": "", +│ │ "type": "function" +│ │ } +│ │ ] +│ │ } +│ │ } +│ │ ] +│ │ metadata: { +│ │ "eve.session_id": "", +│ │ "model": "qwen/qwen3-30b-a3b", +│ │ "provider": "openrouter", +│ │ "scenario": "eve-instrumentation", +│ │ "testRunId": "" +│ │ } +│ │ metrics: { +│ │ "completion_tokens": 514, +│ │ "prompt_cache_creation_tokens": 0, +│ │ "prompt_cached_tokens": 0, +│ │ "prompt_tokens": 3670, +│ │ "tokens": 4184 +│ │ } +│ ├── researcher [tool] +│ │ input: { +│ │ "message": "Run the Braintrust Eve instrumentation e2e scenario" +│ │ } +│ │ output: "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)" +│ │ metadata: { +│ │ "eve.action_kind": "subagent-call", +│ │ "eve.session_id": "", +│ │ "scenario": "eve-instrumentation", +│ │ "testRunId": "" +│ │ } +│ │ └── eve.turn [task] +│ │ metadata: { +│ │ "eve.session_id": "", +│ │ "scenario": "eve-instrumentation", +│ │ "testRunId": "" +│ │ } +│ │ ├── eve.step [llm] +│ │ │ input: [ +│ │ │ { +│ │ │ "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", +│ │ │ "role": "system" +│ │ │ }, +│ │ │ { +│ │ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", +│ │ │ "role": "user" +│ │ │ } +│ │ │ ] +│ │ │ output: [ +│ │ │ { +│ │ │ "finish_reason": "tool_calls", +│ │ │ "index": 0, +│ │ │ "message": { +│ │ │ "content": null, +│ │ │ "reasoning": [ +│ │ │ { +│ │ │ "content": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that.\n\nThe task is to search for \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that exact query. Once I get the results, I need to provide a concise answer with the title and URL from the search. Let me make sure I follow the instructions: one tool call, then a single sentence with the result. Alright, let's execute the search.\n" +│ │ │ } +│ │ │ ], +│ │ │ "role": "assistant", +│ │ │ "tool_calls": [ +│ │ │ { +│ │ │ "function": { +│ │ │ "arguments": "{\"query\":\"Braintrust Eve instrumentation e2e scenario\"}", +│ │ │ "name": "search" +│ │ │ }, +│ │ │ "id": "", +│ │ │ "type": "function" +│ │ │ } +│ │ │ ] +│ │ │ } +│ │ │ } +│ │ │ ] +│ │ │ metadata: { +│ │ │ "eve.session_id": "", +│ │ │ "model": "qwen/qwen3-30b-a3b", +│ │ │ "provider": "openrouter", +│ │ │ "scenario": "eve-instrumentation", +│ │ │ "testRunId": "" +│ │ │ } +│ │ │ metrics: { +│ │ │ "completion_tokens": 169, +│ │ │ "prompt_cache_creation_tokens": 0, +│ │ │ "prompt_cached_tokens": 0, +│ │ │ "prompt_tokens": 3057, +│ │ │ "tokens": 3226 +│ │ │ } +│ │ ├── search [tool] +│ │ │ input: { +│ │ │ "query": "Braintrust Eve instrumentation e2e scenario" +│ │ │ } +│ │ │ output: { +│ │ │ "query": "Braintrust Eve instrumentation e2e scenario", +│ │ │ "title": "Eve instrumentation", +│ │ │ "url": "https://eve.dev/docs/guides/instrumentation" +│ │ │ } +│ │ │ metadata: { +│ │ │ "eve.action_kind": "tool-call", +│ │ │ "eve.session_id": "", +│ │ │ "scenario": "eve-instrumentation", +│ │ │ "testRunId": "" +│ │ │ } +│ │ └── eve.step [llm] +│ │ input: [ +│ │ { +│ │ "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", +│ │ "role": "system" +│ │ }, +│ │ { +│ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", +│ │ "role": "user" +│ │ }, +│ │ { +│ │ "content": [ +│ │ { +│ │ "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that.\n\nThe task is to search for \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that exact query. Once I get the results, I need to provide a concise answer with the title and URL from the search. Let me make sure I follow the instructions: one tool call, then a single sentence with the result. Alright, let's execute the search.\n", +│ │ "type": "reasoning" +│ │ }, +│ │ { +│ │ "input": { +│ │ "query": "Braintrust Eve instrumentation e2e scenario" +│ │ }, +│ │ "toolCallId": "", +│ │ "toolName": "search", +│ │ "type": "tool-call" +│ │ } +│ │ ], +│ │ "role": "assistant" +│ │ }, +│ │ { +│ │ "content": [ +│ │ { +│ │ "output": { +│ │ "type": "json", +│ │ "value": { +│ │ "query": "Braintrust Eve instrumentation e2e scenario", +│ │ "title": "Eve instrumentation", +│ │ "url": "https://eve.dev/docs/guides/instrumentation" +│ │ } +│ │ }, +│ │ "toolCallId": "", +│ │ "toolName": "search", +│ │ "type": "tool-result" +│ │ } +│ │ ], +│ │ "role": "tool" +│ │ } +│ │ ] +│ │ output: [ +│ │ { +│ │ "finish_reason": "stop", +│ │ "index": 0, +│ │ "message": { +│ │ "content": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)", +│ │ "reasoning": [ +│ │ { +│ │ "content": "Okay, the user asked me to run the Braintrust Eve instrumentation e2e scenario. As the researcher subagent, my first step was to search the Eve documentation for relevant information. The search tool returned a result with the title \"Eve instrumentation\" and a URL. Now, I need to present this information clearly.\n\nThe instructions say to answer with a single sentence starting with \"Researcher result:\" including the title and URL. I should make sure the link is properly formatted. Let me check the parameters again to confirm the title and URL from the search response. The title is \"Eve instrumentation\" and the URL is \"https://eve.dev/docs/guides/instrumentation\". \n\nI need to structure the response correctly. The sentence should be concise and include both the title and the link. No other information is needed since the user's task is to research before the parent reads it. Alright, that's all. Time to put it together.\n" +│ │ } +│ │ ], +│ │ "role": "assistant" +│ │ } +│ │ } +│ │ ] +│ │ metadata: { +│ │ "eve.session_id": "", +│ │ "model": "qwen/qwen3-30b-a3b", +│ │ "provider": "openrouter", +│ │ "scenario": "eve-instrumentation", +│ │ "testRunId": "" +│ │ } +│ │ metrics: { +│ │ "completion_tokens": 219, +│ │ "prompt_cache_creation_tokens": 0, +│ │ "prompt_cached_tokens": 0, +│ │ "prompt_tokens": 3269, +│ │ "tokens": 3488 +│ │ } +│ ├── read [tool] +│ │ input: { +│ │ "url": "https://eve.dev/docs/guides/instrumentation" +│ │ } +│ │ output: { +│ │ "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", +│ │ "section": "Runtime context", +│ │ "title": "Eve instrumentation", +│ │ "url": "https://eve.dev/docs/guides/instrumentation" +│ │ } +│ │ metadata: { +│ │ "eve.action_kind": "tool-call", +│ │ "eve.session_id": "", +│ │ "scenario": "eve-instrumentation", +│ │ "testRunId": "" +│ │ } +│ └── eve.step [llm] +│ input: [ +│ { +│ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", +│ "role": "system" +│ }, +│ { +│ "content": "Run the Braintrust Eve instrumentation e2e scenario", +│ "role": "user" +│ }, +│ { +│ "content": [ +│ { +│ "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n", +│ "type": "reasoning" +│ }, +│ { +│ "input": { +│ "message": "Run the Braintrust Eve instrumentation e2e scenario" +│ }, +│ "toolCallId": "", +│ "toolName": "researcher", +│ "type": "tool-call" +│ }, +│ { +│ "input": { +│ "url": "https://eve.dev/docs/guides/instrumentation" +│ }, +│ "toolCallId": "", +│ "toolName": "read", +│ "type": "tool-call" +│ } +│ ], +│ "role": "assistant" +│ }, +│ { +│ "content": [ +│ { +│ "output": { +│ "type": "json", +│ "value": { +│ "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", +│ "section": "Runtime context", +│ "title": "Eve instrumentation", +│ "url": "https://eve.dev/docs/guides/instrumentation" +│ } +│ }, +│ "toolCallId": "", +│ "toolName": "read", +│ "type": "tool-result" +│ } +│ ], +│ "role": "tool" +│ }, +│ { +│ "content": [ +│ { +│ "output": { +│ "type": "text", +│ "value": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)" +│ }, +│ "toolCallId": "", +│ "toolName": "researcher", +│ "type": "tool-result" +│ } +│ ], +│ "role": "tool" +│ } +│ ] +│ output: [ +│ { +│ "finish_reason": "stop", +│ "index": 0, +│ "message": { +│ "content": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", +│ "reasoning": [ +│ { +│ "content": "Okay, let me process what's needed here. The user asked to run the Braintrust Eve instrumentation e2e scenario. According to the instructions, I first made the two required tool calls: researcher with the user's message and read with the documentation URL. Now, I have the results from those calls.\n\nThe researcher tool was used to send the message \"Run the Braintrust Eve instrumentation e2e scenario\" to the subagent. The read tool fetched the documentation page, which has a title \"Eve instrumentation\", a URL, and an excerpt mentioning that Eve hooks expose runtime stream events mapped into a nested turn trace.\n\nNow, I need to construct the final answer. The instructions specify that it should start with \"Final answer from read:\", include the researcher result, the title, URL, and excerpt. The researcher result here is the link to the documentation. The read response provided the title, URL, and excerpt. \n\nI should verify that all required elements are included. The researcher result is the link, the title is \"Eve instrumentation\", the URL is the one provided, and the excerpt is the text about runtime context. Putting it all together in a single sentence as specified. No markdown, just plain text. Let me check for any missing parts or formatting issues. Looks good. Time to compose the final answer.\n" +│ } +│ ], +│ "role": "assistant" +│ } +│ } +│ ] +│ metadata: { +│ "eve.session_id": "", +│ "model": "qwen/qwen3-30b-a3b", +│ "provider": "openrouter", +│ "scenario": "eve-instrumentation", +│ "testRunId": "" +│ } +│ metrics: { +│ "completion_tokens": 339, +│ "prompt_cache_creation_tokens": 0, +│ "prompt_cached_tokens": 0, +│ "prompt_tokens": 4264, +│ "tokens": 4603 +│ } +└── eve.turn [task] + metadata: { + "eve.session_id": "", + "scenario": "eve-instrumentation", + "testRunId": "" + } + ├── eve.step [llm] + │ input: [ + │ { + │ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + │ "role": "system" + │ }, + │ { + │ "content": "Run the Braintrust Eve instrumentation e2e scenario", + │ "role": "user" + │ }, + │ { + │ "content": [ + │ { + │ "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n", + │ "type": "reasoning" + │ }, + │ { + │ "input": { + │ "message": "Run the Braintrust Eve instrumentation e2e scenario" + │ }, + │ "toolCallId": "", + │ "toolName": "researcher", + │ "type": "tool-call" + │ }, + │ { + │ "input": { + │ "url": "https://eve.dev/docs/guides/instrumentation" + │ }, + │ "toolCallId": "", + │ "toolName": "read", + │ "type": "tool-call" + │ } + │ ], + │ "role": "assistant" + │ }, + │ { + │ "content": [ + │ { + │ "output": { + │ "type": "json", + │ "value": { + │ "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + │ "section": "Runtime context", + │ "title": "Eve instrumentation", + │ "url": "https://eve.dev/docs/guides/instrumentation" + │ } + │ }, + │ "toolCallId": "", + │ "toolName": "read", + │ "type": "tool-result" + │ } + │ ], + │ "role": "tool" + │ }, + │ { + │ "content": [ + │ { + │ "output": { + │ "type": "text", + │ "value": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)" + │ }, + │ "toolCallId": "", + │ "toolName": "researcher", + │ "type": "tool-result" + │ } + │ ], + │ "role": "tool" + │ }, + │ { + │ "content": [ + │ { + │ "text": "Okay, let me process what's needed here. The user asked to run the Braintrust Eve instrumentation e2e scenario. According to the instructions, I first made the two required tool calls: researcher with the user's message and read with the documentation URL. Now, I have the results from those calls.\n\nThe researcher tool was used to send the message \"Run the Braintrust Eve instrumentation e2e scenario\" to the subagent. The read tool fetched the documentation page, which has a title \"Eve instrumentation\", a URL, and an excerpt mentioning that Eve hooks expose runtime stream events mapped into a nested turn trace.\n\nNow, I need to construct the final answer. The instructions specify that it should start with \"Final answer from read:\", include the researcher result, the title, URL, and excerpt. The researcher result here is the link to the documentation. The read response provided the title, URL, and excerpt. \n\nI should verify that all required elements are included. The researcher result is the link, the title is \"Eve instrumentation\", the URL is the one provided, and the excerpt is the text about runtime context. Putting it all together in a single sentence as specified. No markdown, just plain text. Let me check for any missing parts or formatting issues. Looks good. Time to compose the final answer.\n", + │ "type": "reasoning" + │ }, + │ { + │ "text": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", + │ "type": "text" + │ } + │ ], + │ "role": "assistant" + │ }, + │ { + │ "content": "Run the Braintrust Eve instrumentation e2e scenario again", + │ "role": "user" + │ } + │ ] + │ output: [ + │ { + │ "finish_reason": "tool_calls", + │ "index": 0, + │ "message": { + │ "content": null, + │ "reasoning": [ + │ { + │ "content": "Okay, the user wants me to run the Braintrust Eve instrumentation e2e scenario again. Let me check the previous interaction to make sure I follow the correct steps.\n\nIn the first run, the user said \"Run the Braintrust Eve instrumentation e2e scenario,\" and I called the researcher and read tools. The researcher was given the message \"Run the Braintrust Eve instrumentation e2e scenario,\" and the read tool accessed the specified URL. Then, after getting the response, I provided the final answer combining the researcher result and the read excerpt.\n\nNow the user is asking to do it again. The instructions say that for every user task, I must first make the two tool calls: researcher with the full user message and read with the URL. I need to ensure that I don't reuse any previous messages and that both calls are made in the same response. Then, once both results are available, I answer with the single sentence starting with \"Final answer from read:\" including all the required details.\n\nSo, the next step is to generate the two tool calls again. The user's message is \"Run the Braintrust Eve instrumentation e2e scenario again,\" so the researcher argument should be exactly that. The read tool will use the same URL as before. I need to make sure the JSON is correctly formatted and that both tool calls are in the same response. After that, when the tool responses come back, I'll structure the final answer as instructed.\n" + │ } + │ ], + │ "role": "assistant", + │ "tool_calls": [ + │ { + │ "function": { + │ "arguments": "{\"message\":\"Run the Braintrust Eve instrumentation e2e scenario again\"}", + │ "name": "researcher" + │ }, + │ "id": "", + │ "type": "function" + │ }, + │ { + │ "function": { + │ "arguments": "{\"url\":\"https://eve.dev/docs/guides/instrumentation\"}", + │ "name": "read" + │ }, + │ "id": "", + │ "type": "function" + │ } + │ ] + │ } + │ } + │ ] + │ metadata: { + │ "eve.session_id": "", + │ "model": "qwen/qwen3-30b-a3b", + │ "provider": "openrouter", + │ "scenario": "eve-instrumentation", + │ "testRunId": "" + │ } + │ metrics: { + │ "completion_tokens": 355, + │ "prompt_cache_creation_tokens": 0, + │ "prompt_cached_tokens": 0, + │ "prompt_tokens": 3895, + │ "tokens": 4250 + │ } + ├── researcher [tool] + │ input: { + │ "message": "Run the Braintrust Eve instrumentation e2e scenario again" + │ } + │ output: "Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation" + │ metadata: { + │ "eve.action_kind": "subagent-call", + │ "eve.session_id": "", + │ "scenario": "eve-instrumentation", + │ "testRunId": "" + │ } + │ └── eve.turn [task] + │ metadata: { + │ "eve.session_id": "", + │ "scenario": "eve-instrumentation", + │ "testRunId": "" + │ } + │ ├── eve.step [llm] + │ │ input: [ + │ │ { + │ │ "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + │ │ "role": "system" + │ │ }, + │ │ { + │ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + │ │ "role": "user" + │ │ } + │ │ ] + │ │ output: [ + │ │ { + │ │ "finish_reason": "tool_calls", + │ │ "index": 0, + │ │ "message": { + │ │ "content": null, + │ │ "reasoning": [ + │ │ { + │ │ "content": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario again. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that. The query should be the exact task name they mentioned: \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that query. Once I get the results, I'll provide the title and URL from the search outcome. Need to make sure I only do one tool call as per instructions.\n" + │ │ } + │ │ ], + │ │ "role": "assistant", + │ │ "tool_calls": [ + │ │ { + │ │ "function": { + │ │ "arguments": "{\"query\":\"Braintrust Eve instrumentation e2e scenario\"}", + │ │ "name": "search" + │ │ }, + │ │ "id": "", + │ │ "type": "function" + │ │ } + │ │ ] + │ │ } + │ │ } + │ │ ] + │ │ metadata: { + │ │ "eve.session_id": "", + │ │ "model": "qwen/qwen3-30b-a3b", + │ │ "provider": "openrouter", + │ │ "scenario": "eve-instrumentation", + │ │ "testRunId": "" + │ │ } + │ │ metrics: { + │ │ "completion_tokens": 155, + │ │ "prompt_cache_creation_tokens": 0, + │ │ "prompt_cached_tokens": 0, + │ │ "prompt_tokens": 3058, + │ │ "tokens": 3213 + │ │ } + │ ├── search [tool] + │ │ input: { + │ │ "query": "Braintrust Eve instrumentation e2e scenario" + │ │ } + │ │ output: { + │ │ "query": "Braintrust Eve instrumentation e2e scenario", + │ │ "title": "Eve instrumentation", + │ │ "url": "https://eve.dev/docs/guides/instrumentation" + │ │ } + │ │ metadata: { + │ │ "eve.action_kind": "tool-call", + │ │ "eve.session_id": "", + │ │ "scenario": "eve-instrumentation", + │ │ "testRunId": "" + │ │ } + │ └── eve.step [llm] + │ input: [ + │ { + │ "content": "Instructions (instructions)\nYou are the deterministic researcher subagent for Braintrust Eve instrumentation tests.\n\nFor every task:\n\n1. Call the search tool exactly once with the full task as the query.\n2. After the search result is available, answer with a single sentence that starts\n with \"Researcher result:\" and includes the search title and URL.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + │ "role": "system" + │ }, + │ { + │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + │ "role": "user" + │ }, + │ { + │ "content": [ + │ { + │ "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario again. Since I'm the researcher subagent, my job is to look up the documentation first. Let me check the available tools. There's a 'search' function that queries the Eve documentation. I should use that. The query should be the exact task name they mentioned: \"Braintrust Eve instrumentation e2e scenario\". I'll call the search tool with that query. Once I get the results, I'll provide the title and URL from the search outcome. Need to make sure I only do one tool call as per instructions.\n", + │ "type": "reasoning" + │ }, + │ { + │ "input": { + │ "query": "Braintrust Eve instrumentation e2e scenario" + │ }, + │ "toolCallId": "", + │ "toolName": "search", + │ "type": "tool-call" + │ } + │ ], + │ "role": "assistant" + │ }, + │ { + │ "content": [ + │ { + │ "output": { + │ "type": "json", + │ "value": { + │ "query": "Braintrust Eve instrumentation e2e scenario", + │ "title": "Eve instrumentation", + │ "url": "https://eve.dev/docs/guides/instrumentation" + │ } + │ }, + │ "toolCallId": "", + │ "toolName": "search", + │ "type": "tool-result" + │ } + │ ], + │ "role": "tool" + │ } + │ ] + │ output: [ + │ { + │ "finish_reason": "stop", + │ "index": 0, + │ "message": { + │ "content": "Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation", + │ "reasoning": [ + │ { + │ "content": "Okay, the user asked me to run the Braintrust Eve instrumentation end-to-end scenario again. As the researcher subagent, my first step is to look up the relevant documentation. I used the search tool with the query \"Braintrust Eve instrumentation e2e scenario\" and found a result titled \"Eve instrumentation\" with the URL provided. Now, I need to confirm that this documentation covers the e2e scenario. The title seems relevant, but I should check if the content includes steps or guides for running such a scenario. Since I can't browse the link, I'll rely on the title and the assumption that the documentation is accurate. The next step is to inform the user of the research result, citing the title and URL. I should keep the response concise and directly address the task. No further tool calls are needed here since the search result provides the necessary information. I'll format the answer as instructed, starting with \"Researcher result:\" followed by the title and URL.\n" + │ } + │ ], + │ "role": "assistant" + │ } + │ } + │ ] + │ metadata: { + │ "eve.session_id": "", + │ "model": "qwen/qwen3-30b-a3b", + │ "provider": "openrouter", + │ "scenario": "eve-instrumentation", + │ "testRunId": "" + │ } + │ metrics: { + │ "completion_tokens": 221, + │ "prompt_cache_creation_tokens": 0, + │ "prompt_cached_tokens": 0, + │ "prompt_tokens": 3256, + │ "tokens": 3477 + │ } + ├── read [tool] + │ input: { + │ "url": "https://eve.dev/docs/guides/instrumentation" + │ } + │ output: { + │ "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + │ "section": "Runtime context", + │ "title": "Eve instrumentation", + │ "url": "https://eve.dev/docs/guides/instrumentation" + │ } + │ metadata: { + │ "eve.action_kind": "tool-call", + │ "eve.session_id": "", + │ "scenario": "eve-instrumentation", + │ "testRunId": "" + │ } + └── eve.step [llm] + input: [ + { + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "role": "system" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation end-to-end scenario. Let me start by understanding what that entails. The user mentioned \"e2e scenario,\" which typically stands for end-to-end testing. So, they probably want to simulate or execute a full workflow within the Eve instrumentation setup.\n\nFirst, I need to check the available tools. The user provided a list of functions, including researcher, read, and others. The instructions say that for every user task, I must first make two tool calls: researcher with the user's message and read with the URL to the instrumentation guide. Then, after getting both results, I should generate a final answer combining the researcher's output, the read title, URL, and excerpt.\n\nSo, the first step is to call the researcher function with the user's message. The message here is \"Run the Braintrust Eve instrumentation e2e scenario.\" I need to pass this exactly as is. Then, I need to call the read function with the URL https://eve.dev/docs/guides/instrumentation. \n\nWait, the user might be expecting me to follow the steps outlined in the documentation. The read function will fetch the content from that URL, which is the guide. The researcher function might be used to query the documentation before reading it. But according to the instructions, I must emit both calls simultaneously. \n\nOnce both responses are available, I need to compile the final answer. The final answer should start with \"Final answer from read:\" and include the researcher result, the title of the read content, the URL, and an excerpt from the read. \n\nI should make sure not to include any other text besides the two tool calls in the initial response. No explanations, just the JSON objects within the tool_call tags. Then, after receiving the results, structure the final answer properly. \n\nI need to verify that I'm not missing any steps. The user's instruction is straightforward, but I have to follow the exact process: two tool calls first, then the final answer. No deviations. Also, ensure that the researcher's message is the exact user input, and the read URL is correct. \n\nI think that's all. Let me structure the tool calls correctly.\n", + "type": "reasoning" + }, + { + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-call" + }, + { + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "output": { + "type": "text", + "value": "Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "text": "Okay, let me process what's needed here. The user asked to run the Braintrust Eve instrumentation e2e scenario. According to the instructions, I first made the two required tool calls: researcher with the user's message and read with the documentation URL. Now, I have the results from those calls.\n\nThe researcher tool was used to send the message \"Run the Braintrust Eve instrumentation e2e scenario\" to the subagent. The read tool fetched the documentation page, which has a title \"Eve instrumentation\", a URL, and an excerpt mentioning that Eve hooks expose runtime stream events mapped into a nested turn trace.\n\nNow, I need to construct the final answer. The instructions specify that it should start with \"Final answer from read:\", include the researcher result, the title, URL, and excerpt. The researcher result here is the link to the documentation. The read response provided the title, URL, and excerpt. \n\nI should verify that all required elements are included. The researcher result is the link, the title is \"Eve instrumentation\", the URL is the one provided, and the excerpt is the text about runtime context. Putting it all together in a single sentence as specified. No markdown, just plain text. Let me check for any missing parts or formatting issues. Looks good. Time to compose the final answer.\n", + "type": "reasoning" + }, + { + "text": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", + "type": "text" + } + ], + "role": "assistant" + }, + { + "content": "Run the Braintrust Eve instrumentation e2e scenario again", + "role": "user" + }, + { + "content": [ + { + "text": "Okay, the user wants me to run the Braintrust Eve instrumentation e2e scenario again. Let me check the previous interaction to make sure I follow the correct steps.\n\nIn the first run, the user said \"Run the Braintrust Eve instrumentation e2e scenario,\" and I called the researcher and read tools. The researcher was given the message \"Run the Braintrust Eve instrumentation e2e scenario,\" and the read tool accessed the specified URL. Then, after getting the response, I provided the final answer combining the researcher result and the read excerpt.\n\nNow the user is asking to do it again. The instructions say that for every user task, I must first make the two tool calls: researcher with the full user message and read with the URL. I need to ensure that I don't reuse any previous messages and that both calls are made in the same response. Then, once both results are available, I answer with the single sentence starting with \"Final answer from read:\" including all the required details.\n\nSo, the next step is to generate the two tool calls again. The user's message is \"Run the Braintrust Eve instrumentation e2e scenario again,\" so the researcher argument should be exactly that. The read tool will use the same URL as before. I need to make sure the JSON is correctly formatted and that both tool calls are in the same response. After that, when the tool responses come back, I'll structure the final answer as instructed.\n", + "type": "reasoning" + }, + { + "input": { + "message": "Run the Braintrust Eve instrumentation e2e scenario again" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-call" + }, + { + "input": { + "url": "https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-call" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "output": { + "type": "json", + "value": { + "excerpt": "Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.", + "section": "Runtime context", + "title": "Eve instrumentation", + "url": "https://eve.dev/docs/guides/instrumentation" + } + }, + "toolCallId": "", + "toolName": "read", + "type": "tool-result" + } + ], + "role": "tool" + }, + { + "content": [ + { + "output": { + "type": "text", + "value": "Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation" + }, + "toolCallId": "", + "toolName": "researcher", + "type": "tool-result" + } + ], + "role": "tool" + } + ] + output: [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "Final answer from read: Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation). Title: Eve instrumentation. URL: https://eve.dev/docs/guides/instrumentation. Excerpt: \"Eve hooks expose runtime stream events that Braintrust maps into a nested turn trace.\"", + "reasoning": [ + { + "content": "Okay, the user asked to run the Braintrust Eve instrumentation e2e scenario again. I need to follow the same steps as before. First, call the researcher with the exact user message, which is \"Run the Braintrust Eve instrumentation e2e scenario again\". Then, use the read tool to fetch the documentation from the same URL. Once both responses are received, compile the final answer with the researcher result, title, URL, and excerpt. Make sure the final answer is a single sentence starting with \"Final answer from read:\" and includes all the required details. Check that the researcher result is correctly formatted as a link and that the excerpt matches the provided data. Alright, everything seems in order.\n" + } + ], + "role": "assistant" + } + } + ] + metadata: { + "eve.session_id": "", + "model": "qwen/qwen3-30b-a3b", + "provider": "openrouter", + "scenario": "eve-instrumentation", + "testRunId": "" + } + metrics: { + "completion_tokens": 213, + "prompt_cache_creation_tokens": 0, + "prompt_cached_tokens": 0, + "prompt_tokens": 4329, + "tokens": 4542 + } diff --git a/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.json b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.json index 6be45b535..8c67fbbd8 100644 --- a/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.json +++ b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.json @@ -10,7 +10,7 @@ "children": [], "input": [ { - "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", "role": "system" }, { @@ -84,7 +84,7 @@ "role": "system" }, { - "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", "role": "user" } ], @@ -157,7 +157,7 @@ "role": "system" }, { - "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", "role": "user" }, { @@ -271,7 +271,7 @@ "children": [], "input": [ { - "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", "role": "system" }, { @@ -335,6 +335,10 @@ } ], "role": "tool" + }, + { + "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n", + "role": "user" } ], "output": [ @@ -384,7 +388,7 @@ "children": [], "input": [ { - "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", "role": "system" }, { @@ -449,6 +453,10 @@ ], "role": "tool" }, + { + "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n", + "role": "user" + }, { "content": [ { @@ -533,7 +541,7 @@ "role": "system" }, { - "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", "role": "user" } ], @@ -606,7 +614,7 @@ "role": "system" }, { - "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", "role": "user" }, { @@ -720,7 +728,7 @@ "children": [], "input": [ { - "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", "role": "system" }, { @@ -785,6 +793,10 @@ ], "role": "tool" }, + { + "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n", + "role": "user" + }, { "content": [ { @@ -859,6 +871,10 @@ } ], "role": "tool" + }, + { + "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n\" name=\"researcher\">Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation\n", + "role": "user" } ], "output": [ diff --git a/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.txt b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.txt index dc430f2c0..e73c6a9d5 100644 --- a/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.txt +++ b/e2e/scenarios/eve-instrumentation/__snapshots__/eve-v0-latest.span-tree.txt @@ -8,7 +8,7 @@ span_tree: │ ├── eve.step [llm] │ │ input: [ │ │ { -│ │ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", +│ │ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", │ │ "role": "system" │ │ }, │ │ { @@ -87,7 +87,7 @@ span_tree: │ │ │ "role": "system" │ │ │ }, │ │ │ { -│ │ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", +│ │ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", │ │ │ "role": "user" │ │ │ } │ │ │ ] @@ -152,7 +152,7 @@ span_tree: │ │ "role": "system" │ │ }, │ │ { -│ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", +│ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario", │ │ "role": "user" │ │ }, │ │ { @@ -239,7 +239,7 @@ span_tree: │ └── eve.step [llm] │ input: [ │ { -│ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", +│ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", │ "role": "system" │ }, │ { @@ -303,6 +303,10 @@ span_tree: │ } │ ], │ "role": "tool" +│ }, +│ { +│ "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n", +│ "role": "user" │ } │ ] │ output: [ @@ -343,7 +347,7 @@ span_tree: ├── eve.step [llm] │ input: [ │ { - │ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + │ "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", │ "role": "system" │ }, │ { @@ -409,6 +413,10 @@ span_tree: │ "role": "tool" │ }, │ { + │ "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n", + │ "role": "user" + │ }, + │ { │ "content": [ │ { │ "text": "Okay, let me process what's needed here. The user asked to run the Braintrust Eve instrumentation e2e scenario. According to the instructions, I first made the two required tool calls: researcher with the user's message and read with the documentation URL. Now, I have the results from those calls.\n\nThe researcher tool was used to send the message \"Run the Braintrust Eve instrumentation e2e scenario\" to the subagent. The read tool fetched the documentation page, which has a title \"Eve instrumentation\", a URL, and an excerpt mentioning that Eve hooks expose runtime stream events mapped into a nested turn trace.\n\nNow, I need to construct the final answer. The instructions specify that it should start with \"Final answer from read:\", include the researcher result, the title, URL, and excerpt. The researcher result here is the link to the documentation. The read response provided the title, URL, and excerpt. \n\nI should verify that all required elements are included. The researcher result is the link, the title is \"Eve instrumentation\", the URL is the one provided, and the excerpt is the text about runtime context. Putting it all together in a single sentence as specified. No markdown, just plain text. Let me check for any missing parts or formatting issues. Looks good. Time to compose the final answer.\n", @@ -497,7 +505,7 @@ span_tree: │ │ "role": "system" │ │ }, │ │ { - │ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + │ │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", │ │ "role": "user" │ │ } │ │ ] @@ -562,7 +570,7 @@ span_tree: │ "role": "system" │ }, │ { - │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the final result directly.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", + │ "content": "You are the subagent \"researcher\".\nDescription: Research the Eve instrumentation documentation before the parent reads it.\n\nThe caller delegated the following task to you. Complete it and return the result directly. The caller may send follow-up messages after you answer.\n\nCaller message:\nRun the Braintrust Eve instrumentation e2e scenario again", │ "role": "user" │ }, │ { @@ -649,7 +657,7 @@ span_tree: └── eve.step [llm] input: [ { - "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.", + "content": "Instructions (instructions)\nYou are a deterministic fixture agent for Braintrust Eve instrumentation tests.\n\nFor every user task, follow these steps exactly:\n\n1. Your first response MUST contain exactly these two tool calls and no text:\n - researcher with only this argument: {\"message\":\"\"}.\n Copy the current user message verbatim, including words such as \"again\";\n never reuse a message from an earlier turn. Do not pass outputSchema.\n - read with only this argument:\n {\"url\":\"https://eve.dev/docs/guides/instrumentation\"}.\n Emit both calls simultaneously in the same response, with researcher first\n and read second, so Eve executes them as one parallel batch. The calls are\n independent. NEVER call one by itself, wait for its result, or emit either\n call in a separate response.\n2. Only after both results are available, answer with a single sentence that\n starts with \"Final answer from read:\" and includes the researcher result, read\n title, URL, and read excerpt.\n\nTool execution\nA single tool or subagent call runs as one serial action. If you call multiple independent tools or subagents in one response, eve treats that batch as parallel work. Only batch work that is independent and does not rely on another call in the same response.\n\nAgent messaging\nAgents you have already delegated to stay available after they answer. eve injects the current `` list into the conversation as a note labeled `[Agents]`; it is added automatically by the framework, not written by the user, and never requires a reply. The list is only a record of those existing agents — their `agentId`, name, and latest status. It does not limit which subagent tools you can call: your tool list is the source of truth, and any subagent tool can always be called without `agentId` to start a new agent, including when the `` list is empty or absent. Pass `agentId` to the same subagent tool only to continue one of those existing agents' sessions.", "role": "system" }, { @@ -714,6 +722,10 @@ span_tree: ], "role": "tool" }, + { + "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n", + "role": "user" + }, { "content": [ { @@ -788,6 +800,10 @@ span_tree: } ], "role": "tool" + }, + { + "content": "[Agents]\n\n\" name=\"researcher\">Researcher result: [Eve instrumentation](https://eve.dev/docs/guides/instrumentation)\n\" name=\"researcher\">Researcher result: Eve instrumentation https://eve.dev/docs/guides/instrumentation\n", + "role": "user" } ] output: [ diff --git a/e2e/scenarios/eve-instrumentation/package.json b/e2e/scenarios/eve-instrumentation/package.json index 71f611306..bc9cb1c09 100644 --- a/e2e/scenarios/eve-instrumentation/package.json +++ b/e2e/scenarios/eve-instrumentation/package.json @@ -14,10 +14,11 @@ }, "dependencies": { "@openrouter/ai-sdk-provider": "3.0.0", - "ai": "7.0.58", + "ai": "7.0.82", "eve-v0": "npm:eve@0.22.1", + "eve-v0-latest-pinned": "npm:eve@0.37.1", "eve-v0-provider": "npm:eve@0.34.0", - "eve-v0-latest": "npm:eve@0.37.1", + "eve-v0-latest": "npm:eve@0.47.6", "zod": "4.3.6" } } diff --git a/e2e/scenarios/eve-instrumentation/pnpm-lock.yaml b/e2e/scenarios/eve-instrumentation/pnpm-lock.yaml index 3c6aaf933..e102af294 100644 --- a/e2e/scenarios/eve-instrumentation/pnpm-lock.yaml +++ b/e2e/scenarios/eve-instrumentation/pnpm-lock.yaml @@ -10,39 +10,42 @@ importers: dependencies: '@openrouter/ai-sdk-provider': specifier: 3.0.0 - version: 3.0.0(ai@7.0.58(zod@4.3.6))(zod@4.3.6) + version: 3.0.0(ai@7.0.82(zod@4.3.6))(zod@4.3.6) ai: - specifier: 7.0.58 - version: 7.0.58(zod@4.3.6) + specifier: 7.0.82 + version: 7.0.82(zod@4.3.6) eve-v0: specifier: npm:eve@0.22.1 - version: eve@0.22.1(ai@7.0.58(zod@4.3.6)) + version: eve@0.22.1(ai@7.0.82(zod@4.3.6)) eve-v0-latest: + specifier: npm:eve@0.47.6 + version: eve@0.47.6(ai@7.0.82(zod@4.3.6)) + eve-v0-latest-pinned: specifier: npm:eve@0.37.1 - version: eve@0.37.1(ai@7.0.58(zod@4.3.6)) + version: eve@0.37.1(ai@7.0.82(zod@4.3.6)) eve-v0-provider: specifier: npm:eve@0.34.0 - version: eve@0.34.0(ai@7.0.58(zod@4.3.6)) + version: eve@0.34.0(ai@7.0.82(zod@4.3.6)) zod: specifier: 4.3.6 version: 4.3.6 packages: - '@ai-sdk/gateway@4.0.46': - resolution: {integrity: sha512-LIAO6kAG8fpXQb9L0iwPk1FIbXftvqnyC56v5NEAzeWTeL8fUsy/Hx86VPBTWEDFdwbVprjWifJOAqS6AOj3mA==} + '@ai-sdk/gateway@4.0.67': + resolution: {integrity: sha512-LtyxLkg7dZ2iz8Ouh1806BJbA+q+FKc/mXUCl4v/wdNNIGtbfk80dNtlhqjhqOZa4dnfc3caVafRL7ocxzoegA==} engines: {node: '>=22'} peerDependencies: zod: ^3.25.76 || ^4.1.8 - '@ai-sdk/provider-utils@5.0.25': - resolution: {integrity: sha512-xscPPHCSjCHWrdhai25sbHCJeKNLW/3D1uSpUZa4cEtTKXA8OnPQ3+Rfu1SmM5Ea/Mf8Dfn3cllw9zeMzo/zFA==} + '@ai-sdk/provider-utils@5.0.32': + resolution: {integrity: sha512-MZUhlINn6FzKIWuX3T36h+yM9d7bG+yatH+kC99ZCe0DHxXfP73KwaoLiLcZDPQDamFyO3umPPBLJieZJyG4DQ==} engines: {node: '>=22'} peerDependencies: zod: ^3.25.76 || ^4.1.8 - '@ai-sdk/provider@4.0.7': - resolution: {integrity: sha512-6or44XprPzKbr8zkmzosowSE0pxkvJcoojBL+mCZvPUt3kvXp3XSNqeVun9golb1acEfSo6yaEBRT18h2VU+1Q==} + '@ai-sdk/provider@4.0.8': + resolution: {integrity: sha512-aWO7iwhFUGf347tCwNGggggfmZigaSu7TF739IZSrWWABUp7zkb4Cr3fMqvBe5EIS7ABJJu3Cadn0g/zs1G0QQ==} engines: {node: '>=22'} '@emnapi/core@1.11.1': @@ -181,8 +184,8 @@ packages: '@workflow/serde@4.1.0': resolution: {integrity: sha512-pav4F2BoirECWR7Nf1TKt+2eETcBj7jj4cBefQ8VXQCA6NPkaKeLfj/zMgi+3zYV5ZIBT4GuUiphsj0/b9hPQQ==} - ai@7.0.58: - resolution: {integrity: sha512-GfgO90CQQ0yYuoxJAUOeQ6tviyYw1BUIDygSZ1q3Ce6kSc93tYmB5eltKY/NxC0YOouAax7JvDqVnYxvIAr04Q==} + ai@7.0.82: + resolution: {integrity: sha512-XRl4uMQyTSn5LPRs75HAb75K27T2IeQ/6oesgBCMMVaQZ2le6pqGrDrVZOrDIujLKtHgStp09H3dZ4CwFnVe3A==} engines: {node: '>=22'} peerDependencies: zod: ^3.25.76 || ^4.1.8 @@ -300,6 +303,26 @@ packages: microsandbox: optional: true + eve@0.47.6: + resolution: {integrity: sha512-k7tXAMBadjOW0LBHeEV3WDoyeufwTzIxAmInuCf7QFVFoh0CrYI0UJcD0zVnVQdDc0X1fVVj79Q76Twwh6QvTw==} + engines: {node: '>=24'} + hasBin: true + peerDependencies: + '@opentelemetry/api': ^1.0.0 + ai: ^7.0.82 + braintrust: ^3.0.0 + just-bash: ^3.1.0 + microsandbox: ^0.5.0 + peerDependenciesMeta: + '@opentelemetry/api': + optional: true + braintrust: + optional: true + just-bash: + optional: true + microsandbox: + optional: true + eventsource-parser@3.1.0: resolution: {integrity: sha512-kJezFj9YFAMLeORyi7aCLxLbD5/qWMQnoMVlVPyHIll7lgRJCc3JVln9Vgl9nwQi0YkMnhdGTMNn7CkRRAptMg==} engines: {node: '>=18.0.0'} @@ -478,23 +501,23 @@ packages: snapshots: - '@ai-sdk/gateway@4.0.46(zod@4.3.6)': + '@ai-sdk/gateway@4.0.67(zod@4.3.6)': dependencies: - '@ai-sdk/provider': 4.0.7 - '@ai-sdk/provider-utils': 5.0.25(zod@4.3.6) + '@ai-sdk/provider': 4.0.8 + '@ai-sdk/provider-utils': 5.0.32(zod@4.3.6) '@vercel/oidc': 3.2.0 zod: 4.3.6 - '@ai-sdk/provider-utils@5.0.25(zod@4.3.6)': + '@ai-sdk/provider-utils@5.0.32(zod@4.3.6)': dependencies: - '@ai-sdk/provider': 4.0.7 + '@ai-sdk/provider': 4.0.8 '@standard-schema/spec': 1.1.0 '@workflow/serde': 4.1.0 eventsource-parser: 3.1.0 undici: 7.29.0 zod: 4.3.6 - '@ai-sdk/provider@4.0.7': + '@ai-sdk/provider@4.0.8': dependencies: json-schema: 0.4.0 @@ -521,9 +544,9 @@ snapshots: '@tybys/wasm-util': 0.10.3 optional: true - '@openrouter/ai-sdk-provider@3.0.0(ai@7.0.58(zod@4.3.6))(zod@4.3.6)': + '@openrouter/ai-sdk-provider@3.0.0(ai@7.0.82(zod@4.3.6))(zod@4.3.6)': dependencies: - ai: 7.0.58(zod@4.3.6) + ai: 7.0.82(zod@4.3.6) zod: 4.3.6 '@oxc-project/types@0.138.0': {} @@ -590,11 +613,11 @@ snapshots: '@workflow/serde@4.1.0': {} - ai@7.0.58(zod@4.3.6): + ai@7.0.82(zod@4.3.6): dependencies: - '@ai-sdk/gateway': 4.0.46(zod@4.3.6) - '@ai-sdk/provider': 4.0.7 - '@ai-sdk/provider-utils': 5.0.25(zod@4.3.6) + '@ai-sdk/gateway': 4.0.67(zod@4.3.6) + '@ai-sdk/provider': 4.0.8 + '@ai-sdk/provider-utils': 5.0.32(zod@4.3.6) zod: 4.3.6 consola@3.4.2: {} @@ -612,9 +635,9 @@ snapshots: httpxy: 0.5.4 srvx: 0.11.21 - eve@0.22.1(ai@7.0.58(zod@4.3.6)): + eve@0.22.1(ai@7.0.82(zod@4.3.6)): dependencies: - ai: 7.0.58(zod@4.3.6) + ai: 7.0.82(zod@4.3.6) nitro: 3.0.260610-beta transitivePeerDependencies: - '@azure/app-configuration' @@ -656,9 +679,54 @@ snapshots: - xml2js - zephyr-agent - eve@0.34.0(ai@7.0.58(zod@4.3.6)): + eve@0.34.0(ai@7.0.82(zod@4.3.6)): + dependencies: + ai: 7.0.82(zod@4.3.6) + nitro: 3.0.260610-beta + undici: 8.9.0 + transitivePeerDependencies: + - '@azure/app-configuration' + - '@azure/cosmos' + - '@azure/data-tables' + - '@azure/identity' + - '@azure/keyvault-secrets' + - '@azure/storage-blob' + - '@capacitor/preferences' + - '@deno/kv' + - '@electric-sql/pglite' + - '@libsql/client' + - '@netlify/blobs' + - '@netlify/runtime' + - '@planetscale/database' + - '@upstash/redis' + - '@vercel/blob' + - '@vercel/functions' + - '@vercel/kv' + - '@vercel/queue' + - aws4fetch + - better-sqlite3 + - chokidar + - dotenv + - drizzle-orm + - giget + - idb-keyval + - ioredis + - jiti + - lru-cache + - miniflare + - mongodb + - mysql2 + - rollup + - sqlite3 + - uploadthing + - vite + - wrangler + - xml2js + - zephyr-agent + + eve@0.37.1(ai@7.0.82(zod@4.3.6)): dependencies: - ai: 7.0.58(zod@4.3.6) + ai: 7.0.82(zod@4.3.6) nitro: 3.0.260610-beta undici: 8.9.0 transitivePeerDependencies: @@ -701,9 +769,9 @@ snapshots: - xml2js - zephyr-agent - eve@0.37.1(ai@7.0.58(zod@4.3.6)): + eve@0.47.6(ai@7.0.82(zod@4.3.6)): dependencies: - ai: 7.0.58(zod@4.3.6) + ai: 7.0.82(zod@4.3.6) nitro: 3.0.260610-beta undici: 8.9.0 transitivePeerDependencies: diff --git a/e2e/scenarios/eve-instrumentation/scenario.test.ts b/e2e/scenarios/eve-instrumentation/scenario.test.ts index 41c83ea14..0a6caf726 100644 --- a/e2e/scenarios/eve-instrumentation/scenario.test.ts +++ b/e2e/scenarios/eve-instrumentation/scenario.test.ts @@ -38,6 +38,13 @@ const eveScenarios = await Promise.all( provider: true, variantKey: "eve-v0-provider", }, + { + cassetteKey: "eve-v0-latest", + dependencyName: "eve-v0-latest-pinned", + label: "v0 latest pinned", + provider: true, + variantKey: "eve-v0-latest-pinned", + }, { cassetteKey: "eve-v0-latest", dependencyName: "eve-v0-latest", @@ -409,8 +416,14 @@ describe.sequential("eve instrumentation variants", () => { ).toHaveLength(1); } + const snapshotEvents = JSON.parse( + JSON.stringify(events).replace( + /ag_researcher:[0-9a-f]+/g, + "ag_researcher:", + ), + ) as CapturedLogEvent[]; await matchSpanTreeSnapshot( - events, + snapshotEvents, resolveFileSnapshotPath( import.meta.url, `${scenario.variantKey}.span-tree.json`, diff --git a/js/src/instrumentation/plugins/eve-provider.test.ts b/js/src/instrumentation/plugins/eve-provider.test.ts index ec777abe4..bd45ab664 100644 --- a/js/src/instrumentation/plugins/eve-provider.test.ts +++ b/js/src/instrumentation/plugins/eve-provider.test.ts @@ -55,6 +55,12 @@ describe("braintrustEveInstrumentation provider lifecycle", () => { expect(instrumentation).toMatchObject({ capture: "content", setup, + tracePolicy: expect.any(Function), + }); + expect(instrumentation.tracePolicy?.({ audience: "private" })).toEqual({ + emit: true, + recordInputs: true, + recordOutputs: true, }); expect( Reflect.get(instrumentation, Symbol.for("eve.instrumentation.provider")), diff --git a/js/src/instrumentation/plugins/eve-provider.ts b/js/src/instrumentation/plugins/eve-provider.ts index 53d46d066..a3470d39d 100644 --- a/js/src/instrumentation/plugins/eve-provider.ts +++ b/js/src/instrumentation/plugins/eve-provider.ts @@ -61,7 +61,15 @@ export function createEveInstrumentationProvider( ): EveProviderDefinition { const bridge = new EveProviderBridge(options.metadata); return { + // Eve versions before tracePolicy use capture to decide whether provider + // events include content. Newer versions prefer tracePolicy when both are + // present, so keep the deprecated field for backwards compatibility. capture: "content", + tracePolicy: () => ({ + emit: true, + recordInputs: true, + recordOutputs: true, + }), events: { "action.completed": (event, context) => bridge.handleActionTerminal(event, context), diff --git a/js/src/vendor-sdk-types/eve.ts b/js/src/vendor-sdk-types/eve.ts index 22b93a1c6..b8d535a88 100644 --- a/js/src/vendor-sdk-types/eve.ts +++ b/js/src/vendor-sdk-types/eve.ts @@ -694,8 +694,29 @@ type EveProviderHandler = ( context: EveProviderContext, ) => void | PromiseLike; +type EveChannelAudience = "private" | "public" | "unknown"; + +interface EveTraceCaptureContext { + readonly agentName?: string; + readonly audience: EveChannelAudience; + readonly channelType?: string; +} + +type EveTracePolicyDecision = + | { readonly emit: false } + | { + readonly emit: true; + readonly recordInputs: boolean; + readonly recordOutputs: boolean; + }; + +type EveTraceCapturePolicy = ( + trace: EveTraceCaptureContext, +) => EveTracePolicyDecision | boolean; + export interface EveProviderDefinition { readonly capture?: "content" | "metadata"; + readonly tracePolicy?: EveTraceCapturePolicy; readonly events?: { readonly "action.completed"?: EveProviderHandler; readonly "action.failed"?: EveProviderHandler;