From e67c49cb5fe5602901271e9a5ca05add801aa523 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Tue, 16 Jun 2026 08:49:50 +0800 Subject: [PATCH 1/9] refactor: foundry e2e --- .../workflows/microsoft-foundry-e2e-eval.yml | 17 ++++++++++++++++- tests/utils/agent-runner.ts | 19 +++++++++++-------- 2 files changed, 27 insertions(+), 9 deletions(-) diff --git a/.github/workflows/microsoft-foundry-e2e-eval.yml b/.github/workflows/microsoft-foundry-e2e-eval.yml index a6b419e32..86d677e7a 100644 --- a/.github/workflows/microsoft-foundry-e2e-eval.yml +++ b/.github/workflows/microsoft-foundry-e2e-eval.yml @@ -23,7 +23,7 @@ permissions: {} jobs: eval: name: Run Microsoft Foundry E2E evals - runs-on: ubuntu-latest + runs-on: windows-latest permissions: id-token: write contents: read @@ -130,11 +130,26 @@ jobs: - name: Build plugin output run: npm run build + - name: Remove unrelated skills + shell: bash + run: | + set -euo pipefail + + skills_dir="output/skills" + if [[ ! -d "${skills_dir}/microsoft-foundry" ]]; then + echo "::error::${skills_dir}/microsoft-foundry not found." + exit 1 + fi + + find "${skills_dir}" -mindepth 1 -maxdepth 1 -type d ! -name "microsoft-foundry" -exec rm -rf -- {} + + - name: Run Microsoft Foundry Vally evals working-directory: tests + shell: bash env: COPILOT_GITHUB_TOKEN: ${{ secrets.COPILOT_GITHUB_TOKEN }} VALLY_RUNS: ${{ inputs.runs }} + VALLY_RUNNER_DISABLE_AZURE_MCP: "true" run: | set -euo pipefail diff --git a/tests/utils/agent-runner.ts b/tests/utils/agent-runner.ts index f3bc366ce..aa6cdfc07 100644 --- a/tests/utils/agent-runner.ts +++ b/tests/utils/agent-runner.ts @@ -700,20 +700,23 @@ export function useAgentRunner(agentRunnerConfig: AgentRunnerConfig) { } const noSkills = process.env.NO_SKILLS === "true"; + const disableAzureMcp = process.env.VALLY_RUNNER_DISABLE_AZURE_MCP === "true"; const model = config.model ?? modelOverride ?? "claude-sonnet-4.6"; const session = await client.createSession({ model: model, onPermissionRequest: approveAll, skillDirectories: noSkills ? [] : [skillDirectory], disabledSkills: disabledSkills, - mcpServers: { - azure: { - type: "stdio", - command: "npx", - args: ["-y", "@azure/mcp", "server", "start"], - tools: ["*"] + ...(disableAzureMcp ? {} : { + mcpServers: { + azure: { + type: "stdio", + command: "npx", + args: ["-y", "@azure/mcp", "server", "start"], + tools: ["*"] + } } - }, + }), systemMessage: config.systemPrompt }); entry.session = session; @@ -1074,4 +1077,4 @@ function sanitizeFileName(name: string): string { .replace(/-+/g, "-") // Collapse multiple dashes .replace(/_+/g, "_") // Collapse multiple underscores .substring(0, 200); // Limit length -} \ No newline at end of file +} From c625309940c2f4eb245cc2933d239cac3658ee78 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Tue, 16 Jun 2026 09:57:54 +0800 Subject: [PATCH 2/9] chore: upgrade vally to 0.6.0 --- .../workflows/microsoft-foundry-e2e-eval.yml | 1 - tests/package-lock.json | 242 ++++++++++++------ tests/package.json | 2 +- 3 files changed, 165 insertions(+), 80 deletions(-) diff --git a/.github/workflows/microsoft-foundry-e2e-eval.yml b/.github/workflows/microsoft-foundry-e2e-eval.yml index 86d677e7a..b4c77a448 100644 --- a/.github/workflows/microsoft-foundry-e2e-eval.yml +++ b/.github/workflows/microsoft-foundry-e2e-eval.yml @@ -161,7 +161,6 @@ jobs: npm run test:vally -- \ --suite foundry-e2e \ --runs "${VALLY_RUNS}" \ - --output jsonl \ --junit \ --threshold 0.8 diff --git a/tests/package-lock.json b/tests/package-lock.json index 89175f3e6..ac3f96b30 100644 --- a/tests/package-lock.json +++ b/tests/package-lock.json @@ -13,7 +13,7 @@ "@eslint/js": "^10.0.0", "@github/copilot": "1.0.49", "@github/copilot-sdk": "^0.3.0", - "@microsoft/vally-cli": "^0.5.0", + "@microsoft/vally-cli": "^0.6.0", "@types/jest": "^30.0.0", "@types/node": "^25.6.0", "cross-env": "^10.1.0", @@ -1161,9 +1161,9 @@ } }, "node_modules/@hono/node-server": { - "version": "2.0.3", - "resolved": "https://registry.npmjs.org/@hono/node-server/-/node-server-2.0.3.tgz", - "integrity": "sha512-a0jV+/HRe3G5zjFID3zObAQFdkl6zpxTuqktdDDXS3MJKcrZIkB8OkLpNBlY/WXFqv2HF4a0takPej+aNFczWA==", + "version": "2.0.5", + "resolved": "https://registry.npmjs.org/@hono/node-server/-/node-server-2.0.5.tgz", + "integrity": "sha512-yQFvDmyDo3y6rEOJZDUYPJ49DIKTPpIk4kGvm40xx4Ejne0Pu9a1+exxPN+C1UppWK/WGZX9F++/Xs231tE86g==", "dev": true, "license": "MIT", "engines": { @@ -1707,70 +1707,72 @@ "license": "MIT" }, "node_modules/@microsoft/vally": { - "version": "0.5.0", - "resolved": "https://registry.npmjs.org/@microsoft/vally/-/vally-0.5.0.tgz", - "integrity": "sha512-R5NhYrLJ724x5k/K82ZaRwRwplRQU3EwT08rONlFe9qM3iMisoWVv9CBj+zYuVdWQuzy4SjwS97wkZ3fb7AjQQ==", + "version": "0.6.0", + "resolved": "https://registry.npmjs.org/@microsoft/vally/-/vally-0.6.0.tgz", + "integrity": "sha512-b283YRDFZXUkKNKY3+1EfMBVbHrBLIs5jfUi7lIQ8N0Y10lVsNnNGkRbtTbd7tLYZajr6AhtnpIroc4RFzo1cQ==", "dev": true, "dependencies": { - "@github/copilot-sdk": "1.0.0-beta.5", + "@github/copilot-sdk": "^1.0.0", "js-tiktoken": "^1.0.21", "picomatch": "^4.0.4", - "yaml": "^2.9.0" + "yaml": "^2.9.0", + "zod": "^4.4.3" } }, "node_modules/@microsoft/vally-cli": { - "version": "0.5.0", - "resolved": "https://registry.npmjs.org/@microsoft/vally-cli/-/vally-cli-0.5.0.tgz", - "integrity": "sha512-emTvEfNo2+9rtglHIjev85tlwwWXHRhbkHpxyWWliIVuK+ICt0BwUMo33pnQ4tfKEaCoxz0kSyCTuerUnj1cxQ==", + "version": "0.6.0", + "resolved": "https://registry.npmjs.org/@microsoft/vally-cli/-/vally-cli-0.6.0.tgz", + "integrity": "sha512-6JzbuB2EpQTuA9snexb7xsos1oLiB8JhWY6Xe2JCOd4clsv1fj3WFhv/jxiU6OMmSC2vQQ3hsM/LDuyAoEUJyw==", "dev": true, "dependencies": { - "@microsoft/vally": "^0.5.0", - "@microsoft/vally-server": "^0.5.0", - "commander": "^14.0.3" + "@microsoft/vally": "^0.6.0", + "@microsoft/vally-server": "^0.6.0", + "commander": "^15.0.0" }, "bin": { "vally": "dist/index.js" } }, "node_modules/@microsoft/vally-server": { - "version": "0.5.0", - "resolved": "https://registry.npmjs.org/@microsoft/vally-server/-/vally-server-0.5.0.tgz", - "integrity": "sha512-2I5rcP0ihnmNT67o7xce3OOqT7RnidckAGtFaYwvtyVzst9qZ2SQ+Tjg/EepTnfNKi50wWSfPCzm2Epn4YYQ1A==", + "version": "0.6.0", + "resolved": "https://registry.npmjs.org/@microsoft/vally-server/-/vally-server-0.6.0.tgz", + "integrity": "sha512-8LPB/NtJo/vYwiNYQdU6oc/f4W08QwQtEiE+M+g5O/dpJQsM9RmPINZrb/DQJl7yngBZQ8VShe1HtqeuwX1wvw==", "dev": true, "dependencies": { - "@hono/node-server": "^2.0.3", - "@microsoft/vally": "^0.5.0", + "@hono/node-server": "^2.0.4", + "@microsoft/vally": "^0.6.0", "better-sqlite3": "^12.10.0", - "hono": "^4.12.21" + "hono": "^4.12.25" } }, "node_modules/@microsoft/vally/node_modules/@github/copilot": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot/-/copilot-1.0.51.tgz", - "integrity": "sha512-yKXbMeApxO8P68/BeSS/lmIRsCprcMdY8MRRp+Vp/QymCv59o4lxDcAIVq2h/CD8vJHoiG4OijdWydd76yoqLw==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot/-/copilot-1.0.63.tgz", + "integrity": "sha512-e8DRYiWJQc4kepVXsXjC8vpDU2FXS/TfR+Z6p/KAojfcwIUZzKMAfCV5D1lD25hV4CryVH1Z9t7mHqChickj0Q==", "dev": true, "license": "SEE LICENSE IN LICENSE.md", "dependencies": { - "detect-libc": "^2.1.2" + "detect-libc": "^2.1.2", + "os-theme": "^0.0.8" }, "bin": { "copilot": "npm-loader.js" }, "optionalDependencies": { - "@github/copilot-darwin-arm64": "1.0.51", - "@github/copilot-darwin-x64": "1.0.51", - "@github/copilot-linux-arm64": "1.0.51", - "@github/copilot-linux-x64": "1.0.51", - "@github/copilot-linuxmusl-arm64": "1.0.51", - "@github/copilot-linuxmusl-x64": "1.0.51", - "@github/copilot-win32-arm64": "1.0.51", - "@github/copilot-win32-x64": "1.0.51" + "@github/copilot-darwin-arm64": "1.0.63", + "@github/copilot-darwin-x64": "1.0.63", + "@github/copilot-linux-arm64": "1.0.63", + "@github/copilot-linux-x64": "1.0.63", + "@github/copilot-linuxmusl-arm64": "1.0.63", + "@github/copilot-linuxmusl-x64": "1.0.63", + "@github/copilot-win32-arm64": "1.0.63", + "@github/copilot-win32-x64": "1.0.63" } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-darwin-arm64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-darwin-arm64/-/copilot-darwin-arm64-1.0.51.tgz", - "integrity": "sha512-i713sW3GzbeLKowGVY6/A97lGkUMJNVdUD0oaUWTWmXX08u+hWsnVKbqL4EQlw7x8xU511X5vkgFMi31DWyCuQ==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-darwin-arm64/-/copilot-darwin-arm64-1.0.63.tgz", + "integrity": "sha512-z6CMBxNDlKvT6bvOpqhu4M2bhb0daEbVwSe9SN9WfDUJbt7bpoL7OKKas428iyPSWHoL2WXwxSsy/FjIwSLV6w==", "cpu": [ "arm64" ], @@ -1785,9 +1787,9 @@ } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-darwin-x64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-darwin-x64/-/copilot-darwin-x64-1.0.51.tgz", - "integrity": "sha512-c67SbMznclcHqlJINXBCwudhqRgE5HNaY9fqMQqu954+ezVa6Q/2hwhCU51PNbYLWtZTGgXsgWnrxOg77hh0ug==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-darwin-x64/-/copilot-darwin-x64-1.0.63.tgz", + "integrity": "sha512-YKd7cXZgAGxhudzrtWdWh2NS35p2G5bV22Gz3jhEyBTqmq45o4sD4OwO87+UpkvM+3nZpwsHaLd3a+ILYX6OXg==", "cpu": [ "x64" ], @@ -1802,13 +1804,16 @@ } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-linux-arm64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-linux-arm64/-/copilot-linux-arm64-1.0.51.tgz", - "integrity": "sha512-MlQeTB4CSPnG2BZTxsPSV5a7rjsqFOzhTCVCNjLeht3ODObWjrIYhtzVF7h/nue9ii96u9RBB0gIAfoBReryTw==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-linux-arm64/-/copilot-linux-arm64-1.0.63.tgz", + "integrity": "sha512-A3DOeEfmsJH9j1N+QLc7WXmESBskbezmhDyhyAJcHkw0ngRbKctuWQf/evUHFMh/kgwy1Lr/+9jXJm3NZqr0MA==", "cpu": [ "arm64" ], "dev": true, + "libc": [ + "glibc" + ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -1819,13 +1824,16 @@ } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-linux-x64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-linux-x64/-/copilot-linux-x64-1.0.51.tgz", - "integrity": "sha512-fniGTwR5KLFfNDjSFbWvZ3Bno+2bXsMdNM0l3dFHwVTHyBqQSXZ3xvEEDadGimCxgKfRDRt1M1FYnUpqhLYf/Q==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-linux-x64/-/copilot-linux-x64-1.0.63.tgz", + "integrity": "sha512-OMKfZJRoDaJOV7vuWX/nFPNdLa9/H+nhajdE83v4YT9mKLXr86aWrkXE3pPoDYsKWvgQFHg4APA6oZPao0Fyow==", "cpu": [ "x64" ], "dev": true, + "libc": [ + "glibc" + ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -1836,13 +1844,16 @@ } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-linuxmusl-arm64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-linuxmusl-arm64/-/copilot-linuxmusl-arm64-1.0.51.tgz", - "integrity": "sha512-vg9sWZw4u/bqHa7ylF/GZeuznt+k4/Em899C++CTBU4CKhtAaxd2TZDsEV0Ap2DXzP2UFxCn77vZoHyxByMI5A==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-linuxmusl-arm64/-/copilot-linuxmusl-arm64-1.0.63.tgz", + "integrity": "sha512-jcIo6B3uHgcOluNfUHp+6atShKKrXYBPLaRyF6aDT699lwI83gW9KTDuEvDs5FDg8qWsWFfOl+al2dkWDYD3CQ==", "cpu": [ "arm64" ], "dev": true, + "libc": [ + "musl" + ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -1853,13 +1864,16 @@ } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-linuxmusl-x64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-linuxmusl-x64/-/copilot-linuxmusl-x64-1.0.51.tgz", - "integrity": "sha512-zxXRdzjshHTQd/LDWmOIDXt0T8nvw66ue6cneAXHhLXWzuiv5mqPKnxuHQyvQDt+IBEyq9utuetlKxcAVo+gYw==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-linuxmusl-x64/-/copilot-linuxmusl-x64-1.0.63.tgz", + "integrity": "sha512-BEdBbEF3fG7VqXzuaAY4JtmbdGSkpJFeb2ZQYaMpq7OP3aS7ssGe1cCX8ehZNegcMM/eb4GC6PXNXsvl3X/PAQ==", "cpu": [ "x64" ], "dev": true, + "libc": [ + "musl" + ], "license": "SEE LICENSE IN LICENSE.md", "optional": true, "os": [ @@ -1870,24 +1884,24 @@ } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-sdk": { - "version": "1.0.0-beta.5", - "resolved": "https://registry.npmjs.org/@github/copilot-sdk/-/copilot-sdk-1.0.0-beta.5.tgz", - "integrity": "sha512-WT/JZXkLi4mZKjyQakqBEv4lU8zHOSERicEB75KN1oLVIvKRvL8WyfP6dqIjeO1xmCKd6/qKVGlQxJChCIzJ1w==", + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/@github/copilot-sdk/-/copilot-sdk-1.0.1.tgz", + "integrity": "sha512-w6AaS0WqqTE/3iyUrZznvgCLQhsUF7ZmEVCneacuHCfOzlH0r6ww9WUmyA0zgqmXO75V0IYrkIcnFke/qJkkDg==", "dev": true, "license": "MIT", "dependencies": { - "@github/copilot": "^1.0.51", + "@github/copilot": "^1.0.61", "vscode-jsonrpc": "^8.2.1", "zod": "^4.3.6" }, "engines": { - "node": ">=20.0.0" + "node": "^20.19.0 || >=22.12.0" } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-win32-arm64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-win32-arm64/-/copilot-win32-arm64-1.0.51.tgz", - "integrity": "sha512-/SP8DfOukjllCXavgBNI0qwJa+8hCFRNK7Q3/Q3qzAOvaWUZZkabKSVZfXaGxerTGpGq009Zg3nyIPR0jfm60w==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-win32-arm64/-/copilot-win32-arm64-1.0.63.tgz", + "integrity": "sha512-7FqUwOmtoeBoOn4zkKQqRL+WGFwektVRSr5Po2FvPAbKxGXGyFXApZTmRLqVcHhMKDRzMb8KLST1LU1TMTY/wg==", "cpu": [ "arm64" ], @@ -1902,9 +1916,9 @@ } }, "node_modules/@microsoft/vally/node_modules/@github/copilot-win32-x64": { - "version": "1.0.51", - "resolved": "https://registry.npmjs.org/@github/copilot-win32-x64/-/copilot-win32-x64-1.0.51.tgz", - "integrity": "sha512-ZB5Jr9m4ZR8gFOwXnYGNfdU+bMFeUgj1OCU3x64Tx5GC6Uln/pf8Ue5LHlsBkBq/NuKvkp/g4GARDIHBCKXEnQ==", + "version": "1.0.63", + "resolved": "https://registry.npmjs.org/@github/copilot-win32-x64/-/copilot-win32-x64-1.0.63.tgz", + "integrity": "sha512-RC/6y9KHdw/YRCrCEksF2RzbeblfBUNE7bkYZxygaQGYThuv1GeZL2YD2jVqxC2LxKzsUmWGvwEMxerfR6pmeQ==", "cpu": [ "x64" ], @@ -1918,6 +1932,36 @@ "copilot-win32-x64": "copilot.exe" } }, + "node_modules/@microsoft/vally/node_modules/os-theme": { + "version": "0.0.8", + "resolved": "https://registry.npmjs.org/os-theme/-/os-theme-0.0.8.tgz", + "integrity": "sha512-u1q3bLSv5uMHNIiPItkfDrHXu6ZFs2juwqxWREFM/uVBa+7Kkhy2v49LmJev2JcinGwqiEccElB/XsH9gwasuA==", + "dev": true, + "license": "MIT", + "optionalDependencies": { + "@os-theme/darwin-arm64": "0.0.8", + "@os-theme/linux-x64": "0.0.8", + "@os-theme/win32-x64": "0.0.8" + }, + "peerDependencies": { + "typescript": "^5" + } + }, + "node_modules/@microsoft/vally/node_modules/typescript": { + "version": "5.9.3", + "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz", + "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", + "dev": true, + "license": "Apache-2.0", + "peer": true, + "bin": { + "tsc": "bin/tsc", + "tsserver": "bin/tsserver" + }, + "engines": { + "node": ">=14.17" + } + }, "node_modules/@napi-rs/wasm-runtime": { "version": "0.2.12", "resolved": "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-0.2.12.tgz", @@ -1944,6 +1988,48 @@ ], "license": "MIT" }, + "node_modules/@os-theme/darwin-arm64": { + "version": "0.0.8", + "resolved": "https://registry.npmjs.org/@os-theme/darwin-arm64/-/darwin-arm64-0.0.8.tgz", + "integrity": "sha512-gMsOs+8Ju396a5yyMWigkbA0dMTxD78U3HzG3mlpiAyn6hfd5dbyI4VGP+sfTB82KGgWLzIhWWTFX5UYY6iX0A==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ] + }, + "node_modules/@os-theme/linux-x64": { + "version": "0.0.8", + "resolved": "https://registry.npmjs.org/@os-theme/linux-x64/-/linux-x64-0.0.8.tgz", + "integrity": "sha512-zvjmBUiSQPjM1RbhpsfCDYMJxW4eLlGmkFPnpteC/03X2lz6CjiX2hfbN2EWLxXjNnIje3Jqaen8IsqEnWrRBg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ] + }, + "node_modules/@os-theme/win32-x64": { + "version": "0.0.8", + "resolved": "https://registry.npmjs.org/@os-theme/win32-x64/-/win32-x64-0.0.8.tgz", + "integrity": "sha512-N3yxKNbVl2IBa/ncDuq55QhwqwUjnYLJxDKMEmYeJbLIV950qZNojPw3scXA6PbfxPZfIiRa8iz1pzNg9XxP8w==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ] + }, "node_modules/@package-json/types": { "version": "0.0.12", "resolved": "https://registry.npmjs.org/@package-json/types/-/types-0.0.12.tgz", @@ -3056,9 +3142,9 @@ } }, "node_modules/better-sqlite3": { - "version": "12.10.0", - "resolved": "https://registry.npmjs.org/better-sqlite3/-/better-sqlite3-12.10.0.tgz", - "integrity": "sha512-CyzaZRQKyHkB2ZInfTTl2nvT33EbDpjkLEbE8/Zck3Ll6O0qqvuGdrJ45HgtH+HykRg88ITY3AdreBGN70aBSQ==", + "version": "12.11.1", + "resolved": "https://registry.npmjs.org/better-sqlite3/-/better-sqlite3-12.11.1.tgz", + "integrity": "sha512-dq9AtApgg5PGFtBzPFSBl3HZQjHok5gaQCM6zh2Yk0aSmDCs1CbnVI8/HgASQkNKsWFpseIO9beg5xxpYhbIfA==", "dev": true, "hasInstallScript": true, "license": "MIT", @@ -3407,13 +3493,13 @@ "license": "MIT" }, "node_modules/commander": { - "version": "14.0.3", - "resolved": "https://registry.npmjs.org/commander/-/commander-14.0.3.tgz", - "integrity": "sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw==", + "version": "15.0.0", + "resolved": "https://registry.npmjs.org/commander/-/commander-15.0.0.tgz", + "integrity": "sha512-z67u4ZhzCL/Tydu1lJARtEZYWbWaN7oYLHbsuzocr6y4N6WZAagG3RQ4FW61V1/0+jImpj293XfrcYnd1qxtPg==", "dev": true, "license": "MIT", "engines": { - "node": ">=20" + "node": ">=22.12.0" } }, "node_modules/comment-parser": { @@ -4594,9 +4680,9 @@ } }, "node_modules/hono": { - "version": "4.12.22", - "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.22.tgz", - "integrity": "sha512-7fvVPbB92zNRsQke+uiRGwtTuef0tB2Dg4hWxYfFNvkQhIltWoyi0ONReM5LWA+jJWS3nfT5lTq+qbsIpX0IQw==", + "version": "4.12.25", + "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.25.tgz", + "integrity": "sha512-2NFaIyNVgJmBs/ecmtGzlmluTFs5cHEWGTdu0t1HBwYzoGXOL5nUQBRMXsXWla5i4KkG//QMzVP88m1+I3fdAQ==", "dev": true, "license": "MIT", "engines": { @@ -6089,9 +6175,9 @@ } }, "node_modules/node-abi/node_modules/semver": { - "version": "7.8.1", - "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.1.tgz", - "integrity": "sha512-rkVq3IXh+4FDGch+KwzX3aV9W3kO54GyEgpvBzSyctDA6Xtd7RJQV1xmXbeQp5v7+VzLOfVqiutSE6GICgPFvg==", + "version": "7.8.4", + "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.4.tgz", + "integrity": "sha512-rUCObTnP32Q08R2uuIrt7r9PlEonuTmtuXYcW6s5kjdlj3xbnwe+21yXptAUYcMAABLkYYTtnmzb3w3EDZfueA==", "dev": true, "license": "ISC", "bin": { @@ -7803,9 +7889,9 @@ } }, "node_modules/zod": { - "version": "4.3.6", - "resolved": "https://registry.npmjs.org/zod/-/zod-4.3.6.tgz", - "integrity": "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg==", + "version": "4.4.3", + "resolved": "https://registry.npmjs.org/zod/-/zod-4.4.3.tgz", + "integrity": "sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ==", "dev": true, "license": "MIT", "funding": { diff --git a/tests/package.json b/tests/package.json index 141f5b468..bcd3d5809 100644 --- a/tests/package.json +++ b/tests/package.json @@ -33,7 +33,7 @@ "@eslint/js": "^10.0.0", "@github/copilot": "1.0.49", "@github/copilot-sdk": "^0.3.0", - "@microsoft/vally-cli": "^0.5.0", + "@microsoft/vally-cli": "^0.6.0", "@types/jest": "^30.0.0", "@types/node": "^25.6.0", "cross-env": "^10.1.0", From 26b3302c616bbad044b1c8e64a49d49ba667e76c Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Tue, 16 Jun 2026 10:28:28 +0800 Subject: [PATCH 3/9] chore: fix --- .github/workflows/microsoft-foundry-e2e-eval.yml | 1 + tests/utils/agent-runner.ts | 5 ++++- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/.github/workflows/microsoft-foundry-e2e-eval.yml b/.github/workflows/microsoft-foundry-e2e-eval.yml index b4c77a448..d25e4fd3c 100644 --- a/.github/workflows/microsoft-foundry-e2e-eval.yml +++ b/.github/workflows/microsoft-foundry-e2e-eval.yml @@ -161,6 +161,7 @@ jobs: npm run test:vally -- \ --suite foundry-e2e \ --runs "${VALLY_RUNS}" \ + --workers 1 \ --junit \ --threshold 0.8 diff --git a/tests/utils/agent-runner.ts b/tests/utils/agent-runner.ts index aa6cdfc07..ed601342b 100644 --- a/tests/utils/agent-runner.ts +++ b/tests/utils/agent-runner.ts @@ -847,7 +847,10 @@ export function useAgentRunner(agentRunnerConfig: AgentRunnerConfig) { console.error("Agent runner error:", errorDetails); throw error; } finally { - if (!isTest()) { + // Jest integration tests clean up in afterEach so reports can be written first. + // Non-Jest test runners such as Vally must clean up here; otherwise Copilot CLI + // child processes keep the Node process alive after results are written. + if (!isTest() || !useJest()) { await cleanup(); } } From 1ca607d448adf2ca15c82153803003a7bb841885 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Tue, 16 Jun 2026 11:51:33 +0800 Subject: [PATCH 4/9] chore: add aic --- ...icrosoft-foundry-e2e-eval-report.prompt.md | 31 ++++++++++++++----- 1 file changed, 24 insertions(+), 7 deletions(-) diff --git a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md index a82834d3f..2ea9de89b 100644 --- a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md +++ b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md @@ -136,16 +136,32 @@ Multi-trial schema example: ## Section: Golden Path Token Cost -Purpose: show average token usage for Golden Path trials only. +Purpose: show average token usage and AI credit cost for Golden Path trials only. + +### Pricing Table + +Use this GitHub Copilot Anthropic pricing table. Prices are USD per 1M tokens. + +| Model | Input | Cached input | Cache write | Output | +|---|---:|---:|---:|---:| +| Claude Sonnet 4.6 | $3.00 | $0.30 | $3.75 | $15.00 | Guidance: -- Calculate token usage from `trajectory.metrics.tokenUsage` when present. -- If that aggregate is missing, sum token usage from token usage events. -- Average across Golden Path trials only. -- Input cache rate is average `cacheReadTokens` divided by average input tokens. +Step 1: calculate average token usage. + +- Calculate token usage directly from `trajectory.metrics.tokenUsage`. +- Each Golden Path trial has complete token usage fields: `inputTokens`, `cacheReadTokens`, `cacheWriteTokens`, and `outputTokens`. +- Average these four fields across Golden Path trials only: `inputTokens`, `cacheReadTokens`, `cacheWriteTokens`, and `outputTokens`. +- Calculate `Total tokens` as average `inputTokens` plus average `outputTokens`. - Round token counts to integers and format them with thousands separators. -- Format input cache rate as a percentage with one decimal place. + +Step 2: calculate average AIC. + +- `100` AI credits equals `$1.00`. +- Determine the model used by each Golden Path trial from `trajectory.metrics.tokenUsage.model`, then use the matching model row in the Pricing Table for `inputRate`, `cachedInputRate`, `cacheWriteRate`, and `outputRate`. +- Calculate `averageAic = ((((averageInputTokens - averageCacheReadTokens - averageCacheWriteTokens) * inputRate) + (averageCacheReadTokens * cachedInputRate) + (averageCacheWriteTokens * cacheWriteRate) + (averageOutputTokens * outputRate)) / 1,000,000) * 100`. +- Format `Average AIC` with two decimal places. Schema example: @@ -158,9 +174,10 @@ Schema example: |---|---:| | Input tokens | 120,000 | | cacheReadTokens | 30,000 | -| Input cache rate | 25.0% | +| cacheWriteTokens | 5,000 | | Output tokens | 8,000 | | Total tokens | 128,000 | +| Average AIC | 8.40 | ``` ## Section: Download From 57de8a464c15b376c2bd8edc10d62581fa663363 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Tue, 16 Jun 2026 12:01:01 +0800 Subject: [PATCH 5/9] chore: add uv cache isolation --- evals/microsoft-foundry/eval.yaml | 2 +- tests/utils/agent-runner.ts | 4 +++- tests/vally/vally-executor.ts | 4 ++++ 3 files changed, 8 insertions(+), 2 deletions(-) diff --git a/evals/microsoft-foundry/eval.yaml b/evals/microsoft-foundry/eval.yaml index a6fb124a2..b6603dc2f 100644 --- a/evals/microsoft-foundry/eval.yaml +++ b/evals/microsoft-foundry/eval.yaml @@ -13,7 +13,7 @@ environment: skills: - ../../plugin/skills/microsoft-foundry -config: +defaults: runs: 5 timeout: "30m" executor: integration-test-agent-runner diff --git a/tests/utils/agent-runner.ts b/tests/utils/agent-runner.ts index ed601342b..5a3c7fc8d 100644 --- a/tests/utils/agent-runner.ts +++ b/tests/utils/agent-runner.ts @@ -114,6 +114,7 @@ const modelOverride = process.env.MODEL_OVERRIDE?.trim(); export interface AgentRunConfig { setup?: (workspace: string) => Promise; + env?: Record; model?: string; prompt: string; shouldEarlyTerminate?: (metadata: AgentMetadata) => boolean; @@ -682,7 +683,8 @@ export function useAgentRunner(agentRunnerConfig: AgentRunnerConfig) { cliPath: getBundledCliPath(), env: { ...process.env, - ...envVar + ...envVar, + ...config.env } }) as CopilotClient; entry.client = client; diff --git a/tests/vally/vally-executor.ts b/tests/vally/vally-executor.ts index 4e4a49a63..0292554fe 100644 --- a/tests/vally/vally-executor.ts +++ b/tests/vally/vally-executor.ts @@ -1,5 +1,6 @@ import type { Executor, ExecutorOptions, ExecutorRegistry, Stimulus, Trajectory, TrajectoryEvent } from "@microsoft/vally"; import { computeMetrics } from "@microsoft/vally"; +import * as path from "node:path"; import type { AgentMetadata, AgentRunConfig } from "../utils/agent-runner.ts"; import { useAgentRunner, createMarkdownReport } from "../utils/agent-runner.ts"; import { listSkills } from "../utils/skill-loader.ts"; @@ -34,6 +35,9 @@ export class IntegrationTestAgentRunner implements Executor { const runConfig: AgentRunConfig = { workspace: workDir, + env: { + UV_CACHE_DIR: path.join(workDir, ".uv-cache"), + }, model: model, prompt: stimulus.prompt, shouldEarlyTerminate: shouldEarlyTerminate, From 408527b8121f670305bdaca34ae1320861a7cc97 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Tue, 16 Jun 2026 12:21:16 +0800 Subject: [PATCH 6/9] fix: add resource creation time definition --- tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md index 2ea9de89b..493909095 100644 --- a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md +++ b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md @@ -88,7 +88,10 @@ Main stages: - Eval suite - Final Output -The `Test agent locally` stage includes creating the local virtual environment, installing `uv`, installing project packages from requirements or equivalent package files, starting the local agent server, and invoking the local agent to verify it responds. +Notes: + +- `Foundry resources creation` starts when the AI begins declaring or working on Foundry project/resource creation, `azd provision`, or equivalent `azd provision` tool-call signals, and ends when `azd provision` has fully completed successfully. +- `Test agent locally` includes creating the local virtual environment, installing `uv`, installing project packages from requirements or equivalent package files, starting the local agent server, and invoking the local agent to verify it responds. Single-trial schema example: From a8b41f1a5ee911bfeb9acd10a51ee27f48d70b91 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Mon, 29 Jun 2026 15:36:00 +0800 Subject: [PATCH 7/9] fix: codes --- tests/utils/agent-runner.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/utils/agent-runner.ts b/tests/utils/agent-runner.ts index bd38d699e..58c0905e3 100644 --- a/tests/utils/agent-runner.ts +++ b/tests/utils/agent-runner.ts @@ -883,7 +883,7 @@ export function useAgentRunner(agentRunnerConfig: AgentRunnerConfig) { tools: ["*"] } } - }, + }), systemMessage: runConfig.systemPrompt, // Disable session telemetry so usage of skills and tools by the test agent runner don't end up sending Copilot CLI telemetry. enableSessionTelemetry: false @@ -1279,4 +1279,4 @@ function sanitizeFileName(name: string): string { .replace(/-+/g, "-") // Collapse multiple dashes .replace(/_+/g, "_") // Collapse multiple underscores .substring(0, 200); // Limit length -} +} \ No newline at end of file From 0d857c48b6a68d2e9defda2b6485a904b0af6ac6 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Wed, 1 Jul 2026 11:20:55 +0800 Subject: [PATCH 8/9] chore: update prompt --- ...icrosoft-foundry-e2e-eval-report.prompt.md | 54 ++++++++++++++++--- 1 file changed, 48 insertions(+), 6 deletions(-) diff --git a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md index 493909095..03caccbd5 100644 --- a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md +++ b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md @@ -6,13 +6,13 @@ You are generating the final Markdown report for a Microsoft Foundry E2E Vally e Create the report as a Markdown file at the path specified by the `REPORT_MD` environment variable. Create parent directories if needed. Do not print the report to stdout; write the complete report to `REPORT_MD`. After writing and verifying the file, the final chat response should only say `Report written to ` followed by the report path. -In CI, Vally writes `eval-results.md` and `results.jsonl` under a timestamped subdirectory of `tests/results/`; the CI runner uses the same repository-relative path. Find the latest `results.jsonl` under `tests/results/`, then use the `eval-results.md` in the same directory. Analyze `results.jsonl` directly. Do not invent numbers. +In CI, Vally writes `eval-results.md` and `results.jsonl` under timestamped subdirectories of `tests/results/`; the CI runner uses the same repository-relative path. Find all `results.jsonl` files under `tests/results/`, then use the `eval-results.md` in the same directory as the Sonnet 4.6 baseline for Raw Results. Analyze `results.jsonl` directly. Do not invent numbers. Only read existing result files in the current working directory and write the final Markdown file to `REPORT_MD`. Do not scan sibling repositories or user directories. Do not run Vally, tests, package install commands, deployment commands, `azd`, `git`, or any command that creates a new evaluation run. Do not create or modify any file except `REPORT_MD` and its parent directory. The report file must contain Markdown only. Do not wrap the report in a code fence. Use normal Markdown pipe tables, not terminal box-drawing tables. Do not include preamble, progress narration, raw event logs, or free-form stage notes outside the requested sections. -Only analyze test records in `results.jsonl` where `trajectory.stimulus.name` starts with `Golden Path`. Ignore all other stimuli and ignore the final `run-summary` record. +Only analyze test records in `results.jsonl` where `trajectory.stimulus.name` starts with `Golden Path`. Ignore all other stimuli and ignore the final `run-summary` record. Determine each trial's model from `trajectory.metadata.model` first, then from `trajectory.metrics.tokenUsage.model`, then from the last `token_usage` event with `data.model`. ## Output Report Structure @@ -22,8 +22,9 @@ The output report must use this section order: 2. `## Golden Path Result` 3. `## Golden Path Time Cost` 4. `## Golden Path Token Cost` -5. `## Download` -6. `## Raw Results` +5. `## Golden Path Model Performance` +6. `## Download` +7. `## Raw Results` Do not add other top-level sections. Do not create a `## Links` section. The schemas below show the required shape and example formatting; calculate the actual values from the input files. @@ -34,6 +35,7 @@ Purpose: highlight the Golden Path result first, without mixing in non-Golden Pa Guidance: - Include only Golden Path trials. +- This section is the Sonnet 4.6 baseline: include only Golden Path trials whose model is `claude-sonnet-4.6`. - Report the overall Golden Path outcome as `PASS` only if every Golden Path trial passed; otherwise report `FAIL`. - Add one table row per Golden Path trial, in chronological trial order. - Use `-` in `Notes` for passed trials. For failed trials, keep the note short and based on the eval result. @@ -66,6 +68,7 @@ Purpose: show Golden Path runtime first as an overall average, then as per-run s Guidance: - Runtime is measured from the first `user_message` event to the last `assistant_message` event for each Golden Path trial. +- This section is the Sonnet 4.6 baseline: include only Golden Path trials whose model is `claude-sonnet-4.6`. - `Total average runtime` must equal the average of the per-run `Total` row values. - Each stage duration is the AI-driven full wall-clock time for that stage: start when the AI begins working on that stage, and end when the AI completes that stage and moves to the next stage. Include AI reasoning, command execution, waiting, result inspection, retries, and verification within the stage. - Each run column's stage durations must sum exactly to that run's `Total` row. Assign all elapsed wall-clock time to exactly one stage. @@ -143,14 +146,20 @@ Purpose: show average token usage and AI credit cost for Golden Path trials only ### Pricing Table -Use this GitHub Copilot Anthropic pricing table. Prices are USD per 1M tokens. +Use this GitHub Copilot pricing table. Prices are USD per 1M tokens. The `Model` column uses the Copilot CLI model id. OpenAI and Microsoft models do not have a separate cache write price in GitHub Copilot pricing, so use `N/A` for `Cache write` and treat cache write tokens as regular input tokens when calculating cost. | Model | Input | Cached input | Cache write | Output | |---|---:|---:|---:|---:| -| Claude Sonnet 4.6 | $3.00 | $0.30 | $3.75 | $15.00 | +| claude-opus-4.8 | $5.00 | $0.50 | $6.25 | $25.00 | +| claude-sonnet-4.6 | $3.00 | $0.30 | $3.75 | $15.00 | +| gpt-5.3-codex | $1.75 | $0.175 | N/A | $14.00 | +| gpt-5-mini | $0.25 | $0.025 | N/A | $2.00 | +| mai-code-1-flash | $0.75 | $0.075 | N/A | $4.50 | Guidance: +This section is the Sonnet 4.6 baseline: include only Golden Path trials whose model is `claude-sonnet-4.6`. + Step 1: calculate average token usage. - Calculate token usage directly from `trajectory.metrics.tokenUsage`. @@ -164,6 +173,7 @@ Step 2: calculate average AIC. - `100` AI credits equals `$1.00`. - Determine the model used by each Golden Path trial from `trajectory.metrics.tokenUsage.model`, then use the matching model row in the Pricing Table for `inputRate`, `cachedInputRate`, `cacheWriteRate`, and `outputRate`. - Calculate `averageAic = ((((averageInputTokens - averageCacheReadTokens - averageCacheWriteTokens) * inputRate) + (averageCacheReadTokens * cachedInputRate) + (averageCacheWriteTokens * cacheWriteRate) + (averageOutputTokens * outputRate)) / 1,000,000) * 100`. +- If `Cache write` is `N/A` for the model, calculate `averageAic = ((((averageInputTokens - averageCacheReadTokens) * inputRate) + (averageCacheReadTokens * cachedInputRate) + (averageOutputTokens * outputRate)) / 1,000,000) * 100`. - Format `Average AIC` with two decimal places. Schema example: @@ -183,6 +193,37 @@ Schema example: | Average AIC | 8.40 | ``` +## Section: Golden Path Model Performance + +Purpose: compare Golden Path performance across every model run in the workflow. + +Guidance: + +- Include every Golden Path trial from every `results.jsonl` file found under `tests/results/`. +- Group trials by model. If a model has multiple Golden Path trials across files or repeated `--runs`, average all of that model's Golden Path trials. +- Runtime for each trial is measured from the first `user_message` event to the last `assistant_message` event, using the same timing rule as `## Golden Path Time Cost`. +- Average token usage directly from `trajectory.metrics.tokenUsage`. +- `Avg total tokens` is average `inputTokens` plus average `outputTokens`. +- Calculate `Avg AIC` using the Pricing Table and the same cost formula as `## Golden Path Token Cost`. +- Round token counts to integers and format them with thousands separators. +- Format average runtime in `x min Y s` format. +- Order rows by the `VALLY_MODELS` environment variable if it is available. Otherwise, order rows by the first chronological appearance of each model in the result files. +- Use `N/A` for unavailable metrics or missing pricing rows. Format `Avg AIC` with two decimal places when it is available. + +Schema example: + +```markdown +## Golden Path Model Performance + +N models analyzed. + +| Model | Golden Path trials | Passed | Failed | Avg total runtime | Avg input tokens | Avg cacheReadTokens | Avg cacheWriteTokens | Avg output tokens | Avg total tokens | Avg AIC | +|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| +| claude-opus-4.8 | 1 | 1 | 0 | 18 min 22 s | 130,000 | 40,000 | 6,000 | 9,000 | 139,000 | 70.25 | +| claude-sonnet-4.6 | 1 | 1 | 0 | 15 min 43 s | 120,000 | 30,000 | 5,000 | 8,000 | 128,000 | 40.28 | +| gpt-5.3-codex | 1 | 1 | 0 | 17 min 9 s | 118,000 | 32,000 | 5,500 | 7,500 | 125,500 | 26.11 | +``` + ## Section: Download Purpose: provide the Vally artifact download link using the existing workflow style. @@ -209,6 +250,7 @@ Guidance: - Include the full `eval-results.md` content. - Keep its table content intact. +- Use the `eval-results.md` in the same directory as the latest `claude-sonnet-4.6` Golden Path result. If no Sonnet 4.6 result exists, use the latest `eval-results.md` under `tests/results/`. - If the source content starts with `## Eval Results`, omit that source heading so `## Raw Results` remains the final top-level report section. Schema example: From 99272c1fdb817e6fdb5b3d8adc3820a6c903a316 Mon Sep 17 00:00:00 2001 From: Chenyi An Date: Wed, 1 Jul 2026 14:24:14 +0800 Subject: [PATCH 9/9] chore: update prompt --- .../microsoft-foundry-e2e-eval-report.prompt.md | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md index 03caccbd5..e322fded4 100644 --- a/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md +++ b/tests/prompts/microsoft-foundry-e2e-eval-report.prompt.md @@ -202,10 +202,10 @@ Guidance: - Include every Golden Path trial from every `results.jsonl` file found under `tests/results/`. - Group trials by model. If a model has multiple Golden Path trials across files or repeated `--runs`, average all of that model's Golden Path trials. - Runtime for each trial is measured from the first `user_message` event to the last `assistant_message` event, using the same timing rule as `## Golden Path Time Cost`. -- Average token usage directly from `trajectory.metrics.tokenUsage`. +- Use token usage from `trajectory.metrics.tokenUsage` to calculate `Avg total tokens` and `Avg AIC`; do not include input/cache/output token detail columns in this table. - `Avg total tokens` is average `inputTokens` plus average `outputTokens`. - Calculate `Avg AIC` using the Pricing Table and the same cost formula as `## Golden Path Token Cost`. -- Round token counts to integers and format them with thousands separators. +- Round `Avg total tokens` to an integer and format it with thousands separators. - Format average runtime in `x min Y s` format. - Order rows by the `VALLY_MODELS` environment variable if it is available. Otherwise, order rows by the first chronological appearance of each model in the result files. - Use `N/A` for unavailable metrics or missing pricing rows. Format `Avg AIC` with two decimal places when it is available. @@ -217,11 +217,11 @@ Schema example: N models analyzed. -| Model | Golden Path trials | Passed | Failed | Avg total runtime | Avg input tokens | Avg cacheReadTokens | Avg cacheWriteTokens | Avg output tokens | Avg total tokens | Avg AIC | -|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| -| claude-opus-4.8 | 1 | 1 | 0 | 18 min 22 s | 130,000 | 40,000 | 6,000 | 9,000 | 139,000 | 70.25 | -| claude-sonnet-4.6 | 1 | 1 | 0 | 15 min 43 s | 120,000 | 30,000 | 5,000 | 8,000 | 128,000 | 40.28 | -| gpt-5.3-codex | 1 | 1 | 0 | 17 min 9 s | 118,000 | 32,000 | 5,500 | 7,500 | 125,500 | 26.11 | +| Model | Avg total tokens | Avg time cost | Avg AIC | +|---|---:|---:|---:| +| claude-opus-4.8 | 139,000 | 18 min 22 s | 70.25 | +| claude-sonnet-4.6 | 128,000 | 15 min 43 s | 40.28 | +| gpt-5.3-codex | 125,500 | 17 min 9 s | 26.11 | ``` ## Section: Download