/*--------------------------------------------------------------------------------------------- * Copyright (c) Microsoft Corporation. All rights reserved. *--------------------------------------------------------------------------------------------*/ import { appendFileSync } from "fs"; import { mkdir, readFile, writeFile } from "fs/promises"; import type { ChatCompletion, ChatCompletionChunk, ChatCompletionCreateParamsBase, ChatCompletionMessageFunctionToolCall, ChatCompletionMessageParam, } from "openai/resources/chat/completions"; import { ChatCompletionStream } from "openai/resources/chat/completions"; import path from "path"; import yaml from "yaml"; import { CapturedExchange, CapturingHttpProxy, PerformRequestOptions, } from "./capturingHttpProxy"; export type { CapturedRequest } from "./capturingHttpProxy"; import { anthropicMessagesEndpoint, anthropicMessagesRequestToChatCompletion, chatCompletionResponseToAnthropicMessage, chatCompletionResponseToAnthropicSseChunks, } from "./anthropicMessagesAdapter"; import { canonicalUserMessageSeparator } from "./modelProtocolAdapterShared"; import { chatCompletionResponseToResponsesApiMessage, chatCompletionResponseToResponsesApiSseChunks, responsesApiRequestToChatCompletion, responsesEndpoint, } from "./responsesApiAdapter"; import { iife, ShellConfig, sleep } from "./util"; export const workingDirPlaceholder = "${workdir}"; const chatCompletionEndpoint = "/chat/completions"; export type ReplayBackend = | "capi" | "anthropic-messages" | "openai-responses" | "openai-completions"; type ReplayProtocol = { endpoint: string; normalizeRequest?: (body: string) => string; responseBody?: (response: ChatCompletion) => unknown; responseChunks: (response: ChatCompletion) => string[]; responseEndChunk?: string; errorBody?: (code: string | undefined, message: string) => unknown; canonicalResponse?: boolean; }; const chatCompletionsProtocol = { endpoint: chatCompletionEndpoint, responseChunks: (response) => convertToStreamingResponseChunks(response).map( (chunk) => `data: ${JSON.stringify(chunk)}\n\n`, ), responseEndChunk: "data: [DONE]\n\n", canonicalResponse: true, } satisfies ReplayProtocol; const replayProtocols: Record = { capi: chatCompletionsProtocol, "openai-completions": { ...chatCompletionsProtocol, normalizeRequest: coalesceAdjacentUserMessages, }, "anthropic-messages": { endpoint: anthropicMessagesEndpoint, normalizeRequest: (body) => coalesceAdjacentUserMessages( anthropicMessagesRequestToChatCompletion(body), ), responseBody: chatCompletionResponseToAnthropicMessage, responseChunks: chatCompletionResponseToAnthropicSseChunks, errorBody: (code, message) => { const type = code ?? "rate_limited"; return { type: "error", error: { type, message } }; }, }, "openai-responses": { endpoint: responsesEndpoint, normalizeRequest: (body) => coalesceAdjacentUserMessages(responsesApiRequestToChatCompletion(body)), responseBody: chatCompletionResponseToResponsesApiMessage, responseChunks: chatCompletionResponseToResponsesApiSseChunks, }, }; const modelEndpoints = new Set( Object.values(replayProtocols).map((protocol) => protocol.endpoint), ); const shellConfig = process.platform === "win32" ? ShellConfig.powerShell : ShellConfig.bash; const normalizedToolNames: Record = { [shellConfig.shellToolName]: "${shell}", [shellConfig.readShellToolName]: "${read_shell}", [shellConfig.writeShellToolName]: "${write_shell}", }; /** * Default model to use when no stored data is available for a given test. * This enables responding to /models without needing to have a capture file. */ const defaultModel = "claude-sonnet-5"; /** * An HTTP proxy that not only captures HTTP exchanges, but also stores them in a file on disk and * replays the stored responses on subsequent runs. * * This only stores and matches CAPI-provided OpenAI chat completions, not arbitrary HTTP traffic, since * the core idea is to store and compare in a normalized form that (1) ignores irrelevant differences (like * timestamps, or references to your working directory path) and (2) writes data files in a simple, * human-readable format where it's easy to reason about diffs when things change. * * To avoid leaving stale files around as tests are modified, it stores things on a one-file-per-test basis, * which is overwritten on each test run. So for as long as a test exists, its data will be kept up-to-date. */ export class ReplayingCapiProxy extends CapturingHttpProxy { private state: ReplayingCapiProxyState | null = null; private startPromise: Promise | null = null; private defaultToolResultNormalizers: ToolResultNormalizer[] = [ { toolName: "*", normalizer: normalizeLargeOutputFilepaths }, { toolName: "*", normalizer: normalizeInterruptedToolResult }, { toolName: "${shell}", normalizer: normalizeShellExitMarkers }, { toolName: "*", normalizer: normalizeGhAuthMessages }, { toolName: "*", normalizer: normalizeAvailableToolNames }, { toolName: "*", normalizer: normalizeBackgroundAgentStartMessage }, { toolName: "read_agent", normalizer: normalizeReadAgentResult }, ]; /** * Per-token responses for `/copilot_internal/user` endpoint. * Key is the Bearer token (without "Bearer " prefix), value is the response body. * When a request arrives with `Authorization: Bearer `, the matching response is returned. * If no match is found, a 401 Unauthorized response is returned. */ private copilotUserByToken = new Map(); /** * If true, cached responses are played back slowly (~ 2KiB/sec). Otherwise streaming responses are sent as fast as possible. */ slowStreaming = false; onStopRequested?: (skipWritingCache: boolean) => Promise | void; constructor( targetUrl: string, filePath?: string, workDir?: string, testInfo?: { file: string; line?: number }, ) { super(targetUrl); // If the instantiator wants to supply config up front as ctor params, we can // skip the need to do a /config POST before other requests. This only makes // sense if the config will be static for the lifetime of the proxy. if (filePath && workDir) { this.state = { filePath, workDir, testInfo, backend: "capi", toolResultNormalizers: [...this.defaultToolResultNormalizers], }; } } async start(): Promise { return (this.startPromise ??= (async () => { await this.loadStoredData(); return super.start(); })()); } async updateConfig(config: Partial): Promise { if (!config.filePath || !config.workDir) { throw new Error("filePath and workDir must be provided in config"); } // Since we're about to switch to a new file, write out any captured exchanges // Note that the final call to stop() will also write out any remaining exchanges. // In CI mode (GITHUB_ACTIONS=true) we never write — the snapshots are read-only. // Otherwise tests that exercise only a subset of a multi-conversation snapshot // would silently overwrite the file with that subset, breaking subsequent runs. if ( this.state?.backend === "capi" && process.env.GITHUB_ACTIONS !== "true" ) { await writeCapturesToDisk(this.exchanges, this.state); } this.state = { filePath: config.filePath, workDir: config.workDir, testInfo: config.testInfo, backend: parseReplayBackend(config.backend), toolResultNormalizers: [...this.defaultToolResultNormalizers], }; this.clearExchanges(); await this.loadStoredData(); } private async loadStoredData(): Promise { if (!this.state) { return; } // Read directly and treat a missing file as "nothing stored" rather than // checking for existence first, which would be a check-then-use race. const content = await readFile(this.state.filePath, "utf-8").catch( (error: NodeJS.ErrnoException) => { if (error.code === "ENOENT") { return undefined; } throw error; }, ); if (content === undefined) { return; } this.state.storedData = yaml.parse(content) as NormalizedData; normalizeToolResultOrder(this.state.storedData.conversations); normalizeStoredUserMessages(this.state.storedData.conversations); normalizeStoredToolMessages(this.state.storedData.conversations); normalizeStoredMessagesForBackend( this.state.storedData.conversations, this.state.backend, ); } async stop(skipWritingCache?: boolean): Promise { await super.stop(); // CAPI is the authoritative capture path. BYOK modes only verify that the // same canonical snapshots replay through each provider protocol. if ( this.state?.backend === "capi" && !skipWritingCache && process.env.GITHUB_ACTIONS !== "true" ) { await writeCapturesToDisk(this.exchanges, this.state); } } addToolResultNormalizer( toolName: string, normalizer: (result: string) => string, ) { if (!this.state) { throw new Error( "ReplayingCapiProxy not yet initialized. Cannot add tool result normalizer.", ); } this.state.toolResultNormalizers.push({ toolName, normalizer }); } /** * Register a per-token response for the `/copilot_internal/user` endpoint. * When a request with `Authorization: Bearer ` arrives, the matching response is returned. */ setCopilotUserByToken(token: string, response: CopilotUserResponse): void { this.copilotUserByToken.set(token, response); } override performRequest(options: PerformRequestOptions): void { void iife(async () => { const commonResponseHeaders = { "x-github-request-id": "some-request-id", }; try { // Handle /copilot-user-config endpoint for configuring per-token user responses if ( options.requestOptions.path === "/copilot-user-config" && options.requestOptions.method === "POST" ) { const config = JSON.parse(options.body!) as { token: string; response: CopilotUserResponse; }; this.copilotUserByToken.set(config.token, config.response); options.onResponseStart(200, {}); options.onResponseEnd(); return; } // Handle /config endpoint for updating proxy configuration if ( options.requestOptions.path === "/config" && options.requestOptions.method === "POST" ) { await this.updateConfig(JSON.parse(options.body!)); options.onResponseStart(200, {}); options.onResponseEnd(); return; } // Handle /stop endpoint for stopping the proxy if ( options.requestOptions.path?.startsWith("/stop") && options.requestOptions.method === "POST" ) { const skipWritingCache = options.requestOptions.path.includes( "skipWritingCache=true", ); options.onResponseStart(200, {}); options.onResponseEnd(); await this.onStopRequested?.(skipWritingCache); await this.stop(skipWritingCache); process.exit(0); } // Handle /exchanges endpoint for retrieving captured exchanges if ( options.requestOptions.path === "/exchanges" && options.requestOptions.method === "GET" ) { const protocol = replayProtocols[this.state?.backend ?? "capi"]; const parsedExchanges = await Promise.all( this.exchanges .filter((exchange) => exchange.request.url === protocol.endpoint) .map((exchange) => parseHttpExchange( protocol.normalizeRequest?.(exchange.request.body) ?? exchange.request.body, protocol.canonicalResponse ? exchange.response?.body : undefined, exchange.request.headers, ), ), ); options.onResponseStart(200, {}); options.onData(Buffer.from(JSON.stringify(parsedExchanges))); options.onResponseEnd(); return; } // Handle /requests endpoint for retrieving all captured outbound requests. if ( options.requestOptions.path === "/requests" && options.requestOptions.method === "GET" ) { const requests = this.exchanges .map((exchange) => exchange.request) .filter((request) => request.url !== "/requests"); options.onResponseStart(200, { "content-type": "application/json" }); options.onData(Buffer.from(JSON.stringify(requests))); options.onResponseEnd(); return; } // Handle /copilot_internal/user endpoint for per-session auth. // This must run before the state guard below: the CLI authenticates and // calls /copilot_internal/user at startup, which can race ahead of the // per-test POST /config (e.g. the Go harness spawns the CLI before the // first ConfigureForTest). The response only depends on the token map, // which is populated independently of `state`. if (options.requestOptions.path === "/copilot_internal/user") { const headers = options.requestOptions.headers; const headerMap = headers as | Record | undefined; const rawAuthHeader = Array.isArray(headers) ? undefined : (headerMap?.authorization ?? headerMap?.Authorization); const authHeader = Array.isArray(rawAuthHeader) ? rawAuthHeader[0] : typeof rawAuthHeader === "string" ? rawAuthHeader : undefined; const token = authHeader?.replace("Bearer ", ""); const registered = token ? this.copilotUserByToken.get(token) : undefined; // The CLI gates third-party MCP servers behind the copilot user's // `is_mcp_enabled` flag (a null/missing value disables them). Default // it to true so e2e MCP servers are enabled unless a test opts out. const userResponse = registered ? ({ is_mcp_enabled: true, ...registered } as CopilotUserResponse) : undefined; if (userResponse) { const headers = { "content-type": "application/json", ...commonResponseHeaders, }; options.onResponseStart(200, headers); options.onData(Buffer.from(JSON.stringify(userResponse))); options.onResponseEnd(); } else { options.onResponseStart(401, commonResponseHeaders); options.onData( Buffer.from(JSON.stringify({ message: "Bad credentials" })), ); options.onResponseEnd(); } return; } const state = this.state; if (!state) { throw new Error( "ReplayingCapiProxy not yet initialized. Either pass filePath and workDir to the constructor, " + "or post configuration to /config before making other HTTP requests.", ); } // Handle /models endpoint // Use stored models if available, otherwise use default model if (options.requestOptions.path === "/models") { const models = state.storedData?.models && state.storedData.models.length > 0 ? state.storedData.models : [defaultModel]; const modelsResponse = createGetModelsResponse(models); const body = JSON.stringify(modelsResponse); const headers = { "content-type": "application/json", ...commonResponseHeaders, }; options.onResponseStart(200, headers); options.onData(Buffer.from(body)); options.onResponseEnd(); return; } // Keep GitHub MCP tests hermetic while still capturing the request at // the CAPI proxy. The tests only need a successful transport handshake; // no fake tools are exposed. if (options.requestOptions.path === "/mcp") { if (options.requestOptions.method !== "POST") { options.onResponseStart(200, commonResponseHeaders); options.onResponseEnd(); return; } const request = JSON.parse(options.body ?? "{}") as { id?: string | number; method?: string; params?: { protocolVersion?: string }; }; if (request.id === undefined) { options.onResponseStart(202, commonResponseHeaders); options.onResponseEnd(); return; } const result = request.method === "initialize" ? { protocolVersion: request.params?.protocolVersion ?? "2025-03-26", capabilities: { tools: {} }, serverInfo: { name: "e2e-github-mcp", version: "1.0.0" }, } : request.method === "tools/list" ? { tools: [] } : {}; options.onResponseStart(200, { "content-type": "application/json", ...commonResponseHeaders, }); options.onData( Buffer.from( JSON.stringify({ jsonrpc: "2.0", id: request.id, result }), ), ); options.onResponseEnd(); return; } // Handle memory endpoints - return stub responses in tests // Matches: /agents/*/memory/*/enabled, /agents/*/memory/*/recent, etc. if (options.requestOptions.path?.match(/\/agents\/.*\/memory\//)) { let body: string; if (options.requestOptions.path.includes("/enabled")) { body = JSON.stringify({ enabled: false }); } else if (options.requestOptions.path.includes("/recent")) { body = JSON.stringify({ memories: [] }); } else { body = JSON.stringify({}); } const headers = { "content-type": "application/json", ...commonResponseHeaders, }; options.onResponseStart(200, headers); options.onData(Buffer.from(body)); options.onResponseEnd(); return; } const requestPath = options.requestOptions.path ?? ""; const protocol = replayProtocols[state.backend]; if ( modelEndpoints.has(requestPath) && state.backend !== "capi" && requestPath !== protocol.endpoint ) { const message = `Expected ${protocol.endpoint} for backend ${state.backend}, received ${requestPath}`; options.onResponseStart(400, { "content-type": "application/json", ...commonResponseHeaders, }); options.onData( Buffer.from( JSON.stringify({ error: { type: "protocol_mismatch", message }, }), ), ); options.onResponseEnd(); return; } const isModelRequest = requestPath === protocol.endpoint; // Every protocol enters the existing Chat Completions snapshot matcher. const normalizedBody = isModelRequest && options.body ? (protocol.normalizeRequest?.(options.body) ?? options.body) : options.body; if (state.storedData && isModelRequest && normalizedBody) { const streamingIsRequested = (JSON.parse(normalizedBody) as { stream?: boolean }).stream === true; const savedError = await findSavedChatCompletionError( state.storedData, normalizedBody, state.workDir, state.toolResultNormalizers, ); if (savedError) { const headers = { "content-type": "application/json", ...commonResponseHeaders, ...(savedError.retryAfterSeconds !== undefined ? { "retry-after": String(savedError.retryAfterSeconds) } : {}), }; options.onResponseStart(savedError.status, headers); options.onData( Buffer.from( JSON.stringify( (protocol.errorBody ?? openAIErrorBody)( savedError.code, savedError.message ?? "Rate limited by test snapshot", ), ), ), ); options.onResponseEnd(); return; } const savedResponse = await findSavedChatCompletionResponse( state.storedData, normalizedBody, state.workDir, state.toolResultNormalizers, ); if (savedResponse) { await this.respondWithProtocol( options, protocol, savedResponse, streamingIsRequested, commonResponseHeaders, ); return; } // Check if this request matches a snapshot with no response (e.g., timeout tests). // If so, hang forever so the client-side timeout can trigger. if ( await isRequestOnlySnapshot( state.storedData, normalizedBody, state.workDir, state.toolResultNormalizers, ) ) { const headers = { "content-type": streamingIsRequested ? "text/event-stream" : "application/json", ...commonResponseHeaders, }; options.onResponseStart(200, headers); // Never call onResponseEnd - hang indefinitely for timeout tests. // Returning here keeps the HTTP response open without leaking a pending Promise. return; } } // Beyond this point, we're only going to be able to supply responses in CI if we have a snapshot, // and we only store snapshots for chat completion. For anything else (e.g., custom-agents fetches), // return 404 so the CLI treats them as unavailable instead of erroring. if (!isModelRequest) { const headers = { "content-type": "application/json", "x-github-request-id": "proxy-not-found", }; options.onResponseStart(404, headers); options.onData( Buffer.from(JSON.stringify({ error: "Not found by test proxy" })), ); options.onResponseEnd(); return; } // Fallback to normal proxying if no cached response found // This implicitly captures the new exchange too const isCI = process.env.GITHUB_ACTIONS === "true"; if (isCI || state.backend !== "capi") { await exitWithNoMatchingRequestError( options, state.testInfo, state.workDir, state.toolResultNormalizers, state.storedData, normalizedBody, ); return; } super.performRequest(options); } catch (err) { options.onError(err as Error | string); } }); } private async respondWithProtocol( options: PerformRequestOptions, protocol: ReplayProtocol, response: ChatCompletion, streaming: boolean, commonHeaders: Record, ): Promise { if (!streaming) { options.onResponseStart(200, { "content-type": "application/json", ...commonHeaders, }); options.onData( Buffer.from( JSON.stringify(protocol.responseBody?.(response) ?? response), ), ); options.onResponseEnd(); return; } options.onResponseStart(200, { "content-type": "text/event-stream", ...commonHeaders, }); for (const chunk of protocol.responseChunks(response)) { options.onData(Buffer.from(chunk)); if (this.slowStreaming) { await sleep(100); } } if (protocol.responseEndChunk) { options.onData(Buffer.from(protocol.responseEndChunk)); } options.onResponseEnd(); } } async function writeCapturesToDisk( exchanges: readonly CapturedExchange[], state: ReplayingCapiProxyState, ) { const data = await transformHttpExchanges( exchanges, state.workDir, state.toolResultNormalizers, ); const preservedErrors = state.storedData?.errors; if (preservedErrors && preservedErrors.length > 0) { data.errors = preservedErrors; data.models = [ ...new Set([ ...(state.storedData?.models ?? []), ...data.models, ...preservedErrors .map((error) => error.model) .filter((model): model is string => model !== undefined), ]), ]; } if (data.conversations.length > 0) { let yamlText = yaml.stringify(data, { lineWidth: 120 }); // We have to normalize line endings explicitly, because yaml.stringify uses Unix-style even on Windows, // and Git will restore the files with CRLF on Windows so they will appear to be changed if (process.platform === "win32") { yamlText = yamlText.replace(/\r?\n/g, "\r\n"); } await mkdir(path.dirname(state.filePath), { recursive: true }); await writeFileIfDifferent(state.filePath, yamlText); } } /** * Produces a human-readable explanation of why no stored conversation matched * a given request. For each stored conversation it reports the first reason * matching failed, mirroring the logic in {@link findAssistantIndexAfterPrefix}. */ function diagnoseMatchFailure( requestMessages: NormalizedMessage[], rawMessages: unknown[], storedData: NormalizedData | undefined, ): string { const lines: string[] = []; lines.push( `Request has ${requestMessages.length} normalized messages (${rawMessages.length} raw).`, ); if (!storedData || storedData.conversations.length === 0) { lines.push("No stored conversations to match against."); return lines.join("\n"); } for (let c = 0; c < storedData.conversations.length; c++) { const saved = storedData.conversations[c].messages; // Same check as findAssistantIndexAfterPrefix: request must be a strict prefix if (requestMessages.length >= saved.length) { lines.push( `Conversation ${c} (${saved.length} messages): ` + `skipped — request has ${requestMessages.length} messages, need fewer than ${saved.length}.`, ); continue; } // Find the first message that doesn't match let mismatchIndex = -1; for (let i = 0; i < requestMessages.length; i++) { if (JSON.stringify(requestMessages[i]) !== JSON.stringify(saved[i])) { mismatchIndex = i; break; } } if (mismatchIndex >= 0) { const raw = mismatchIndex < rawMessages.length ? JSON.stringify(rawMessages[mismatchIndex]).slice(0, 300) : "(no raw message)"; lines.push( `Conversation ${c} (${saved.length} messages): mismatch at message ${mismatchIndex}:`, ` request: ${JSON.stringify(requestMessages[mismatchIndex]).slice(0, 200)}`, ` saved: ${JSON.stringify(saved[mismatchIndex]).slice(0, 200)}`, ` raw (pre-normalization): ${raw}`, ); } else { // Prefix matched, but the next saved message isn't an assistant turn const nextRole = saved[requestMessages.length]?.role ?? "(end of conversation)"; lines.push( `Conversation ${c} (${saved.length} messages): ` + `prefix matched, but next saved message is "${nextRole}" (need "assistant").`, ); } } return lines.join("\n"); } async function exitWithNoMatchingRequestError( options: PerformRequestOptions, testInfo: { file: string; line?: number } | undefined, workDir: string, toolResultNormalizers: ToolResultNormalizer[], storedData?: NormalizedData, requestBody?: string, ) { let diagnostics: string; try { const normalized = await parseAndNormalizeRequest( requestBody ?? options.body, workDir, toolResultNormalizers, ); const requestMessages = normalized.conversations[0]?.messages ?? []; let rawMessages: unknown[] = []; try { rawMessages = ( JSON.parse(requestBody ?? options.body ?? "{}") as { messages?: unknown[]; } ).messages ?? []; } catch { /* non-JSON body */ } diagnostics = diagnoseMatchFailure( requestMessages, rawMessages, storedData, ); } catch (e) { diagnostics = `(unable to parse request for diagnostics: ${e})`; } const errorMessage = `No cached response found for ${options.requestOptions.method} ${options.requestOptions.path}.\n${diagnostics}`; // Format as GitHub Actions annotation when test location is available const annotation = [ testInfo?.file ? `file=${testInfo.file}` : "", typeof testInfo?.line === "number" ? `line=${testInfo.line}` : "", ] .filter(Boolean) .join(","); process.stderr.write( `::error${annotation ? ` ${annotation}` : ""}::${errorMessage}\n`, ); options.onError(new Error(errorMessage)); } async function findSavedChatCompletionResponse( storedData: NormalizedData, requestBody: string | undefined, workDir: string, toolResultNormalizers: ToolResultNormalizer[], ): Promise { // Normalize the incoming request the same way we normalize for caching const normalized = await parseAndNormalizeRequest( requestBody, workDir, toolResultNormalizers, ); const requestMessages = normalized.conversations[0]?.messages ?? []; const requestModel = normalized.models[0]; if (!requestModel) { throw new Error("Unable to determine model from request"); } const backgroundAgentIds = extractBackgroundAgentIds(requestBody); // Now find a matching cached conversation (i.e., one for which this request is a prefix) for (const conversation of storedData.conversations) { const replyIndex = findAssistantIndexAfterPrefix( requestMessages, conversation.messages, ); if (replyIndex !== undefined) { return createOpenAIResponse( requestModel, conversation.messages, replyIndex, workDir, backgroundAgentIds, ); } } return undefined; } async function findSavedChatCompletionError( storedData: NormalizedData, requestBody: string | undefined, workDir: string, toolResultNormalizers: ToolResultNormalizer[], ): Promise { const normalized = await parseAndNormalizeRequest( requestBody, workDir, toolResultNormalizers, ); const requestMessages = normalized.conversations[0]?.messages ?? []; const requestModel = normalized.models[0]; for (const error of storedData.errors ?? []) { if (error.model && error.model !== requestModel) { continue; } if ( requestMessages.length === error.messages.length && requestMessages.every( (msg, i) => JSON.stringify(msg) === JSON.stringify(error.messages[i]), ) ) { return error; } } return undefined; } // Checks if the request matches a snapshot that has no assistant response. // This handles timeout test scenarios where the snapshot only records the request. async function isRequestOnlySnapshot( storedData: NormalizedData, requestBody: string | undefined, workDir: string, toolResultNormalizers: ToolResultNormalizer[], ): Promise { const normalized = await parseAndNormalizeRequest( requestBody, workDir, toolResultNormalizers, ); const requestMessages = normalized.conversations[0]?.messages ?? []; for (const conversation of storedData.conversations) { if ( requestMessages.length === conversation.messages.length && requestMessages.every( (msg, i) => JSON.stringify(msg) === JSON.stringify(conversation.messages[i]), ) ) { return true; } } return false; } async function parseAndNormalizeRequest( requestBody: string | undefined, workDir: string, toolResultNormalizers: ToolResultNormalizer[], ) { const fakeRequest = { request: { url: chatCompletionEndpoint, body: requestBody }, } as CapturedExchange; return await transformHttpExchanges( [fakeRequest], workDir, toolResultNormalizers, ); } // Takes raw HTTP traffic and turns it into the normalized form that we store on disk async function transformHttpExchanges( httpExchanges: readonly CapturedExchange[], workDir: string, toolResultNormalizers: ToolResultNormalizer[], ): Promise { const chatCompletionExchanges = httpExchanges .filter((e) => e.request.url === chatCompletionEndpoint) .filter(excludeFailedResponses); const allTurns = await Promise.all( chatCompletionExchanges.map((e) => transformHttpExchange(e.request.body, e.response?.body), ), ); const dedupedExchanges = removePrefixConversations( allTurns.map((t) => t.conversation), ); const dedupedModels = new Set( allTurns.map((t) => t.model ?? "").filter((m) => !!m), ); normalizeToolCalls(dedupedExchanges, toolResultNormalizers); normalizeToolResultOrder(dedupedExchanges); normalizeFilenames(dedupedExchanges, workDir); return { models: Array.from(dedupedModels), conversations: dedupedExchanges }; } function parseReplayBackend(value: unknown): ReplayBackend { if (value === undefined || value === null || value === "") return "capi"; if (typeof value === "string" && Object.hasOwn(replayProtocols, value)) { return value as ReplayBackend; } throw new Error(`Unsupported replay backend: ${String(value)}`); } function coalesceAdjacentUserMessages(requestBody: string): string { const request = JSON.parse(requestBody) as { messages?: Array<{ role?: string; content?: unknown; [key: string]: unknown; }>; }; if (!request.messages) return requestBody; const messages: NonNullable = []; for (const message of request.messages) { const previous = messages.at(-1); if ( previous?.role === "user" && message.role === "user" && typeof previous.content === "string" && typeof message.content === "string" ) { previous.content = `${previous.content.trimEnd()}${canonicalUserMessageSeparator}${message.content.trimStart()}`; } else { messages.push(message); } } for (const message of messages) { if (message.role === "user" && typeof message.content === "string") { message.content = normalizeUserMessage(message.content).replace( /\n{5,}/g, canonicalUserMessageSeparator, ); } } request.messages = messages; return JSON.stringify(request); } function openAIErrorBody( code: string | undefined, message: string, ): unknown { const type = code ?? "rate_limited"; return { error: { message, type, code: type } }; } function normalizeFilenames( conversations: NormalizedConversation[], workDir: string, ): void { // Replace occurrences of the workDir path with workingDirPlaceholder to avoid diffs due to different test run locations // We do so case-insensitively and with both / and \ to cover different OSes // We also normalize any slashes in the rest of the path (e.g., C:\my\workdir\path\to\file.txt -> ${workdir}/path/to/file.txt) workDir = workDir.replace(/\\/g, "/").replace(/\/+$/, ""); const escaped = workDir.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); const workDirPattern = new RegExp( escaped.replace(/\//g, "[\\\\/]+") + "([\\\\/]+[^\\s\"'`,]*)?", "gi", ); const workDirReplacer = (_: string, rest?: string) => workingDirPlaceholder + (rest?.replace(/[\\/]+/g, "/") ?? ""); // Match non-rooted Windows paths like abc\def\something.ext and flip slashes to / // We don't need to match absolute paths because the only legit ones should be inside workdir which // is handled above. Plus there's nothing we could do to normalize them since we don't know their base. const windowsFnPattern = /(? path.replace(/\\/g, "/"); for (const conv of conversations) { for (const msg of conv.messages) { if (msg.content) { msg.content = msg.content.replace(workDirPattern, workDirReplacer); msg.content = msg.content.replace(windowsFnPattern, windowsFnReplacer); } for (const tc of msg.tool_calls ?? []) { if (tc.function?.arguments) { tc.function.arguments = tc.function.arguments.replace( workDirPattern, workDirReplacer, ); tc.function.arguments = tc.function.arguments.replace( windowsFnPattern, windowsFnReplacer, ); } } } } } function normalizeToolCalls( conversations: NormalizedConversation[], resultNormalizers: ToolResultNormalizer[], ) { // We normalize: // - Tool call IDs (mapping from tooluse_rjaaFdJRRhqAZevU_1aBSA etc to toolcall_0, toolcall_1, etc) // - Tool names (e.g., bash/powershell -> ${shell}) // - Tool call results that may vary between execution environments // This is so that we're not storing random or environment-specific data in snapshots, and so we can // still match cached responses even if these details change. for (const conv of conversations) { const idMap = new Map(); const backgroundAgentNamesByToolCallId = new Map(); const backgroundAgentNamesByRuntimeId = new Map(); const precedingMessages: NormalizedMessage[] = []; let counter = 0; let unnamedBackgroundAgentCounter = 0; for (const msg of conv.messages) { for (const tc of msg.tool_calls ?? []) { // Normalize ID in tool calls idMap.set(tc.id, (tc.id = idMap.get(tc.id) ?? `toolcall_${counter++}`)); // Normalize name const originalToolName = tc.function?.name; const normalizedToolName = originalToolName && normalizedToolNames[originalToolName]; if (normalizedToolName) { tc.function!.name = normalizedToolName; } if (tc.function?.name === "task") { const configuredName = getBackgroundAgentName( tc.function.arguments, ); const fallbackName = unnamedBackgroundAgentCounter === 0 ? "background-agent" : `background-agent-${unnamedBackgroundAgentCounter}`; backgroundAgentNamesByToolCallId.set( tc.id, configuredName ?? fallbackName, ); unnamedBackgroundAgentCounter++; } else if ( tc.function?.name === "read_agent" || tc.function?.name === "write_agent" ) { tc.function.arguments = normalizeBackgroundAgentArguments( tc.function.arguments, backgroundAgentNamesByRuntimeId, ); } } if (msg.role === "tool" && msg.tool_call_id) { // Normalize ID in tool results msg.tool_call_id = idMap.get(msg.tool_call_id) ?? msg.tool_call_id; // Normalize result if (msg.content) { const precedingToolCall = precedingMessages .flatMap((m) => m.tool_calls ?? []) .find((tc) => tc.id === msg.tool_call_id); if (precedingToolCall) { if (precedingToolCall.function?.name === "task") { const runtimeId = getBackgroundAgentId(msg.content); const stableName = backgroundAgentNamesByToolCallId.get( msg.tool_call_id, ); if (runtimeId && stableName) { backgroundAgentNamesByRuntimeId.set(runtimeId, stableName); msg.content = normalizeBackgroundAgentStartMessageWithId( msg.content, stableName, ); } } for (const normalizer of resultNormalizers) { if ( precedingToolCall.function?.name === normalizer.toolName || normalizer.toolName === "*" ) { msg.content = normalizer.normalizer(msg.content); } } msg.content = replaceBackgroundAgentIds( msg.content, backgroundAgentNamesByRuntimeId, ); } } } precedingMessages.push(msg); } } } function normalizeToolResultOrder(conversations: NormalizedConversation[]) { for (const conv of conversations) { for (let start = 0; start < conv.messages.length; ) { if (conv.messages[start].role !== "tool") { start++; continue; } let end = start + 1; while (end < conv.messages.length && conv.messages[end].role === "tool") { end++; } conv.messages .slice(start, end) .sort(compareToolResultMessages) .forEach((message, index) => { conv.messages[start + index] = message; }); start = end; } } } function compareToolResultMessages( left: NormalizedMessage, right: NormalizedMessage, ) { return compareToolCallIds(left.tool_call_id, right.tool_call_id); } function compareToolCallIds(left?: string, right?: string) { const leftNumber = parseNormalizedToolCallId(left); const rightNumber = parseNormalizedToolCallId(right); if (leftNumber !== undefined && rightNumber !== undefined) { return leftNumber - rightNumber; } return (left ?? "").localeCompare(right ?? ""); } function parseNormalizedToolCallId(id?: string) { const match = id?.match(/^toolcall_(\d+)$/); return match ? Number(match[1]) : undefined; } // As we capture LLM calls, we see: // - Request A, response AB // - Request ABC, response ABCD // - Request ABCDE, response ABCDEF // Among these, it's only necessary to keep the longest conversation (ABCDEF) since this contains all // information from the shorter ones. Avoiding duplication makes it reasonable for humans to reason // about diffs in the stored conversations when things change. function removePrefixConversations( conversations: NormalizedConversation[], ): NormalizedConversation[] { const result = [...conversations]; for (let i = result.length - 1; i >= 0; i--) { for (let j = i - 1; j >= 0; j--) { if (isPrefix(result[j].messages, result[i].messages)) { result.splice(j, 1); i--; // adjust index since we removed an element before current position } } } return result; } function isPrefix( shorter: NormalizedMessage[], longer: NormalizedMessage[], ): boolean { if (shorter.length >= longer.length) { return false; } return shorter.every( (msg, idx) => JSON.stringify(msg) === JSON.stringify(longer[idx]), ); } async function parseHttpExchange( requestBody: string, responseBody: string | undefined, requestHeaders?: Record, ): Promise { const request = JSON.parse(requestBody) as ChatCompletionCreateParamsBase; const response = await parseOpenAIResponse(responseBody); return { request, response, requestHeaders }; } // Converts a single HTTP exchange (request + response) into a normalized conversation async function transformHttpExchange( requestBody: string, responseBody: string | undefined, ): Promise<{ conversation: NormalizedConversation; model?: string }> { const { request, response } = await parseHttpExchange( requestBody, responseBody, ); const messages = request.messages.map(transformOpenAIRequestMessage); if (response?.choices?.length) { messages.push(...transformOpenAIResponseChoice(response.choices)); } return { conversation: { messages }, model: request.model }; } // Transforms a single OpenAI-style outbound request message into normalized form // We use this to look up whether we already have a cached response for it function transformOpenAIRequestMessage( m: ChatCompletionMessageParam, ): NormalizedMessage { let content: string | undefined; if (m.role === "system") { // System message changes too often to include in snapshots - just store placeholder content = "${system}"; } else if (m.role === "user" && typeof m.content === "string") { content = normalizeUserMessage(m.content); } else if (m.role === "user" && Array.isArray(m.content)) { // Multimodal user messages have array content with text and image_url parts. // Extract and normalize text parts; represent image_url parts as a stable marker. const parts: string[] = []; for (const part of m.content) { if ( typeof part === "object" && part.type === "text" && typeof part.text === "string" ) { parts.push(normalizeUserMessage(part.text)); } else if (typeof part === "object" && part.type === "image_url") { parts.push("[image]"); } } content = parts.join("\n") || undefined; } else if (m.role === "tool" && typeof m.content === "string") { // If it's a JSON tool call result, normalize the whitespace and property ordering. // For successful tool results wrapped in {resultType, textResultForLlm}, unwrap to // just the inner value so snapshots stay stable across envelope format changes. try { const parsed = JSON.parse(m.content); if ( parsed && typeof parsed === "object" && parsed.resultType === "success" && "textResultForLlm" in parsed ) { content = typeof parsed.textResultForLlm === "string" ? parsed.textResultForLlm : JSON.stringify(sortJsonKeys(parsed.textResultForLlm)); } else { content = JSON.stringify(sortJsonKeys(parsed)); } } catch { content = m.content.trim(); } } else if (typeof m.content === "string") { content = m.content; } const msg: NormalizedMessage = { role: m.role }; if ("tool_call_id" in m && m.tool_call_id) { msg.tool_call_id = m.tool_call_id; } if (content) msg.content = content; if ("tool_calls" in m && m.tool_calls?.length) { msg.tool_calls = m.tool_calls.map(transformOpenAIToolCall); } return msg; } function normalizeUserMessage(content: string): string { return normalizeSkillContextFrontmatter(content) .replace( taskCompletionNotificationPattern, taskCompletionNotificationReplacement, ) .replace(/.*?<\/current_datetime>/g, "") .replace(/[\s\S]*?<\/reminder>/g, "") .replace(/[\s\S]*?<\/system_reminder>/g, "") .replace(/[\s\S]*?<\/agent_instructions>/g, "") .replace(/^\s*\[\[PLAN\]\]\s*/, "") .replace( /Please create a detailed summary of the conversation so far\. The history is being compacted[\s\S]*/, "${compaction_prompt}", ) .trim(); } const taskCompletionNotificationPattern = /Agent "([^"]+)" \(([^)]+)\) (?:has completed successfully|has finished processing and is now idle)\. Use read_agent with agent_id "[^"]+" to (?:retrieve (?:unread results|the full results)|read the results, or write_agent to send follow-up messages)\./g; const taskCompletionNotificationReplacement = 'Agent "$1" ($2) has completed successfully. Use read_agent with agent_id "$1" to retrieve the full results.'; function normalizeStoredUserMessages(conversations: NormalizedConversation[]) { for (const conversation of conversations) { for (const message of conversation.messages) { if (message.role === "user" && typeof message.content === "string") { message.content = message.content.replace( taskCompletionNotificationPattern, taskCompletionNotificationReplacement, ); } } } } function normalizeStoredMessagesForBackend( conversations: NormalizedConversation[], backend: ReplayBackend, ) { if (backend === "capi") return; for (const conversation of conversations) { conversation.messages = coalesceMessages( conversation.messages, backend !== "openai-completions", ); } } function coalesceMessages( messages: NormalizedMessage[], coalesceAssistantMessages: boolean, ): NormalizedMessage[] { const result: NormalizedMessage[] = []; for (const message of messages) { const previous = result.at(-1); const shouldCoalesce = previous?.role === message.role && ((coalesceAssistantMessages && message.role === "assistant") || message.role === "user"); if (!shouldCoalesce) { result.push(message); continue; } const separator = message.role === "user" ? canonicalUserMessageSeparator : ""; const previousContent = previous.content ?? ""; const currentContent = message.content ?? ""; const content = `${previousContent}${previousContent && currentContent ? separator : ""}${currentContent}`; if (content) previous.content = content; const toolCalls = [ ...(previous.tool_calls ?? []), ...(message.tool_calls ?? []), ]; if (toolCalls.length) previous.tool_calls = toolCalls; } return result; } // Apply runtime-dependent tool result normalization to snapshots recorded by // older CLI versions as well as to live requests. function normalizeStoredToolMessages(conversations: NormalizedConversation[]) { for (const conversation of conversations) { for (const message of conversation.messages) { if (message.role === "tool" && typeof message.content === "string") { message.content = normalizeInterruptedToolResult(message.content); message.content = normalizeAvailableToolNames(message.content); message.content = normalizeBackgroundAgentStartMessage(message.content); message.content = normalizeReadAgentResult(message.content); } } } } function normalizeSkillContextFrontmatter(content: string): string { // Runtime versions may include or omit SKILL.md metadata in the prompt context. return content.replace( /(]*>\s*Base directory for this skill:[^\r\n]*(?:\r?\n)+)---\r?\n(?:(?!<\/skill-context>)[\s\S])*?\r?\n---(?:\r?\n)+/g, "$1", ); } function normalizeLargeOutputFilepaths(result: string): string { // Replaces filenames like 1774637043987-copilot-tool-output-tk7puw.txt with PLACEHOLDER-copilot-tool-output-PLACEHOLDER return result .replace( /\d+-copilot-tool-output-[a-z0-9.]+/g, "PLACEHOLDER-copilot-tool-output-PLACEHOLDER", ) .replace( /(?:[A-Za-z]:)?[^\s"'`]*[\\/]session-state[\\/]temp[\\/]PLACEHOLDER-copilot-tool-output-PLACEHOLDER/g, "/session-state/temp/PLACEHOLDER-copilot-tool-output-PLACEHOLDER", ); } function normalizeShellExitMarkers(result: string): string { return result.replace( /\r\n]+?\s+completed with exit code (-?\d+)>/g, "", ); } // The `gh` CLI emits different "not authenticated" help text depending on the // environment (local dev vs. inside GitHub Actions). Normalize both forms to a // stable placeholder so snapshots don't drift between environments. function normalizeGhAuthMessages(result: string): string { let normalized = result; // GitHub Actions form normalized = normalized.replace( /gh: To use GitHub CLI in a GitHub Actions workflow, set the GH_TOKEN environment variable\. Example:\s*\n\s*env:\s*\n\s*GH_TOKEN: \$\{\{ github\.token \}\}/g, "${gh_auth_required}", ); // Local dev form normalized = normalized.replace( /To get started with GitHub CLI, please run:\s*gh auth login\s*\n\s*Alternatively, populate the GH_TOKEN environment variable with a GitHub API authentication token\./g, "${gh_auth_required}", ); // When the GitHub CLI is run under the local CONNECT proxy on Windows, it // can try its auth probe before trusting the generated CA. This is still the // same unauthenticated-GitHub condition from the snapshot's perspective. normalized = normalized.replace( /[^\n]*Post "https:\/\/api\.github\.com\/graphql": tls: failed to verify certificate: x509: certificate signed by unknown authority\s*\n/g, "${gh_auth_required}\n", ); return normalizeGh401AuthMessages(normalized); } function normalizeGh401AuthMessages(result: string): string { const lines = result.split(/\r?\n/); const normalizedLines: string[] = []; let changed = false; for (let i = 0; i < lines.length; i++) { if ( /(?:HTTP|GraphQL)[ \t:]+401/.test(lines[i]) && lines[i].includes("Requires authentication") ) { let replaced = false; for (let j = i + 1; j < lines.length; j++) { if (/^$/.test(lines[j].trim())) { normalizedLines.push("${gh_auth_required}"); normalizedLines.push(""); i = j; changed = true; replaced = true; break; } } if (replaced) { continue; } } normalizedLines.push(lines[i]); } return changed ? normalizedLines.join("\n") : result; } function normalizeReadAgentResult(result: string): string { const normalized = result .replace( /^Agent is idle \(waiting for messages\)\./, "Agent completed.", ) .replace(/^Agent completed\. (.*), status: idle,/, "Agent completed. $1, status: completed,") .replace( /, total_turns: \d+(?=\r?\n|$)/, ", total_turns: 0, duration: 0s", ) .replace(/\r?\n\r?\n\[Turn \d+\]\r?\n/, "\n\n"); return normalized .replace(/\belapsed: \d+(?:\.\d+)?s\b/g, "elapsed: 0s") .replace(/\bduration: \d+(?:\.\d+)?s\b/g, "duration: 0s"); } // When a model calls a tool that doesn't exist (e.g., the removed report_intent // tool), the runtime replies with "Available tools that can be called are ." // That enumeration is both platform-specific (shell tool family names differ // across OSes) and runtime-version-specific (built-in tools such as write_agent // are added or removed over time). Some runtime builds omit the enumeration // entirely, so remove the optional suffix and retain only the stable error. function normalizeAvailableToolNames(result: string): string { return result.replace( /(Tool '[^']+' does not exist\.) Available tools that can be called are [^.]*\./g, "$1", ); } function normalizeInterruptedToolResult(result: string): string { return result.replace( /^(?:Session aborted|Failed to execute `[^`]+` tool(?: with arguments: [\s\S]*?)? due to error: (?:Error: )?Session aborted||unknown attachedShellSession handle \d+)$/, "The execution of this tool, or a previous tool was interrupted.", ); } function normalizeBackgroundAgentStartMessage(result: string): string { return normalizeBackgroundAgentStartMessageWithId(result); } function normalizeBackgroundAgentStartMessageWithId( result: string, stableId?: string, ): string { return result.replace( /^Agent started in background with agent_id: (.*?)(\. You'll be notified when it completes\. Tell the user you're waiting and end your response, or continue unrelated work until notified\.).*$/s, (_match, runtimeId: string, suffix: string) => `Agent started in background with agent_id: ${stableId ?? runtimeId}${suffix}`, ); } function getBackgroundAgentName(argumentsJson: string): string | undefined { try { const args = JSON.parse(argumentsJson) as unknown; if ( args && typeof args === "object" && "name" in args && typeof args.name === "string" ) { return args.name; } } catch { // Invalid tool arguments are left for the runtime to report. } return undefined; } function getBackgroundAgentId(content: string): string | undefined { const match = content.match( /^Agent started in background with agent_id: (.*?)\. You'll be notified when it completes\./s, ); if (match?.[1]) { return match[1]; } try { const parsed = JSON.parse(content) as unknown; if ( parsed && typeof parsed === "object" && "textResultForLlm" in parsed && typeof parsed.textResultForLlm === "string" ) { return getBackgroundAgentId(parsed.textResultForLlm); } } catch { // Plain-text tool results are expected. } return undefined; } function replaceBackgroundAgentIds( content: string, backgroundAgentNamesByRuntimeId: Map, ): string { let normalized = content; for (const [runtimeId, stableName] of backgroundAgentNamesByRuntimeId) { normalized = normalized.split(runtimeId).join(stableName); } return normalized; } function rewriteBackgroundAgentArguments( argumentsJson: string, replacements: Map, ): string { try { const args = JSON.parse(argumentsJson) as unknown; if (!args || typeof args !== "object") { return argumentsJson; } if ( "agent_id" in args && typeof args.agent_id === "string" && replacements.has(args.agent_id) ) { args.agent_id = replacements.get(args.agent_id); } if ("agent_ids" in args && Array.isArray(args.agent_ids)) { args.agent_ids = args.agent_ids.map((agentId: unknown) => typeof agentId === "string" ? (replacements.get(agentId) ?? agentId) : agentId, ); } return JSON.stringify(args); } catch { return argumentsJson; } } function normalizeBackgroundAgentArguments( argumentsJson: string, backgroundAgentNamesByRuntimeId: Map, ): string { return rewriteBackgroundAgentArguments( argumentsJson, backgroundAgentNamesByRuntimeId, ); } function extractBackgroundAgentIds( requestBody: string | undefined, ): Map { const result = new Map(); if (!requestBody) { return result; } try { const request = JSON.parse(requestBody) as { messages?: Array<{ role?: string; tool_call_id?: string; content?: unknown; tool_calls?: Array<{ id?: string; function?: { name?: string; arguments?: string }; }>; }>; }; const backgroundAgentNamesByToolCallId = new Map(); let unnamedBackgroundAgentCounter = 0; for (const message of request.messages ?? []) { for (const toolCall of message.tool_calls ?? []) { if ( toolCall.id && toolCall.function?.name === "task" && toolCall.function.arguments ) { const configuredName = getBackgroundAgentName( toolCall.function.arguments, ); const fallbackName = unnamedBackgroundAgentCounter === 0 ? "background-agent" : `background-agent-${unnamedBackgroundAgentCounter}`; backgroundAgentNamesByToolCallId.set( toolCall.id, configuredName ?? fallbackName, ); unnamedBackgroundAgentCounter++; } } if ( message.role === "tool" && message.tool_call_id && typeof message.content === "string" ) { const stableName = backgroundAgentNamesByToolCallId.get( message.tool_call_id, ); const runtimeId = getBackgroundAgentId(message.content); if (stableName && runtimeId) { result.set(stableName, runtimeId); } } } } catch { // Request parsing failures are handled by the normal replay matcher. } return result; } function expandBackgroundAgentArguments( argumentsJson: string, backgroundAgentIds: Map, ): string { return rewriteBackgroundAgentArguments(argumentsJson, backgroundAgentIds); } // Transforms a single OpenAI-style inbound response message into normalized form function transformOpenAIResponseChoice( choices: ChatCompletion.Choice[], ): NormalizedMessage[] { // Maps each choice to a separate assistant message. // This is clearly wrong, since choices are meant to be alternatives (from which the client // should pick one). However CAPI frequently returns collections of tool calls as separate choices, // and our chat-completion-client.ts logic handles this by treating them as sequential messages. // So, we have to do the same thing here. return choices.map((choice) => { const tool_calls = choice.message.tool_calls?.map(transformOpenAIToolCall) ?? []; const msg: NormalizedMessage = { role: "assistant" }; msg.content = choice.message.content ?? undefined; msg.refusal = choice.message.refusal ?? undefined; if (tool_calls.length) msg.tool_calls = tool_calls; return msg; }); } function transformOpenAIToolCall(tc: { id: string; type: string; function?: { name: string; arguments: string }; }): NormalizedToolCall { return { id: tc.id, type: tc.type, function: tc.function && { name: tc.function.name, arguments: normalizeToolCallArguments(tc.function.arguments), }, }; } function normalizeToolCallArguments(args: string): string { if (!args || args.trim() === "") { return "{}"; } try { return JSON.stringify(JSON.parse(args)); } catch { return args; } } // Takes raw HTTP response data and turns it into an OpenAI ChatCompletion object, regardless of whether // it's a streaming or non-streaming response async function parseOpenAIResponse( responseBody: string | undefined, ): Promise { // Check if it's a streaming response (Server-Sent Events format) if (responseBody?.startsWith("data:")) { const lines = responseBody .split("\n") .filter((line) => line.startsWith("data:") && !line.includes("[DONE]")); const chunks = lines.map( (line) => JSON.parse(line.slice(5)) as ChatCompletionChunk, ); // Convert the sequence of chunks into a final ChatCompletion object // TODO: Do we need to apply fixCAPIStreamingToolCalling normalization here? const readableStream = new ReadableStream({ async start(controller) { for (const chunk of chunks) { controller.enqueue( new TextEncoder().encode(JSON.stringify(chunk) + "\n"), ); } controller.close(); }, }); return await ChatCompletionStream.fromReadableStream( readableStream, ).finalChatCompletion(); } else if (responseBody) { return JSON.parse(responseBody) as ChatCompletion; } } // Checks if requestMessages is a prefix of savedMessages, // and returns the index of the next assistant message if found. function findAssistantIndexAfterPrefix( requestMessages: NormalizedMessage[], savedMessages: NormalizedMessage[], ): number | undefined { const logFile = process.env.PROXY_DEBUG_LOG; const log = (msg: string) => { if (logFile) try { appendFileSync(logFile, msg + "\n"); } catch {} }; if (requestMessages.length >= savedMessages.length) { log(`prefix check failed: request.length=${requestMessages.length} >= saved.length=${savedMessages.length}`); return undefined; } for (let i = 0; i < requestMessages.length; i++) { const reqMsg = JSON.stringify(requestMessages[i]); const savedMsg = JSON.stringify(savedMessages[i]); if (reqMsg !== savedMsg) { log(`mismatch at index ${i}:`); log(` REQ: ${reqMsg.substring(0, 1000)}`); log(` SAVED: ${savedMsg.substring(0, 1000)}`); return undefined; } } // The next message after the prefix should be an assistant message const nextIndex = requestMessages.length; if ( nextIndex < savedMessages.length && savedMessages[nextIndex].role === "assistant" ) { log(`MATCH found at index ${nextIndex}`); return nextIndex; } log(`no assistant at nextIndex=${nextIndex}, saved.length=${savedMessages.length}`); return undefined; } function expandWorkDir( content: string | undefined, workDir: string, jsonEscape: boolean, ): string | undefined { if (!content) { return content; } const workDirValue = jsonEscape ? JSON.stringify(workDir).replaceAll('"', "") : workDir; return content.replace(/\$\{workdir\}/g, workDirValue); } function expandToolName(name: string): string { for (const [fullName, normalized] of Object.entries(normalizedToolNames)) { if (name === normalized) { return fullName; } } return name; } // Turns a normalized message back into a full OpenAI ChatCompletion that we can replay as a response function createOpenAIResponse( model: string, messages: NormalizedMessage[], responseStartIndex: number, workDir: string, backgroundAgentIds: Map, ): ChatCompletion { // Here we recreate the strange CAPI/productcode behavior of using multiple choices to represent // multiple assistant messages in a row. This is the inverse of the logic in transformOpenAIResponseChoice(). // So, find all successive assistant messages starting from responseStartIndex. const choices: ChatCompletion.Choice[] = []; for ( let index = 0; messages[responseStartIndex + index]?.role === "assistant"; index++ ) { const assistantMessage = messages[responseStartIndex + index]; const toolCalls = assistantMessage.tool_calls?.map((tc, idx) => ({ id: tc.id || `call_${idx}`, type: "function" as const, function: { name: expandToolName(tc.function?.name ?? ""), arguments: expandBackgroundAgentArguments( expandWorkDir(tc.function?.arguments, workDir, true) ?? "{}", backgroundAgentIds, ), }, })); choices.push({ index, message: { role: "assistant", content: expandWorkDir(assistantMessage.content, workDir, false) ?? null, refusal: assistantMessage.refusal ?? null, tool_calls: toolCalls, }, finish_reason: toolCalls?.length ? "tool_calls" : "stop", logprobs: null, }); } return { id: "cached-completion", object: "chat.completion", created: Math.floor(Date.now() / 1000), model, choices, usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0, }, }; } const STREAM_CHUNK_SIZE = 200; function convertToStreamingResponseChunks( completion: ChatCompletion, ): ChatCompletionChunk[] { const choice = completion.choices[0]; const content = choice.message.content ?? ""; const toolCalls = choice.message.tool_calls?.filter( (tc): tc is ChatCompletionMessageFunctionToolCall => tc.type === "function", ); const makeChunk = ( delta: ChatCompletionChunk.Choice.Delta, ): ChatCompletionChunk => ({ id: completion.id, object: "chat.completion.chunk", created: completion.created, model: completion.model, choices: [{ index: 0, delta, finish_reason: null, logprobs: null }], }); const chunks: ChatCompletionChunk[] = []; // Content chunks for (let i = 0; i < content.length; i += STREAM_CHUNK_SIZE) { chunks.push( makeChunk({ role: "assistant", content: content.slice(i, i + STREAM_CHUNK_SIZE), }), ); } // Tool call argument chunks for (const [tcIdx, tc] of (toolCalls ?? []).entries()) { const args = tc.function.arguments; for (let i = 0; i < args.length; i += STREAM_CHUNK_SIZE) { chunks.push( makeChunk({ role: "assistant", tool_calls: [ { index: tcIdx, id: tc.id, type: "function", function: { name: i === 0 ? tc.function.name : "", arguments: args.slice(i, i + STREAM_CHUNK_SIZE), }, }, ], }), ); } } // Set finish_reason on last chunk if (chunks.length === 0) { chunks.push(makeChunk({ role: "assistant" })); } chunks[chunks.length - 1].choices[0].finish_reason = choice.finish_reason; return chunks; } function createGetModelsResponse(modelIds: string[]) { // Obviously the following might not match any given model. We could track the original responses from /models, // but that risks invalidating the caches too frequently and making this unmaintainable. If this approximation // turns out to be insufficient, we can tweak the logic here based on known model IDs. return { data: modelIds.map((id) => ({ id, name: id, capabilities: { supports: { vision: true }, limits: { max_context_window_tokens: 128000 }, }, })), }; } async function writeFileIfDifferent(filePath: string, contents: string) { // Read directly instead of checking for existence first, which would be a // check-then-use race on the same path. const existingContents = await readFile(filePath, "utf-8").catch( (error: NodeJS.ErrnoException) => { if (error.code === "ENOENT") { return undefined; } throw error; }, ); if (existingContents === contents) { return; } await writeFile(filePath, contents, "utf-8"); } function excludeFailedResponses(exchange: CapturedExchange): boolean { const status = exchange.response?.statusCode; return status === undefined || (status >= 200 && status < 300); } export type ToolResultNormalizer = { toolName: string; normalizer: (result: string) => string; }; /** * Response shape for the `/copilot_internal/user` endpoint. * Used by per-session auth tests to mock GitHub identity resolution. */ export type CopilotUserResponse = { login: string; copilot_plan?: string; token_based_billing?: boolean; is_mcp_enabled?: boolean; endpoints?: { api?: string; telemetry?: string; }; analytics_tracking_id?: string; quota_snapshots?: Record< string, { entitlement?: number; overage_count?: number; overage_permitted?: boolean; percent_remaining?: number; timestamp_utc?: string; unlimited?: boolean; } >; }; export type ParsedHttpExchange = { request: ChatCompletionCreateParamsBase; response: ChatCompletion | undefined; requestHeaders?: Record; }; // We want to be able to reuse the proxy across multiple tests, so it needs to be reconfigurable // and resettable on demand. By holding all state in one place it's easy to manage. type ReplayingCapiProxyState = { filePath: string; workDir: string; testInfo?: { file: string; line?: number }; backend: ReplayBackend; storedData?: NormalizedData | undefined; toolResultNormalizers: ToolResultNormalizer[]; }; interface NormalizedToolCall { id: string; type: string; function?: { name: string; arguments: string; }; } interface NormalizedMessage { role: string; content?: string; refusal?: string; tool_calls?: NormalizedToolCall[]; tool_call_id?: string; } interface NormalizedConversation { messages: NormalizedMessage[]; } interface NormalizedErrorResponse { model?: string; status: number; code?: string; message?: string; retryAfterSeconds?: number; messages: NormalizedMessage[]; } export interface NormalizedData { models: string[]; errors?: NormalizedErrorResponse[]; conversations: NormalizedConversation[]; } function sortJsonKeys(obj: unknown): unknown { if (Array.isArray(obj)) return obj.map(sortJsonKeys); if (obj && typeof obj === "object") { return Object.keys(obj) .sort() .reduce( (acc, key) => { acc[key] = sortJsonKeys((obj as Record)[key]); return acc; }, {} as Record, ); } return obj; }