diff --git a/API.md b/API.md index dd2d69b6..61575c1f 100644 --- a/API.md +++ b/API.md @@ -145,6 +145,29 @@ For SWE Atlas runs, `judgeModel` may also be provided: The judge always uses the server-configured DigitalOcean inference endpoint and key. `judgeModel` is ignored for non-SWE-Atlas benchmarks. +TAU Airline always uses `openai-gpt-5.4-mini` for its simulated customer through a +DigitalOcean inference endpoint. When `inference.baseUrl` is +`https://inference.do-ai.run/v1` or +`https://inference.do-ai-test.run/v1`, the simulator reuses that endpoint and +`inference.apiKey`. For OpenRouter or a custom candidate endpoint, provide a +separate top-level `simulatorApiKey`; the simulator then uses +`https://inference.do-ai.run/v1`: + +```json +{ + "benchmark": "tau_bench_verified_airline", + "simulatorApiKey": "", + "inference": { + "baseUrl": "https://openrouter.ai/api/v1", + "apiKey": "", + "model": "provider/model" + } +} +``` + +Both inference secrets are launch-only and are never returned, logged, or +persisted. + Inference constraints: - `model` accepts letters, digits, `.`, `_`, `:`, `-`, and `/`. @@ -595,11 +618,14 @@ Request: { "sampleId": "sample-id", "originalEpoch": 0, - "apiKey": "" + "apiKey": "", + "simulatorApiKey": "" } ``` -The source item must exist in the benchmark-specific report. +`simulatorApiKey` is required only when retrying a TAU Airline run whose +candidate used OpenRouter or a custom endpoint. The source item must exist in +the benchmark-specific report. Response: `202 Accepted` diff --git a/README.md b/README.md index 8ede4f9c..15b6b905 100644 --- a/README.md +++ b/README.md @@ -136,12 +136,6 @@ export DO_MODEL_CATALOG_TOKEN='replace-me' # OpenRouter API key used only to list OpenRouter models. # Its successful catalog response is cached by the server for two hours. export OPENROUTER_MODEL_CATALOG_TOKEN='replace-me' -# Optional TAU Airline user-simulator endpoint. When the API key is set, TAU -# uses this independent OpenAI-compatible endpoint for its simulated customer. -export TAU_AIRLINE_USER_SIMULATOR_API_KEY='replace-me' -export TAU_AIRLINE_USER_SIMULATOR_BASE_URL='https://generativelanguage.googleapis.com/v1beta/openai' -export TAU_AIRLINE_USER_SIMULATOR_MODEL='gemini-2.5-flash' - export MYSQL_HOST='replace-me.db.ondigitalocean.com' export MYSQL_PORT='25060' export MYSQL_USER='doadmin' @@ -220,7 +214,7 @@ curl -X POST http://127.0.0.1:8080/runs \ }' ``` -The API and dashboard default GPQA and TAU to three epochs and concurrency three. Deep SWE, SWE-bench Verified, Terminal-Bench, and SWE Atlas default to one epoch and concurrency one because each evaluation provisions disposable sandbox workers. All benchmarks default to reasoning effort `high`, a one-hour full-response timeout per attempt, six retries for retryable inference errors, and retry-on-error behavior. GPQA defaults to temperature `1`; TAU, Deep SWE, SWE-bench Verified, Terminal-Bench, and SWE Atlas default to temperature `0`. Request timeouts (`408`), rate limits (`429`), transport failures, and server errors (`5xx`) are retryable; `maxRetries: 0` disables inference retries. Every value can be overridden. Set `"benchmark": "tau_bench_verified_airline"` to run the 50-task TAU suite, `"benchmark": "deep_swe"` for the 113-task Deep SWE suite, `"benchmark": "swe_bench_verified"` for the 500-task SWE-bench Verified suite, `"benchmark": "terminal_bench"` for the 89-task Terminal-Bench 2.1 suite, or `"benchmark": "swe_atlas_qa"` for the 124-task SWE Atlas QA suite. SWE Atlas runs accept a top-level `"judgeModel"` selected independently from `inference.model`; the judge always uses `https://inference.do-ai.run/v1` and the server-side `SWE_ATLAS_JUDGE_API_KEY`. The dashboard shows this field only for SWE Atlas. SWE-bench uses Harbor's deterministic verifier and does not use a judge model. Each SWE-bench task uses one worker for both candidate work and verification; hidden tests are uploaded only after the candidate patch is captured. The patch and verifier report are saved in sample metadata in the result Parquet and therefore included in the normal Spaces run bundle. Terminal-Bench and SWE Atlas also use one worker per concurrent evaluation, with agent and verifier sharing the worker. SWE Atlas task metadata requests 16 CPUs and 16 GiB RAM, so configure its dedicated size override before increasing concurrency. If `TAU_AIRLINE_USER_SIMULATOR_API_KEY` is configured, its model, base URL, and key are used for TAU's simulated customer while `inference` continues to configure the evaluated agent model; its default model is `gemini-2.5-flash`, and `TAU_AIRLINE_USER_SIMULATOR_MODEL` overrides it. Optional inference fields are `temperature`, `endpointId`, `costTier`, `sort`, `providerOnly`, `allowFallbacks`, `cloudflareVersion`, `costQualityTradeoff`, `pinModel`, and `completionTimeoutMs`. `completionTimeoutMs` limits the complete inference attempt, including connection and streamed-body processing. The existing `timeoutMs` remains the connection/response-header timeout. When OpenRouter is selected, the dashboard exposes provider routing only inside Advanced configuration. Selecting DigitalOcean sends `providerOnly: ["digitalocean"]` and disables provider fallback by default, so OpenRouter must use DigitalOcean or fail the request. Execution can use `start` plus either `end` or `limit`. Set `execution.unordered` to `true` for rolling concurrency without input-order head-of-line blocking; omitted or `false` preserves ordered result emission. +The API and dashboard default GPQA and TAU to three epochs and concurrency three. Deep SWE, SWE-bench Verified, Terminal-Bench, and SWE Atlas default to one epoch and concurrency one because each evaluation provisions disposable sandbox workers. All benchmarks default to reasoning effort `high`, a one-hour full-response timeout per attempt, six retries for retryable inference errors, and retry-on-error behavior. GPQA defaults to temperature `1`; TAU, Deep SWE, SWE-bench Verified, Terminal-Bench, and SWE Atlas default to temperature `0`. Request timeouts (`408`), rate limits (`429`), transport failures, and server errors (`5xx`) are retryable; `maxRetries: 0` disables inference retries. Every value can be overridden. Set `"benchmark": "tau_bench_verified_airline"` to run the 50-task TAU suite, `"benchmark": "deep_swe"` for the 113-task Deep SWE suite, `"benchmark": "swe_bench_verified"` for the 500-task SWE-bench Verified suite, `"benchmark": "terminal_bench"` for the 89-task Terminal-Bench 2.1 suite, or `"benchmark": "swe_atlas_qa"` for the 124-task SWE Atlas QA suite. SWE Atlas runs accept a top-level `"judgeModel"` selected independently from `inference.model`; the judge always uses `https://inference.do-ai.run/v1` and the server-side `SWE_ATLAS_JUDGE_API_KEY`. The dashboard shows this field only for SWE Atlas. SWE-bench uses Harbor's deterministic verifier and does not use a judge model. Each SWE-bench task uses one worker for both candidate work and verification; hidden tests are uploaded only after the candidate patch is captured. The patch and verifier report are saved in sample metadata in the result Parquet and therefore included in the normal Spaces run bundle. Terminal-Bench and SWE Atlas also use one worker per concurrent evaluation, with agent and verifier sharing the worker. SWE Atlas task metadata requests 16 CPUs and 16 GiB RAM, so configure its dedicated size override before increasing concurrency. TAU Airline always uses `openai-gpt-5.4-mini` for its simulated customer through DigitalOcean inference. A production or test DigitalOcean candidate reuses its `inference.baseUrl` and `inference.apiKey`; an OpenRouter or custom candidate requires a separate top-level `simulatorApiKey`, which is used only with `https://inference.do-ai.run/v1`. The dashboard requests this token only when required, and neither secret is persisted. Optional inference fields are `temperature`, `endpointId`, `costTier`, `sort`, `providerOnly`, `allowFallbacks`, `cloudflareVersion`, `costQualityTradeoff`, `pinModel`, and `completionTimeoutMs`. `completionTimeoutMs` limits the complete inference attempt, including connection and streamed-body processing. The existing `timeoutMs` remains the connection/response-header timeout. When OpenRouter is selected, the dashboard exposes provider routing only inside Advanced configuration. Selecting DigitalOcean sends `providerOnly: ["digitalocean"]` and disables provider fallback by default, so OpenRouter must use DigitalOcean or fail the request. Execution can use `start` plus either `end` or `limit`. Set `execution.unordered` to `true` for rolling concurrency without input-order head-of-line blocking; omitted or `false` preserves ordered result emission. Use `"swe_atlas_qa"` for the 124-task Codebase Q&A track, `"swe_atlas_tw"` for the 90-task Test Writing track, or `"swe_atlas_rf"` for the 65-task Refactoring track. All three use the same independently selected judge model and default to one epoch and concurrency one in managed runs. diff --git a/src/benchmarks/benchmark-config.ts b/src/benchmarks/benchmark-config.ts index b55e826f..6fcf9677 100644 --- a/src/benchmarks/benchmark-config.ts +++ b/src/benchmarks/benchmark-config.ts @@ -16,10 +16,7 @@ import { ORI_CHANNELS, ORI_REASONING_EFFORTS, } from "./agent-cli/schema"; -import { - TAU3_BENCH_BANKING_META, - TAU_BENCH_AIRLINE_META, -} from "./benchmark-meta"; +import { TAU3_BENCH_BANKING_META } from "./benchmark-meta"; import { DEFAULT_STEP_LIMIT as DEEP_SWE_DEFAULT_STEP_LIMIT } from "./deep-swe/schema"; import { DracoPanelConfigSchema } from "./draco/schemas"; import { SearchLaneConfigSchema } from "./search/core/config"; @@ -99,7 +96,6 @@ export type MmluProBenchmarkConfig = z.infer< >; export const TauBenchOptionsSchema = z.object({ - userModel: zDefaultedText(TAU_BENCH_AIRLINE_META.userModel), userReasoningEffort: z.enum(REASONING_EFFORTS).default("medium"), }); diff --git a/src/benchmarks/benchmark-meta.ts b/src/benchmarks/benchmark-meta.ts index 21d22656..77936619 100644 --- a/src/benchmarks/benchmark-meta.ts +++ b/src/benchmarks/benchmark-meta.ts @@ -25,7 +25,7 @@ export const TAU_BENCH_AIRLINE_META = { id: "tau_bench_verified_airline", defaultEpochs: 3, temperature: 0, - userModel: "openai/gpt-5.4-mini", + userModel: "openai-gpt-5.4-mini", } as const satisfies BenchmarkMeta; export const TAU3_BENCH_BANKING_META = { diff --git a/src/benchmarks/tau-bench-airline/airline.test.ts b/src/benchmarks/tau-bench-airline/airline.test.ts index 7b98f440..3e9175bc 100644 --- a/src/benchmarks/tau-bench-airline/airline.test.ts +++ b/src/benchmarks/tau-bench-airline/airline.test.ts @@ -26,8 +26,8 @@ import { benchmarkIds, getBenchmark } from "../registry"; import { compareActionWithToolCall } from "./action-match"; import { airlineRecordToSample, - resolveAirlineUserModel, TAU_BENCH_AIRLINE_ID, + TAU_BENCH_AIRLINE_USER_SIMULATOR_MODEL, } from "./benchmark"; import { ensureAirlineData, @@ -120,24 +120,13 @@ describe("tau_bench_verified_airline registry", () => { expect(b?.id).toBe("tau_bench_verified_airline"); expect(b?.temperature).toBe(0); expect(b?.defaultEpochs).toBe(3); - expect(b?.userModel).toBe("openai/gpt-5.4-mini"); + expect(b?.userModel).toBe("openai-gpt-5.4-mini"); }); it("appears in benchmarkIds()", () => { expect(benchmarkIds()).toContain("tau_bench_verified_airline"); }); - it("uses the DigitalOcean user-model slug only for DigitalOcean inference", () => { - expect( - resolveAirlineUserModel( - "openai/gpt-5.4-mini", - "https://inference.do-ai.run/v1" - ) - ).toBe("openai-gpt-5.4-mini"); - expect( - resolveAirlineUserModel( - "openai/gpt-5.4-mini", - "https://openrouter.ai/api/v1" - ) - ).toBe("openai/gpt-5.4-mini"); + it("fixes the user simulator to the DigitalOcean model slug", () => { + expect(TAU_BENCH_AIRLINE_USER_SIMULATOR_MODEL).toBe("openai-gpt-5.4-mini"); }); }); describe("tau_bench_verified_airline config", () => { diff --git a/src/benchmarks/tau-bench-airline/benchmark.ts b/src/benchmarks/tau-bench-airline/benchmark.ts index 8ef1afe8..31deb483 100644 --- a/src/benchmarks/tau-bench-airline/benchmark.ts +++ b/src/benchmarks/tau-bench-airline/benchmark.ts @@ -19,6 +19,7 @@ import { Solver } from "../../harness/solver"; import { Either } from "../../internal/either"; import { definedValues } from "../../internal/guards"; import { parseSchema } from "../../internal/zod"; +import { isDigitalOceanInferenceBaseUrl } from "../../providers/digitalocean-inference"; import { makeOpenRouterModelLayer } from "../../providers/openrouter-model"; import { makeResponsesModelLayer, @@ -40,25 +41,8 @@ export const TAU_BENCH_AIRLINE_TEMPERATURE = TAU_BENCH_AIRLINE_META.temperature; export const TAU_BENCH_AIRLINE_ID = TAU_BENCH_AIRLINE_META.id; -const DIGITALOCEAN_INFERENCE_BASE_URLS = new Set([ - "https://inference.do-ai.run/v1", - "https://inference.do-ai-test.run/v1", -]); - -export const DIGITALOCEAN_MODEL_SLUGS: Readonly> = { - "openai/gpt-5.4-mini": "openai-gpt-5.4-mini", -}; - -export function resolveAirlineUserModel( - model: string, - baseUrl: string | undefined -): string { - const normalizedBaseUrl = baseUrl?.replace(/\/+$/, ""); - return normalizedBaseUrl !== undefined && - DIGITALOCEAN_INFERENCE_BASE_URLS.has(normalizedBaseUrl) - ? (DIGITALOCEAN_MODEL_SLUGS[model] ?? model) - : model; -} +export const TAU_BENCH_AIRLINE_USER_SIMULATOR_MODEL = + TAU_BENCH_AIRLINE_META.userModel; export function airlineRecordToSample( record: Readonly>, @@ -123,11 +107,17 @@ function makeAirlineLayer( } const userSimulator = input.userSimulator; const userSimulatorBaseUrl = userSimulator?.baseUrl ?? input.baseUrl; - const defaultUserModel = resolveAirlineUserModel( - benchmarkConfig.userModel, - userSimulatorBaseUrl - ); - const userSimulatorModel = userSimulator?.model ?? defaultUserModel; + if ( + userSimulatorBaseUrl === undefined || + !isDigitalOceanInferenceBaseUrl(userSimulatorBaseUrl) + ) { + return layerFail( + new Error( + "TAU Airline user simulator requires a DigitalOcean inference endpoint and access token" + ) + ); + } + const userSimulatorModel = TAU_BENCH_AIRLINE_USER_SIMULATOR_MODEL; const solverOpts: SolverOpts = definedValues({ endpointId: benchmarkConfig.endpointId, userModelConfig: definedValues({ diff --git a/src/benchmarks/tau-bench-airline/user-simulator.ts b/src/benchmarks/tau-bench-airline/user-simulator.ts index 80ed5296..df58b60d 100644 --- a/src/benchmarks/tau-bench-airline/user-simulator.ts +++ b/src/benchmarks/tau-bench-airline/user-simulator.ts @@ -13,7 +13,7 @@ import { retrySalted, withRetryAttemptLogging } from "../../runtime/retry"; import type { UserModelConfig } from "./types"; import { USER_SIM_GUIDELINES } from "./user-sim-guidelines"; -const USER_FALLBACK_MODEL = "openai/gpt-5.4-mini"; +const USER_FALLBACK_MODEL = "openai-gpt-5.4-mini"; class UserSimError extends TaggedError("UserSimError")<{ readonly message: string; diff --git a/src/benchmarks/types.ts b/src/benchmarks/types.ts index f638dc82..0026bb02 100644 --- a/src/benchmarks/types.ts +++ b/src/benchmarks/types.ts @@ -21,7 +21,6 @@ export interface BenchmarkRunInput< readonly userSimulator?: { readonly apiKey: string; readonly baseUrl: string; - readonly model: string; }; readonly benchmarkConfig: Config; readonly sessionId: string; diff --git a/src/cli/index.test.ts b/src/cli/index.test.ts index 53e06b89..13ea16ed 100644 --- a/src/cli/index.test.ts +++ b/src/cli/index.test.ts @@ -368,16 +368,37 @@ describe("bench-harness CLI", () => { retrievalConfig: "bm25_grep", }); }); - it("uses server-provided Gemini settings for the TAU user simulator", () => { + it("reuses DigitalOcean candidate credentials for the TAU simulator", () => { + for (const baseUrl of [ + "https://inference.do-ai.run/v1", + "https://inference.do-ai-test.run/v1/", + ]) { + expect( + tauAirlineUserSimulatorFromEnv({}, "candidate-key", baseUrl) + ).toEqual({ + apiKey: "candidate-key", + baseUrl: baseUrl.replace(/\/+$/u, ""), + }); + } + }); + + it("requires a separate DigitalOcean token for non-DO candidates", () => { expect( - tauAirlineUserSimulatorFromEnv({ - TAU_AIRLINE_USER_SIMULATOR_API_KEY: "gemini-key", - }) + tauAirlineUserSimulatorFromEnv( + { TAU_AIRLINE_USER_SIMULATOR_API_KEY: "simulator-key" }, + "candidate-key", + "https://openrouter.ai/api/v1" + ) ).toEqual({ - apiKey: "gemini-key", - baseUrl: "https://generativelanguage.googleapis.com/v1beta/openai", - model: "gemini-2.5-flash", + apiKey: "simulator-key", + baseUrl: "https://inference.do-ai.run/v1", }); - expect(tauAirlineUserSimulatorFromEnv({})).toBeUndefined(); + expect(() => + tauAirlineUserSimulatorFromEnv( + {}, + "candidate-key", + "https://inference.example.com/v1" + ) + ).toThrow("TAU_AIRLINE_USER_SIMULATOR_API_KEY"); }); }); diff --git a/src/cli/index.ts b/src/cli/index.ts index d9d0b56b..76a40489 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -30,6 +30,10 @@ import { runHarnessPromise } from "../internal/effect-logger"; import { Either } from "../internal/either"; import { definedValues, isMember } from "../internal/guards"; import { parseSchema } from "../internal/zod"; +import { + DEFAULT_DIGITALOCEAN_INFERENCE_BASE_URL, + isDigitalOceanInferenceBaseUrl, +} from "../providers/digitalocean-inference"; import { makeLocalResultStore } from "../results/result-store"; import { datasetSizeById, runBenchmarkById } from "../runner/run-by-id"; @@ -176,25 +180,31 @@ function resolveApiKey(): string { } export function tauAirlineUserSimulatorFromEnv( - env: NodeJS.ProcessEnv = process.env -): - | { - readonly apiKey: string; - readonly baseUrl: string; - readonly model: string; - } - | undefined { + env: NodeJS.ProcessEnv, + inferenceApiKey: string, + inferenceBaseUrl: string | undefined +): { + readonly apiKey: string; + readonly baseUrl: string; +} { + if ( + inferenceBaseUrl !== undefined && + isDigitalOceanInferenceBaseUrl(inferenceBaseUrl) + ) { + return { + apiKey: inferenceApiKey, + baseUrl: inferenceBaseUrl.replace(/\/+$/u, ""), + }; + } const apiKey = env["TAU_AIRLINE_USER_SIMULATOR_API_KEY"]?.trim(); if (!apiKey) { - return undefined; + throw new Error( + "Set TAU_AIRLINE_USER_SIMULATOR_API_KEY when the TAU candidate uses OpenRouter or a custom inference endpoint." + ); } return { apiKey, - baseUrl: - env["TAU_AIRLINE_USER_SIMULATOR_BASE_URL"]?.trim() || - "https://generativelanguage.googleapis.com/v1beta/openai", - model: - env["TAU_AIRLINE_USER_SIMULATOR_MODEL"]?.trim() || "gemini-2.5-flash", + baseUrl: DEFAULT_DIGITALOCEAN_INFERENCE_BASE_URL, }; } @@ -207,13 +217,17 @@ function main(): Promise { ); } const apiKey = resolveApiKey(); - const tauAirlineUserSimulator = - args.benchmark === "tau_bench_verified_airline" - ? tauAirlineUserSimulatorFromEnv() - : undefined; const baseUrl = getOrNull( runSync(string("OPENROUTER_BASE_URL").pipe(option)) ); + const tauAirlineUserSimulator = + args.benchmark === "tau_bench_verified_airline" + ? tauAirlineUserSimulatorFromEnv( + process.env, + apiKey, + baseUrl ?? undefined + ) + : undefined; const epochs = args.epochs ?? benchmark.defaultEpochs; const range = resolveRange(args); const sessionId = resolveSessionId(); diff --git a/src/providers/digitalocean-inference.ts b/src/providers/digitalocean-inference.ts new file mode 100644 index 00000000..25242c03 --- /dev/null +++ b/src/providers/digitalocean-inference.ts @@ -0,0 +1,18 @@ +export const DIGITALOCEAN_INFERENCE_BASE_URLS = [ + "https://inference.do-ai.run/v1", + "https://inference.do-ai-test.run/v1", +] as const; + +export const DEFAULT_DIGITALOCEAN_INFERENCE_BASE_URL = + DIGITALOCEAN_INFERENCE_BASE_URLS[0]; + +function withoutTrailingSlashes(value: string): string { + return value.replace(/\/+$/u, ""); +} + +export function isDigitalOceanInferenceBaseUrl(value: string): boolean { + const normalized = withoutTrailingSlashes(value); + return DIGITALOCEAN_INFERENCE_BASE_URLS.some( + (baseUrl) => baseUrl === normalized + ); +} diff --git a/src/runner/run-by-id.ts b/src/runner/run-by-id.ts index c518f0f7..8396cbaf 100644 --- a/src/runner/run-by-id.ts +++ b/src/runner/run-by-id.ts @@ -55,7 +55,6 @@ export interface RunBenchmarkInput { readonly userSimulator?: { readonly apiKey: string; readonly baseUrl: string; - readonly model: string; }; readonly benchmarkConfig: BenchmarkRunConfig; readonly epochs: number; diff --git a/src/server/dashboard-html.ts b/src/server/dashboard-html.ts index d56ba86c..3ac5cd85 100644 --- a/src/server/dashboard-html.ts +++ b/src/server/dashboard-html.ts @@ -233,7 +233,7 @@ export const DASHBOARD_HTML = ` + @@ -446,6 +450,12 @@ export const DASHBOARD_HTML = ` const startForm = document.getElementById("start-form"); const startBenchmark = document.getElementById("start-benchmark"); const tauUserSimulatorDefault = document.getElementById("tau-user-simulator-default"); + const tauSimulatorApiKeyLabel = document.getElementById( + "tau-simulator-api-key-label" + ); + const tauSimulatorApiKey = document.getElementById( + "tau-simulator-api-key" + ); const sweAtlasJudgeModelLabel = document.getElementById( "swe-atlas-judge-model-label" ); @@ -665,8 +675,28 @@ export const DASHBOARD_HTML = ` } } + function isDigitalOceanCandidateBaseUrl(value) { + return ( + value === "https://inference.do-ai.run/v1" || + value === "https://inference.do-ai-test.run/v1" + ); + } + + function configureTauSimulatorControls() { + const isTau = startBenchmark.value === "tau_bench_verified_airline"; + const requiresSimulatorApiKey = + isTau && !isDigitalOceanCandidateBaseUrl(inferenceBaseUrl.value); + tauSimulatorApiKeyLabel.hidden = !requiresSimulatorApiKey; + tauSimulatorApiKey.disabled = !requiresSimulatorApiKey; + tauSimulatorApiKey.required = requiresSimulatorApiKey; + if (!requiresSimulatorApiKey) { + tauSimulatorApiKey.value = ""; + } + } + async function configureInferenceControls() { const isOther = inferenceBaseUrl.value === "other"; + configureTauSimulatorControls(); const isOpenRouter = inferenceBaseUrl.value === "https://openrouter.ai/api/v1"; openrouterProviderLabel.hidden = !isOpenRouter; @@ -2693,7 +2723,13 @@ export const DASHBOARD_HTML = ` } } - function openReportRetryPanel(document, runId, item, resultFields) { + function openReportRetryPanel( + document, + runId, + item, + resultFields, + requiresSimulatorApiKey = false + ) { document.getElementById("diagnostic-retry-panel")?.remove(); if (!document.getElementById("diagnostic-retry-style")) { const style = document.createElement("style"); @@ -2738,6 +2774,18 @@ export const DASHBOARD_HTML = ` apiKey.required = true; apiKeyLabel.appendChild(apiKey); form.appendChild(apiKeyLabel); + const simulatorApiKey = document.createElement("input"); + if (requiresSimulatorApiKey) { + const simulatorApiKeyLabel = document.createElement("label"); + simulatorApiKeyLabel.appendChild( + document.createTextNode("DigitalOcean simulator access token") + ); + simulatorApiKey.type = "password"; + simulatorApiKey.autocomplete = "off"; + simulatorApiKey.required = true; + simulatorApiKeyLabel.appendChild(simulatorApiKey); + form.appendChild(simulatorApiKeyLabel); + } const triggerLabel = document.createElement("label"); triggerLabel.appendChild( document.createTextNode("Run-trigger password") @@ -2829,11 +2877,15 @@ export const DASHBOARD_HTML = ` sampleId: item.sampleId, originalEpoch: Number(item.epoch), apiKey: apiKey.value, + ...(requiresSimulatorApiKey + ? { simulatorApiKey: simulatorApiKey.value } + : {}), }), } ); const job = await response.json(); apiKey.value = ""; + simulatorApiKey.value = ""; triggerSecret.value = ""; await poll(job.id); } catch (error) { @@ -4675,16 +4727,22 @@ export const DASHBOARD_HTML = ` retry.title = "Rerun this complete TAU scenario without changing the original score"; retry.addEventListener("click", () => - openReportRetryPanel(document, id, item, (retried) => [ - ["Result", retried.status], - ["Reward", retried.reward], - ["Agent latency", formatLatencyMs(retried.latencyMs)], - ["Termination", retried.terminationReason], - ["Final agent answer", retried.finalAgentAnswer], - ["Actual tool calls", retried.actualToolCalls], - ["Conversation", conversationText(retried.conversation)], - ["Scorer explanation", retried.scorerExplanation], - ]) + openReportRetryPanel( + document, + id, + item, + (retried) => [ + ["Result", retried.status], + ["Reward", retried.reward], + ["Agent latency", formatLatencyMs(retried.latencyMs)], + ["Termination", retried.terminationReason], + ["Final agent answer", retried.finalAgentAnswer], + ["Actual tool calls", retried.actualToolCalls], + ["Conversation", conversationText(retried.conversation)], + ["Scorer explanation", retried.scorerExplanation], + ], + !isDigitalOceanCandidateBaseUrl(report.inference?.baseUrl) + ) ); itemActions.appendChild(retry); card.appendChild(itemActions); @@ -4947,6 +5005,7 @@ export const DASHBOARD_HTML = ` ); setStartFormValue("triggerSecret", ""); setStartFormValue("apiKey", ""); + setStartFormValue("simulatorApiKey", ""); startDialog.showModal(); await Promise.all([ @@ -5031,6 +5090,9 @@ export const DASHBOARD_HTML = ` const pinModel = String(data.get("pinModel") || "").trim(); const maxRetries = String(data.get("maxRetries") || "").trim(); const logLevel = String(data.get("logLevel") || "").trim(); + const simulatorApiKey = String( + data.get("simulatorApiKey") || "" + ).trim(); const isOther = data.get("baseUrl") === "other"; const isOpenRouter = data.get("baseUrl") === "https://openrouter.ai/api/v1"; @@ -5039,6 +5101,8 @@ export const DASHBOARD_HTML = ` triggeredByEmail: String(data.get("triggeredByEmail")), ...(isSweAtlasBenchmark(benchmark) && judgeModel !== "" && { judgeModel }), + ...(benchmark === "tau_bench_verified_airline" && + simulatorApiKey !== "" && { simulatorApiKey }), ...(logLevel === "" ? {} : { logLevel }), inference: { baseUrl: String( @@ -5107,6 +5171,7 @@ export const DASHBOARD_HTML = ` sweAtlasJudgeModel.disabled = !isSweAtlas; sweAtlasJudgeModel.required = isSweAtlas; tauUserSimulatorDefault.hidden = !isTau; + configureTauSimulatorControls(); } const scheduleRunFilterReload = () => { diff --git a/src/server/dashboard.test.ts b/src/server/dashboard.test.ts index 2cd5a3ca..798c72c4 100644 --- a/src/server/dashboard.test.ts +++ b/src/server/dashboard.test.ts @@ -576,7 +576,11 @@ describe("benchmark runs dashboard", () => { expect(body).toContain( "more than one epoch can take a long time to finish and is not recommended" ); - expect(body).toContain("TAU user simulator default: gemini-2.5-flash"); + expect(body).toContain( + "TAU always uses openai-gpt-5.4-mini as the simulated customer" + ); + expect(body).toContain('name="simulatorApiKey"'); + expect(body).toContain("configureTauSimulatorControls"); expect(body).toContain( '...(temperature === "" ? {} : { temperature: Number(temperature) })' ); diff --git a/src/server/index.test.ts b/src/server/index.test.ts index 46c6f706..4991fe80 100644 --- a/src/server/index.test.ts +++ b/src/server/index.test.ts @@ -64,6 +64,7 @@ describe("benchmark API request validation", () => { const tau = parseSchema(RunRequestSchema, { ...base, benchmark: "tau_bench_verified_airline", + simulatorApiKey: "simulator-secret", inference: inferenceWithoutTemperature, }); const deepSwe = parseSchema(RunRequestSchema, { @@ -155,6 +156,38 @@ describe("benchmark API request validation", () => { } }); + it("requires a simulator token only for TAU runs on non-DO endpoints", () => { + for (const baseUrl of [ + "https://inference.do-ai.run/v1", + "https://inference.do-ai-test.run/v1/", + ]) { + const parsed = parseSchema(RunRequestSchema, { + ...validArgs(), + benchmark: "tau_bench_verified_airline", + inference: { ...validArgs().inference, baseUrl }, + }); + expect(Either.isRight(parsed)).toBe(true); + } + + const missing = parseSchema(RunRequestSchema, { + ...validArgs(), + benchmark: "tau_bench_verified_airline", + }); + expect(Either.isLeft(missing)).toBe(true); + + const provided = parseSchema(RunRequestSchema, { + ...validArgs(), + benchmark: "tau_bench_verified_airline", + simulatorApiKey: "simulator-secret", + }); + expect(Either.isRight(provided)).toBe(true); + if (Either.isRight(provided)) { + expect(resolveRunRequest(provided.right).simulatorApiKey).toBe( + "simulator-secret" + ); + } + }); + it("preserves explicit overrides of benchmark defaults", () => { const parsed = parseSchema(RunRequestSchema, { ...validArgs(), @@ -185,6 +218,7 @@ describe("benchmark API request validation", () => { const tauAirline = parseSchema(RunRequestSchema, { ...validArgs(), benchmark: "tau_bench_verified_airline", + simulatorApiKey: "simulator-secret", }); const deepSwe = parseSchema(RunRequestSchema, { ...validArgs(), diff --git a/src/server/index.ts b/src/server/index.ts index 1e3b4405..e79d5553 100644 --- a/src/server/index.ts +++ b/src/server/index.ts @@ -191,6 +191,7 @@ export const RunRequestSchema = z .transform((value) => value.toLowerCase()), inference: InferenceRequestSchema, execution: ExecutionRequestSchema.default({}), + simulatorApiKey: z.string().min(1).optional(), judgeModel: z.string().min(1).regex(IDENTIFIER).optional(), logLevel: z .string() @@ -208,6 +209,18 @@ export const RunRequestSchema = z message: `Sandbox benchmark concurrency cannot exceed ${MAX_SANDBOX_CONCURRENCY}`, }); } + if ( + request.benchmark === "tau_bench_verified_airline" && + !isDigitalOceanInferenceBaseUrl(request.inference.baseUrl) && + request.simulatorApiKey === undefined + ) { + context.addIssue({ + code: "custom", + path: ["simulatorApiKey"], + message: + "simulatorApiKey is required when TAU Airline uses OpenRouter or a custom inference endpoint", + }); + } }); const RetryArmSchema = z.object({ @@ -245,6 +258,7 @@ const ReportRetryRequestSchema = z.object({ sampleId: z.string().min(1).max(500).regex(IDENTIFIER), originalEpoch: z.number().int().nonnegative().max(20), apiKey: z.string().min(1), + simulatorApiKey: z.string().min(1).optional(), }); const ReportRetrySourceSchema = z.object({ @@ -258,6 +272,7 @@ const ReportRetrySourceSchema = z.object({ export function resolveRunRequest(request: z.infer): { readonly apiKey: string; + readonly simulatorApiKey?: string; readonly args: RunArgs; } { const isSandboxBenchmark = usesSandboxWorkers(request.benchmark); @@ -267,6 +282,9 @@ export function resolveRunRequest(request: z.infer): { ); return { apiKey, + ...(request.simulatorApiKey !== undefined && { + simulatorApiKey: request.simulatorApiKey, + }), args: { benchmark: request.benchmark, triggeredByEmail: request.triggeredByEmail, @@ -704,6 +722,7 @@ export async function tauAirlineReportResponse( record.id, result.file, new Uint8Array(readFileSync(result.path)), + record.args.inference, download ); return response; @@ -771,6 +790,7 @@ async function createTauAirlineReportResponse( runId: string, file: string, bytes: Uint8Array, + inference: RunInference, download: boolean ): Promise { const rows = await readResultRows(asyncBufferFromBytes(bytes)); @@ -785,15 +805,18 @@ async function createTauAirlineReportResponse( /[^A-Za-z0-9._-]/gu, "_" ); - return new Response(JSON.stringify({ runId, file, ...report }, null, 2), { - headers: { - "Content-Type": "application/json; charset=utf-8", - "Cache-Control": "private, no-store", - ...(download && { - "Content-Disposition": `attachment; filename="${safeFilename}"`, - }), - }, - }); + return new Response( + JSON.stringify({ runId, file, inference, ...report }, null, 2), + { + headers: { + "Content-Type": "application/json; charset=utf-8", + "Cache-Control": "private, no-store", + ...(download && { + "Content-Disposition": `attachment; filename="${safeFilename}"`, + }), + }, + } + ); } type RemoteRunMetadata = NonNullable< @@ -888,6 +911,7 @@ async function remoteTauAirlineReportResponse( metadata.id, result.file, result.bytes, + metadata.args.inference, download ); } catch (error) { @@ -933,7 +957,7 @@ async function handleCreateRun(request: Request): Promise { if (Either.isLeft(parsed)) { return json({ error: firstZodIssueMessage(parsed.left) }, 400); } - const { apiKey, args } = resolveRunRequest(parsed.right); + const { apiKey, simulatorApiKey, args } = resolveRunRequest(parsed.right); const rangeError = validateRange(args); if (rangeError !== null) { return json({ error: rangeError }, 400); @@ -942,6 +966,7 @@ async function handleCreateRun(request: Request): Promise { return json( await startRun(args, { apiKey, + ...(simulatorApiKey !== undefined && { simulatorApiKey }), maxActiveRuns: maxActiveRuns(), }), 202 @@ -1358,6 +1383,19 @@ export async function handleRequest(request: Request): Promise { if (Either.isLeft(parsed)) { return json({ error: firstZodIssueMessage(parsed.left) }, 400); } + if ( + metadata.args.benchmark === "tau_bench_verified_airline" && + !isDigitalOceanInferenceBaseUrl(metadata.args.inference.baseUrl) && + parsed.right.simulatorApiKey === undefined + ) { + return json( + { + error: + "simulatorApiKey is required when TAU Airline uses OpenRouter or a custom inference endpoint", + }, + 400 + ); + } const localRecord = getRun(runId); const reportResponse = await reportRetrySourceResponse( metadata, @@ -1399,6 +1437,9 @@ export async function handleRequest(request: Request): Promise { sampleId: parsed.right.sampleId, originalEpoch: parsed.right.originalEpoch, apiKey: parsed.right.apiKey, + ...(parsed.right.simulatorApiKey !== undefined && { + simulatorApiKey: parsed.right.simulatorApiKey, + }), }), 202 ); diff --git a/src/server/model-catalog.ts b/src/server/model-catalog.ts index f1447c69..038f2977 100644 --- a/src/server/model-catalog.ts +++ b/src/server/model-catalog.ts @@ -1,10 +1,10 @@ import { Either } from "../internal/either"; import { firstZodIssueMessage, parseSchema, z } from "../internal/zod"; -export const DIGITALOCEAN_INFERENCE_BASE_URLS = [ - "https://inference.do-ai.run/v1", - "https://inference.do-ai-test.run/v1", -] as const; +export { + DIGITALOCEAN_INFERENCE_BASE_URLS, + isDigitalOceanInferenceBaseUrl, +} from "../providers/digitalocean-inference"; export const OPENROUTER_INFERENCE_BASE_URL = "https://openrouter.ai/api/v1"; const CATALOG_URL = "https://api.digitalocean.com/v2/gen-ai/models/catalog"; @@ -70,14 +70,6 @@ function timestamp(value: string | undefined): number { return value === undefined ? Number.NEGATIVE_INFINITY : Date.parse(value); } -export function isDigitalOceanInferenceBaseUrl( - value: string -): value is (typeof DIGITALOCEAN_INFERENCE_BASE_URLS)[number] { - return DIGITALOCEAN_INFERENCE_BASE_URLS.includes( - value as (typeof DIGITALOCEAN_INFERENCE_BASE_URLS)[number] - ); -} - export function isOpenRouterInferenceBaseUrl(value: string): boolean { return value === OPENROUTER_INFERENCE_BASE_URL; } diff --git a/src/server/report-retry.ts b/src/server/report-retry.ts index dd632232..214555d1 100644 --- a/src/server/report-retry.ts +++ b/src/server/report-retry.ts @@ -173,6 +173,7 @@ export function startReportRetry(input: { readonly sampleId: string; readonly originalEpoch: number; readonly apiKey: string; + readonly simulatorApiKey?: string; }): ReportRetryJob { if (!supportsReportRetry(input.args.benchmark)) { throw new Error( @@ -222,7 +223,8 @@ export function startReportRetry(input: { input.apiKey, id, requestLogPath, - resultsDir + resultsDir, + input.simulatorApiKey ), stdin: "ignore", stdout: descriptor, diff --git a/src/server/run-registry.test.ts b/src/server/run-registry.test.ts index a39e5bc6..705d2895 100644 --- a/src/server/run-registry.test.ts +++ b/src/server/run-registry.test.ts @@ -16,6 +16,10 @@ const originalMysqlPassword = process.env["MYSQL_PASSWORD"]; const originalSweAtlasJudgeKey = process.env["SWE_ATLAS_JUDGE_API_KEY"]; const originalSweAtlasJudgeBaseUrl = process.env["SWE_ATLAS_JUDGE_BASE_URL"]; const originalSweAtlasJudgeModel = process.env["SWE_ATLAS_JUDGE_MODEL"]; +const originalTauSimulatorModel = + process.env["TAU_AIRLINE_USER_SIMULATOR_MODEL"]; +const originalTauSimulatorBaseUrl = + process.env["TAU_AIRLINE_USER_SIMULATOR_BASE_URL"]; afterEach(() => { process.env["SPACES_SECRET_ACCESS_KEY"] = originalSpacesSecret; @@ -25,6 +29,9 @@ afterEach(() => { process.env["SWE_ATLAS_JUDGE_API_KEY"] = originalSweAtlasJudgeKey; process.env["SWE_ATLAS_JUDGE_BASE_URL"] = originalSweAtlasJudgeBaseUrl; process.env["SWE_ATLAS_JUDGE_MODEL"] = originalSweAtlasJudgeModel; + process.env["TAU_AIRLINE_USER_SIMULATOR_MODEL"] = originalTauSimulatorModel; + process.env["TAU_AIRLINE_USER_SIMULATOR_BASE_URL"] = + originalTauSimulatorBaseUrl; }); function args(): RunArgs { @@ -228,6 +235,46 @@ describe("GPQA child invocation", () => { expect(env["MYSQL_PASSWORD"]).toBeUndefined(); }); + it("routes TAU simulator credentials through DigitalOcean inference", () => { + const tauArgs: RunArgs = { + ...args(), + benchmark: "tau_bench_verified_airline", + }; + for (const baseUrl of [ + "https://inference.do-ai.run/v1", + "https://inference.do-ai-test.run/v1", + ]) { + const env = childEnvironment( + { ...tauArgs, inference: { ...tauArgs.inference, baseUrl } }, + "candidate-secret", + "run-id", + "requests.jsonl", + "results", + "separate-secret" + ); + expect(env["TAU_AIRLINE_USER_SIMULATOR_API_KEY"]).toBe( + "candidate-secret" + ); + } + + process.env["TAU_AIRLINE_USER_SIMULATOR_MODEL"] = "ignored-model"; + process.env["TAU_AIRLINE_USER_SIMULATOR_BASE_URL"] = + "https://ignored.example/v1"; + const externalEnv = childEnvironment( + tauArgs, + "candidate-secret", + "run-id", + "requests.jsonl", + "results", + "separate-secret" + ); + expect(externalEnv["TAU_AIRLINE_USER_SIMULATOR_API_KEY"]).toBe( + "separate-secret" + ); + expect(externalEnv["TAU_AIRLINE_USER_SIMULATOR_MODEL"]).toBeUndefined(); + expect(externalEnv["TAU_AIRLINE_USER_SIMULATOR_BASE_URL"]).toBeUndefined(); + }); + it("injects SWE Atlas judge credentials only into Atlas children", () => { process.env["SWE_ATLAS_JUDGE_API_KEY"] = "judge-secret"; process.env["SWE_ATLAS_JUDGE_BASE_URL"] = "https://judge.example/v1"; diff --git a/src/server/run-registry.ts b/src/server/run-registry.ts index 57f07922..41ad39a5 100644 --- a/src/server/run-registry.ts +++ b/src/server/run-registry.ts @@ -14,6 +14,7 @@ import { join } from "node:path"; import { eLog, iLog, wLog } from "../internal/log"; import { z } from "../internal/zod"; +import { isDigitalOceanInferenceBaseUrl } from "../providers/digitalocean-inference"; import { asyncBufferFromBytes, readResultRows, @@ -717,7 +718,8 @@ export function childEnvironment( apiKey: string, id: string, requestLogPath: string, - resultsDir: string + resultsDir: string, + simulatorApiKey?: string ): Record { const excluded = new Set([ "BENCH_API_TOKEN", @@ -727,9 +729,6 @@ export function childEnvironment( "REQUEST_LOG_CONSOLE", "SPACES_ACCESS_KEY_ID", "SPACES_SECRET_ACCESS_KEY", - "TAU_AIRLINE_USER_SIMULATOR_API_KEY", - "TAU_AIRLINE_USER_SIMULATOR_BASE_URL", - "TAU_AIRLINE_USER_SIMULATOR_MODEL", "SWE_ATLAS_JUDGE_API_KEY", "SWE_ATLAS_JUDGE_BASE_URL", "SWE_ATLAS_JUDGE_MODEL", @@ -738,6 +737,7 @@ export function childEnvironment( Object.entries(process.env).filter( ([name]) => !excluded.has(name) && + !name.startsWith("TAU_AIRLINE_USER_SIMULATOR_") && !name.startsWith("MYSQL_") && !name.startsWith("SPACES_") ) @@ -751,12 +751,11 @@ export function childEnvironment( BENCH_PROGRESS_FILE: join(resultsDir, "..", "progress.json"), REQUEST_LOG_FILE: requestLogPath, ...(args.benchmark === "tau_bench_verified_airline" && { - TAU_AIRLINE_USER_SIMULATOR_API_KEY: - process.env["TAU_AIRLINE_USER_SIMULATOR_API_KEY"], - TAU_AIRLINE_USER_SIMULATOR_BASE_URL: - process.env["TAU_AIRLINE_USER_SIMULATOR_BASE_URL"], - TAU_AIRLINE_USER_SIMULATOR_MODEL: - process.env["TAU_AIRLINE_USER_SIMULATOR_MODEL"], + TAU_AIRLINE_USER_SIMULATOR_API_KEY: isDigitalOceanInferenceBaseUrl( + args.inference.baseUrl + ) + ? apiKey + : simulatorApiKey, }), ...(isSweAtlasBenchmark(args.benchmark) && { SWE_ATLAS_JUDGE_API_KEY: process.env["SWE_ATLAS_JUDGE_API_KEY"], @@ -1070,6 +1069,7 @@ export async function startRun( args: RunArgs, options: { readonly apiKey: string; + readonly simulatorApiKey?: string; readonly maxActiveRuns: number; } ): Promise { @@ -1148,7 +1148,8 @@ export async function startRun( options.apiKey, id, requestLogPath, - resultsDir + resultsDir, + options.simulatorApiKey ), stdin: "ignore", stdout: fd,