diff --git a/docs/config/providers.mdx b/docs/config/providers.mdx index 42c60efe2f7..bd60e921a7e 100644 --- a/docs/config/providers.mdx +++ b/docs/config/providers.mdx @@ -73,6 +73,27 @@ Azure OpenAI env vars configure the OpenAI provider with Azure backend. {/* END PROVIDER_ENV_VARS */} +## AI Calls from Bash Commands + +Scripts that the agent runs in the bash tool (test harnesses, `make bug-bash`, SDK scripts) can call Anthropic and OpenAI directly. Turn on **Settings → Providers → Bash commands** to count that spend (it is off by default). Each bash command then gets a local endpoint and a workspace key, and Xum adds the usage to that workspace's Costs tab and to Analytics (source `headless:bash_proxy`). + +| Variable | Value | +| ------------------------------------------------------------- | ------------------------------------- | +| `ANTHROPIC_BASE_URL` | `http://127.0.0.1:/anthropic` | +| `OPENAI_BASE_URL` | `http://127.0.0.1:/openai/v1` | +| `ANTHROPIC_API_KEY`, `ANTHROPIC_AUTH_TOKEN`, `OPENAI_API_KEY` | `xum-proxy-…` (one key per workspace) | + +- Xum sends the calls with the provider keys from this page. The command never sees the real key. +- Only Local and Worktree commands get the variables. SSH, Coder, Docker and devcontainer commands keep their environment. +- Bash commands in `xum run` and `xum workflow` sessions do not get the variables. +- Commands in untrusted projects, and every command while `XUM_DISABLE_PROJECT_AUTOMATION=1` is set, get no variables: Xum blanks provider keys there, so repo code cannot spend through the proxy. Revoking a project's trust also refuses calls from its commands that are still running. +- A provider without a Xum API key (for example OpenAI with Codex OAuth only) gets no variables. A provider that your [project secrets](/config/project-secrets) configure keeps the secret values. This includes `OPENAI_ORG_ID` and `OPENAI_PROJECT_ID`. +- The proxy allows Messages, Responses, Chat Completions, token counting and model listing. Other paths get 404. +- The port and keys change when Xum restarts. A background command started before the restart gets connection errors. Start it again. +- A key stops working when its workspace is removed. +- A script that exports its own `ANTHROPIC_API_KEY` but keeps the proxy URL gets 401. Set both the key and the base URL, or neither. +- Agent CLIs that read these variables also go through the proxy. `claude -p` in a bash command then bills the Xum Anthropic key instead of a Claude subscription login. + ## Advanced: Manual Configuration For advanced options not exposed in the UI, edit `~/.xum/providers.jsonc` directly: diff --git a/src/browser/features/Settings/Sections/ProvidersSection.stories.tsx b/src/browser/features/Settings/Sections/ProvidersSection.stories.tsx index d1044833cb5..95b110df6ae 100644 --- a/src/browser/features/Settings/Sections/ProvidersSection.stories.tsx +++ b/src/browser/features/Settings/Sections/ProvidersSection.stories.tsx @@ -218,3 +218,24 @@ export const CoderModelRouting: Story = { await canvas.findByText(/openai-removed is not a known OpenAI provider/); }, }; + +/** The Bash commands switch saves through the API and stays on. */ +export const BashCommandsSwitch: Story = { + render: () => ( + setupSettingsStory({ providersConfig: {} })}> + + + ), + play: async ({ canvasElement }) => { + const canvas = within(canvasElement); + const toggle = await canvas.findByRole( + "switch", + { name: "Count AI calls from bash commands" }, + { timeout: 5000 } + ); + await waitFor(() => expect(toggle).toHaveAttribute("aria-checked", "false")); + await userEvent.click(toggle); + // A failed save rolls the switch back. + await waitFor(() => expect(toggle).toHaveAttribute("aria-checked", "true")); + }, +}; diff --git a/src/browser/features/Settings/Sections/ProvidersSection.tsx b/src/browser/features/Settings/Sections/ProvidersSection.tsx index 425acd79bfe..85bc30138b2 100644 --- a/src/browser/features/Settings/Sections/ProvidersSection.tsx +++ b/src/browser/features/Settings/Sections/ProvidersSection.tsx @@ -33,6 +33,7 @@ import { useWorkspaceContext } from "@/browser/contexts/WorkspaceContext"; import { ProviderIcon, ProviderWithIcon } from "@/browser/components/ProviderIcon/ProviderIcon"; import { getStoredAuthToken } from "@/browser/components/AuthTokenModal/AuthTokenModal"; import { useAPI } from "@/browser/contexts/API"; +import { ConfigSwitchSetting, type SettingsApi } from "./ConfigSwitchSetting"; import { useSettings } from "@/browser/contexts/SettingsContext"; import { useProvidersConfig } from "@/browser/hooks/useProvidersConfig"; import { @@ -463,6 +464,11 @@ function GatewayRoutePriorityList({ ); } +const loadBashAiProxyEnabled = async (api: SettingsApi) => + (await api.config.getConfig()).bashAiProxyEnabled; +const saveBashAiProxyEnabled = (api: SettingsApi, enabled: boolean) => + api.config.updateBashAiProxyEnabled({ enabled }); + export function ProvidersSection() { const { providersExpandedProvider, @@ -3444,6 +3450,19 @@ export function ProvidersSection() { )} +
+
Bash commands
+ +
+ {config && !hasAnyConfiguredProvider && (
No providers are currently enabled. You won't be able to send messages until you diff --git a/src/browser/stories/mocks/orpc.ts b/src/browser/stories/mocks/orpc.ts index cdab7820cb9..3ee513c560d 100644 --- a/src/browser/stories/mocks/orpc.ts +++ b/src/browser/stories/mocks/orpc.ts @@ -600,6 +600,7 @@ export function createMockORPCClient(options: MockORPCClientOptions = {}): APICl let chatTranscriptFullWidth = initialChatTranscriptFullWidth; let keepScreenAwake = initialKeepScreenAwake; let toolSearchEnabled = true; + let bashAiProxyEnabled = false; let agentHeartbeatsEnabled = false; let runtimeEnablement: Record = initialRuntimeEnablement ?? { local: true, @@ -850,6 +851,7 @@ export function createMockORPCClient(options: MockORPCClientOptions = {}): APICl llmDebugLogs: false, keepScreenAwake, toolSearchEnabled, + bashAiProxyEnabled, agentHeartbeatsEnabled, }), saveConfig: (input: { @@ -963,6 +965,11 @@ export function createMockORPCClient(options: MockORPCClientOptions = {}): APICl notifyConfigChanged(); return Promise.resolve(undefined); }, + updateBashAiProxyEnabled: (input: { enabled: boolean }) => { + bashAiProxyEnabled = input.enabled; + notifyConfigChanged(); + return Promise.resolve(undefined); + }, updateAgentHeartbeatsEnabled: (input: { enabled: boolean }) => { agentHeartbeatsEnabled = input.enabled; notifyConfigChanged(); diff --git a/src/browser/testUtils.ts b/src/browser/testUtils.ts index 9beccadb7f9..67cd7a68dd0 100644 --- a/src/browser/testUtils.ts +++ b/src/browser/testUtils.ts @@ -161,6 +161,7 @@ export function createTestConfig(overrides: Partial = {}): Tes llmDebugLogs: false, keepScreenAwake: false, toolSearchEnabled: true, + bashAiProxyEnabled: false, agentHeartbeatsEnabled: false, goalDefaults: DEFAULT_GOAL_DEFAULTS, ...overrides, diff --git a/src/browser/utils/commandIds.ts b/src/browser/utils/commandIds.ts index cdb91e45590..05ada997720 100644 --- a/src/browser/utils/commandIds.ts +++ b/src/browser/utils/commandIds.ts @@ -105,6 +105,7 @@ export const CommandIds = { settingsOpen: () => "settings:open" as const, settingsOpenSection: (section: string) => `settings:open:${section}` as const, settingsToggleKeepScreenAwake: () => "settings:toggle-keep-screen-awake" as const, + settingsToggleBashAiProxy: () => "settings:toggle-bash-ai-proxy" as const, openServerWindow: () => "remote-connection:open-server-window" as const, coderDisconnect: () => "providers:coder:disconnect" as const, coderRefreshModels: () => "providers:coder:refresh-models" as const, diff --git a/src/browser/utils/commands/sources.test.ts b/src/browser/utils/commands/sources.test.ts index cdfe9f71e58..810c14857c5 100644 --- a/src/browser/utils/commands/sources.test.ts +++ b/src/browser/utils/commands/sources.test.ts @@ -1352,6 +1352,30 @@ test("toggle keep screen awake command inverts the persisted config flag", async } }); +test("toggle bash AI proxy command inverts the persisted config flag", async () => { + let bashAiProxyEnabled = false; + const updateBashAiProxyEnabled = mock((input: { enabled: boolean }) => { + bashAiProxyEnabled = input.enabled; + return Promise.resolve(); + }); + const actions = getActions({ + api: createTestApiClient({ + config: { + getConfig: () => Promise.resolve(createTestConfig({ bashAiProxyEnabled })), + updateBashAiProxyEnabled, + }, + }), + }); + const toggleAction = actions.find((a) => a.id === "settings:toggle-bash-ai-proxy"); + + expect(toggleAction).toBeDefined(); + await toggleAction!.run(); + expect(updateBashAiProxyEnabled).toHaveBeenLastCalledWith({ enabled: true }); + await toggleAction!.run(); + expect(updateBashAiProxyEnabled).toHaveBeenLastCalledWith({ enabled: false }); + expect(bashAiProxyEnabled).toBe(false); +}); + test("analytics rebuild command calls route and dispatches toast feedback", async () => { const rebuildDatabase = mock(() => Promise.resolve({ success: true, workspacesIngested: 4 })); diff --git a/src/browser/utils/commands/sources.ts b/src/browser/utils/commands/sources.ts index fb6c436f29a..37036b6cdfa 100644 --- a/src/browser/utils/commands/sources.ts +++ b/src/browser/utils/commands/sources.ts @@ -2043,6 +2043,23 @@ export function buildCoreSources(p: BuildSourcesParams): Array<() => CommandActi ] ); + // Keyboard route for the Settings → Providers → Bash commands switch. + actions.push(() => [ + { + id: CommandIds.settingsToggleBashAiProxy(), + title: "Toggle Count AI Calls from Bash Commands", + subtitle: "Route bash commands' Anthropic and OpenAI calls through Xum to count their cost", + section: section.settings, + keywords: ["bash", "proxy", "cost", "usage", "anthropic", "openai", "billing"], + run: async () => { + if (!p.api) return; + // The flag lives in config.json (not localStorage), so read the current value first. + const cfg = await p.api.config.getConfig(); + await p.api.config.updateBashAiProxyEnabled({ enabled: !cfg.bashAiProxyEnabled }); + }, + }, + ]); + // Settings if (p.onOpenSettings) { const openSettings = p.onOpenSettings; diff --git a/src/common/config/schemas/appConfigOnDisk.ts b/src/common/config/schemas/appConfigOnDisk.ts index cb050e740fa..5d47729e3b2 100644 --- a/src/common/config/schemas/appConfigOnDisk.ts +++ b/src/common/config/schemas/appConfigOnDisk.ts @@ -177,6 +177,8 @@ export const AppConfigOnDiskSchema = z keepScreenAwake: z.boolean().optional(), /** Defer MCP tool definitions behind tool_catalog_search. Absent = on. */ toolSearchEnabled: z.boolean().optional(), + /** Route AI calls from bash tool commands through Xum's cost-tracking proxy. Absent = off. */ + bashAiProxyEnabled: z.boolean().optional(), /** Expose the `heartbeat` tool so agents can schedule their own recurring turns. Absent = off. */ agentHeartbeatsEnabled: z.boolean().optional(), heartbeatDefaultPrompt: z.string().optional(), diff --git a/src/common/orpc/schemas/api.ts b/src/common/orpc/schemas/api.ts index 8937b6dbbc3..b8e094df85f 100644 --- a/src/common/orpc/schemas/api.ts +++ b/src/common/orpc/schemas/api.ts @@ -3149,6 +3149,7 @@ export const config = { llmDebugLogs: z.boolean(), keepScreenAwake: z.boolean(), toolSearchEnabled: z.boolean(), + bashAiProxyEnabled: z.boolean(), agentHeartbeatsEnabled: z.boolean(), heartbeatDefaultPrompt: z.string().optional(), heartbeatDefaultIntervalMs: z.number().optional(), @@ -3287,6 +3288,7 @@ export const config = { updateLlmDebugLogs: booleanToggleRoute, updateKeepScreenAwake: booleanToggleRoute, updateToolSearchEnabled: booleanToggleRoute, + updateBashAiProxyEnabled: booleanToggleRoute, updateAgentHeartbeatsEnabled: booleanToggleRoute, updateHeartbeatDefaultPrompt: { input: z diff --git a/src/common/types/project.ts b/src/common/types/project.ts index aa96fe5d5d9..ef7a7736835 100644 --- a/src/common/types/project.ts +++ b/src/common/types/project.ts @@ -104,6 +104,12 @@ export interface ProjectsConfig { * discovers them via tool_catalog_search. Absent = on; only `false` disables it. */ toolSearchEnabled?: boolean; + /** + * Route Anthropic/OpenAI calls from bash tool commands through Xum's local proxy so their + * spend shows in the Costs tab and Analytics. Absent = off: with it on, agent CLIs such as + * `claude -p` bill the Xum API key instead of a subscription login. + */ + bashAiProxyEnabled?: boolean; /** * Expose the `heartbeat` tool so agents can schedule their own recurring (paid) turns. * Absent = off; users opt in from Settings. diff --git a/src/node/config.test.ts b/src/node/config.test.ts index e54c877d1e8..a71ff9dce17 100644 --- a/src/node/config.test.ts +++ b/src/node/config.test.ts @@ -622,6 +622,23 @@ describe("Config", () => { }); }); + describe("bash AI proxy setting", () => { + // Off by default: with the proxy vars set, `claude -p` bills the API key, not a subscription. + it("is off until the user opts in, and opting out removes the key", async () => { + expect(config.loadConfigOrDefault().bashAiProxyEnabled).toBeUndefined(); + + await config.updateBashAiProxyEnabled(true); + expect(new Config(tempDir).loadConfigOrDefault().bashAiProxyEnabled).toBe(true); + + await config.updateBashAiProxyEnabled(false); + const persisted = JSON.parse(fs.readFileSync(path.join(tempDir, "config.json"), "utf-8")) as { + bashAiProxyEnabled?: boolean; + }; + expect(persisted.bashAiProxyEnabled).toBeUndefined(); + expect(new Config(tempDir).loadConfigOrDefault().bashAiProxyEnabled).toBeUndefined(); + }); + }); + describe("persistent sub-agent retention migration", () => { it.each([ ["missing", undefined], diff --git a/src/node/config/index.ts b/src/node/config/index.ts index 6129bc5fa94..52f8987ee01 100644 --- a/src/node/config/index.ts +++ b/src/node/config/index.ts @@ -2284,6 +2284,7 @@ export class Config { llmDebugLogs: parseOptionalBoolean(parsed.llmDebugLogs), keepScreenAwake: parseOptionalBoolean(parsed.keepScreenAwake), toolSearchEnabled: parseOptionalBoolean(parsed.toolSearchEnabled), + bashAiProxyEnabled: parseOptionalBoolean(parsed.bashAiProxyEnabled), agentHeartbeatsEnabled: parseOptionalBoolean(parsed.agentHeartbeatsEnabled), heartbeatDefaultPrompt: parseOptionalNonEmptyString(parsed.heartbeatDefaultPrompt), heartbeatDefaultIntervalMs: parseOptionalHeartbeatIntervalMs( @@ -2406,6 +2407,11 @@ export class Config { data.toolSearchEnabled = false; } + // Default-off flag: only the opt-in is written. + if (parseOptionalBoolean(config.bashAiProxyEnabled) === true) { + data.bashAiProxyEnabled = true; + } + if (parseOptionalBoolean(config.agentHeartbeatsEnabled) === true) { data.agentHeartbeatsEnabled = true; } @@ -2833,6 +2839,7 @@ export class Config { llmDebugLogs: config.llmDebugLogs === true, keepScreenAwake: config.keepScreenAwake === true, toolSearchEnabled: config.toolSearchEnabled !== false, + bashAiProxyEnabled: config.bashAiProxyEnabled === true, agentHeartbeatsEnabled: config.agentHeartbeatsEnabled === true, heartbeatDefaultPrompt: config.heartbeatDefaultPrompt ?? undefined, heartbeatDefaultIntervalMs: config.heartbeatDefaultIntervalMs ?? undefined, @@ -2887,6 +2894,14 @@ export class Config { }); } + async updateBashAiProxyEnabled(enabled: boolean): Promise { + await this.editConfig((config) => { + if (enabled) config.bashAiProxyEnabled = true; + else delete config.bashAiProxyEnabled; + return config; + }); + } + async updateAgentHeartbeatsEnabled(enabled: boolean): Promise { await this.editConfig((config) => { if (enabled) config.agentHeartbeatsEnabled = true; diff --git a/src/node/orpc/router.ts b/src/node/orpc/router.ts index 569ed95ec51..9c93991d64f 100644 --- a/src/node/orpc/router.ts +++ b/src/node/orpc/router.ts @@ -574,6 +574,16 @@ export const router = (authToken?: string) => { yield* atomicPromise(async () => context.config.updateToolSearchEnabled(input.enabled)); }) ), + updateBashAiProxyEnabled: t + .input(schemas.config.updateBashAiProxyEnabled.input) + .output(schemas.config.updateBashAiProxyEnabled.output) + .handler( + handlerGen(function* ({ context }, input) { + yield* atomicPromise(async () => + context.config.updateBashAiProxyEnabled(input.enabled) + ); + }) + ), updateAgentHeartbeatsEnabled: t .input(schemas.config.updateAgentHeartbeatsEnabled.input) .output(schemas.config.updateAgentHeartbeatsEnabled.output) diff --git a/src/node/services/agentSkills/builtInSkillContent.generated.ts b/src/node/services/agentSkills/builtInSkillContent.generated.ts index ca15a42240c..fef00e63b33 100644 --- a/src/node/services/agentSkills/builtInSkillContent.generated.ts +++ b/src/node/services/agentSkills/builtInSkillContent.generated.ts @@ -5014,6 +5014,27 @@ export const BUILTIN_SKILL_FILES: Record> = { "", "{/* END PROVIDER_ENV_VARS */}", "", + "## AI Calls from Bash Commands", + "", + "Scripts that the agent runs in the bash tool (test harnesses, `make bug-bash`, SDK scripts) can call Anthropic and OpenAI directly. Turn on **Settings → Providers → Bash commands** to count that spend (it is off by default). Each bash command then gets a local endpoint and a workspace key, and Xum adds the usage to that workspace's Costs tab and to Analytics (source `headless:bash_proxy`).", + "", + "| Variable | Value |", + "| ------------------------------------------------------------- | ------------------------------------- |", + "| `ANTHROPIC_BASE_URL` | `http://127.0.0.1:/anthropic` |", + "| `OPENAI_BASE_URL` | `http://127.0.0.1:/openai/v1` |", + "| `ANTHROPIC_API_KEY`, `ANTHROPIC_AUTH_TOKEN`, `OPENAI_API_KEY` | `xum-proxy-…` (one key per workspace) |", + "", + "- Xum sends the calls with the provider keys from this page. The command never sees the real key.", + "- Only Local and Worktree commands get the variables. SSH, Coder, Docker and devcontainer commands keep their environment.", + "- Bash commands in `xum run` and `xum workflow` sessions do not get the variables.", + "- Commands in untrusted projects, and every command while `XUM_DISABLE_PROJECT_AUTOMATION=1` is set, get no variables: Xum blanks provider keys there, so repo code cannot spend through the proxy. Revoking a project's trust also refuses calls from its commands that are still running.", + "- A provider without a Xum API key (for example OpenAI with Codex OAuth only) gets no variables. A provider that your [project secrets](/config/project-secrets) configure keeps the secret values. This includes `OPENAI_ORG_ID` and `OPENAI_PROJECT_ID`.", + "- The proxy allows Messages, Responses, Chat Completions, token counting and model listing. Other paths get 404.", + "- The port and keys change when Xum restarts. A background command started before the restart gets connection errors. Start it again.", + "- A key stops working when its workspace is removed.", + "- A script that exports its own `ANTHROPIC_API_KEY` but keeps the proxy URL gets 401. Set both the key and the base URL, or neither.", + "- Agent CLIs that read these variables also go through the proxy. `claude -p` in a bash command then bills the Xum Anthropic key instead of a Claude subscription login.", + "", "## Advanced: Manual Configuration", "", "For advanced options not exposed in the UI, edit `~/.xum/providers.jsonc` directly:", diff --git a/src/node/services/aiService.test.ts b/src/node/services/aiService.test.ts index 8735b438f52..7c8a734bc8b 100644 --- a/src/node/services/aiService.test.ts +++ b/src/node/services/aiService.test.ts @@ -4786,6 +4786,41 @@ describe("AIService.streamMessage multi-project trust gating", () => { expect(trustedFromFirstGetToolsCall(harness.getToolsForModelSpy)).toBe(false); }); + // Untrusted repos and the project-automation kill switch get blanked provider keys in bash, so + // the bash AI proxy must not hand them a proxy URL either (it would pair with a blank key). + for (const [label, trusted, killSwitch, expected] of [ + ["trusted project", true, false, "http://proxy"], + ["untrusted project", false, false, undefined], + ["project automation kill switch", true, true, undefined], + ] as const) { + it(`adds bash AI proxy variables only for shared-trusted execution (${label})`, async () => { + using xumHome = new DisposableTempDir("ai-service-bash-ai-proxy-trust"); + const projectPath = path.join(xumHome.path, "project-a"); + await fs.mkdir(projectPath, { recursive: true }); + const workspaceId = "workspace-bash-ai-proxy-trust"; + const harness = createHarness(xumHome.path, createTrustMetadata(workspaceId, [projectPath])); + harness.service.turnRequestBuilderBindings.bashAiProxy = { + envFor: () => Promise.resolve({ ANTHROPIC_BASE_URL: "http://proxy" }), + }; + await harness.config.editConfig((cfg) => { + cfg.projects.set(projectPath, { workspaces: [], trusted }); + return cfg; + }); + const previous = process.env.XUM_DISABLE_PROJECT_AUTOMATION; + if (killSwitch) process.env.XUM_DISABLE_PROJECT_AUTOMATION = "1"; + try { + await streamOnce(harness, workspaceId); + } finally { + if (previous === undefined) delete process.env.XUM_DISABLE_PROJECT_AUTOMATION; + else process.env.XUM_DISABLE_PROJECT_AUTOMATION = previous; + } + const toolConfig = harness.getToolsForModelSpy.mock.calls[0]?.[1] as + | { xumEnv?: Record } + | undefined; + expect(toolConfig?.xumEnv?.ANTHROPIC_BASE_URL).toBe(expected); + }); + } + it("uses the persisted workspace root as cwd for multi-project ssh startup", async () => { using xumHome = new DisposableTempDir("ai-service-multi-project-persisted-cwd"); const projectAPath = path.join(xumHome.path, "project-a"); diff --git a/src/node/services/bashAiProxy/bashAiProxyService.test.ts b/src/node/services/bashAiProxy/bashAiProxyService.test.ts new file mode 100644 index 00000000000..d1f212dbb3d --- /dev/null +++ b/src/node/services/bashAiProxy/bashAiProxyService.test.ts @@ -0,0 +1,327 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as http from "node:http"; +import type { AddressInfo } from "node:net"; + +import type { ChatUsageDisplay } from "@/common/utils/tokens/usageAggregator"; +import type { AiSdkUsageLike } from "@/common/utils/tokens/usageHelpers"; +import type { ProviderConfigRaw } from "@/node/utils/providerRequirements"; + +import { BashAiProxyService } from "./bashAiProxyService"; + +interface SeenRequest { + method: string; + url: string; + headers: http.IncomingHttpHeaders; + body: string; +} + +const zero = { tokens: 0 }; +const emptyDisplayUsage: ChatUsageDisplay = { + input: zero, + cached: zero, + cacheCreate: zero, + output: zero, + reasoning: zero, +}; + +interface RecordCall { + workspaceId: string; + modelString: string; + usage: AiSdkUsageLike; +} + +const anthropicSse = [ + `event: message_start\ndata: ${JSON.stringify({ + type: "message_start", + message: { model: "claude-sonnet-5-5", usage: { input_tokens: 11, output_tokens: 1 } }, + })}\n\n`, + `event: message_delta\ndata: ${JSON.stringify({ type: "message_delta", usage: { output_tokens: 4 } })}\n\n`, +].join(""); + +/** A loopback stand-in for the provider: records each request and answers by path. */ +async function startUpstream() { + const seen: SeenRequest[] = []; + const server = http.createServer((req, res) => { + let body = ""; + req.on("data", (chunk: Buffer) => (body += chunk.toString())); + req.on("end", () => { + seen.push({ method: req.method ?? "", url: req.url ?? "", headers: req.headers, body }); + if (req.url === "/v1/messages") { + res.writeHead(200, { "content-type": "text/event-stream" }); + res.end(anthropicSse); + } else if (req.url === "/v1/chat/completions") { + res.writeHead(200, { "content-type": "application/json" }); + res.end( + JSON.stringify({ model: "gpt-6.1", usage: { prompt_tokens: 3, completion_tokens: 2 } }) + ); + } else if (req.url === "/v1/responses") { + res.writeHead(307, { location: "https://elsewhere.example/v1/responses" }); + res.end(); + } else { + res.writeHead(404); + res.end(); + } + }); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const port = (server.address() as AddressInfo).port; + return { seen, server, baseUrl: `http://127.0.0.1:${port}/v1` }; +} + +describe("BashAiProxyService", () => { + let upstream: Awaited>; + let proxy: BashAiProxyService; + let enabled: boolean; + let removed: Set; + let configs: Record; + let recorded: RecordCall[]; + let liveDeltas: string[]; + let untrusted: Set; + + beforeEach(async () => { + upstream = await startUpstream(); + enabled = true; + configs = { + anthropic: { apiKey: "real-anthropic-key", baseUrl: upstream.baseUrl }, + openai: { + apiKey: "real-openai-key", + baseUrl: upstream.baseUrl, + headers: { "x-gateway-team": "qa", Authorization: "Bearer not-this-one" }, + }, + }; + recorded = []; + liveDeltas = []; + removed = new Set(); + untrusted = new Set(); + proxy = makeProxy(); + }); + + function makeProxy(): BashAiProxyService { + return new BashAiProxyService({ + isEnabled: () => enabled, + workspaceExists: (workspaceId) => !removed.has(workspaceId), + isWorkspaceTrusted: (workspaceId) => Promise.resolve(!untrusted.has(workspaceId)), + loadProviderConfig: (provider) => configs[provider] ?? {}, + recordUsage: (workspaceId, modelString, usage) => { + recorded.push({ workspaceId, modelString, usage }); + return Promise.resolve( + workspaceId === "removed-ws" + ? undefined + : { model: modelString, usage: emptyDisplayUsage } + ); + }, + onUsageRecorded: (workspaceId) => liveDeltas.push(workspaceId), + }); + } + + afterEach(async () => { + await proxy.stop(); + upstream.server.close(); + }); + + test("both Anthropic SDK path styles reach /v1/messages with the Xum key and record once each", async () => { + const env = await proxy.envFor("ws-1", "worktree", []); + const base = env.ANTHROPIC_BASE_URL; + expect(env.ANTHROPIC_API_KEY).toStartWith("xum-proxy-"); + expect(env.ANTHROPIC_AUTH_TOKEN).toBe(env.ANTHROPIC_API_KEY); + + // AI SDK style (base + /messages) and official SDK style (base + /v1/messages). + for (const path of ["/messages", "/v1/messages"]) { + const res = await fetch(`${base}${path}`, { + method: "POST", + headers: { "x-api-key": env.ANTHROPIC_API_KEY, "anthropic-version": "2023-06-01" }, + body: JSON.stringify({ model: "claude-sonnet-5-5", stream: true }), + }); + expect(res.status).toBe(200); + expect(await res.text()).toBe(anthropicSse); + } + + expect(upstream.seen.map((r) => r.url)).toEqual(["/v1/messages", "/v1/messages"]); + for (const request of upstream.seen) { + expect(request.headers["x-api-key"]).toBe("real-anthropic-key"); + expect(request.headers["anthropic-version"]).toBe("2023-06-01"); + expect(JSON.stringify(request.headers)).not.toContain("xum-proxy-"); + } + expect(recorded).toEqual([ + { + workspaceId: "ws-1", + modelString: "anthropic:claude-sonnet-5-5", + usage: { inputTokens: 11, cachedInputTokens: 0, outputTokens: 4 }, + }, + { + workspaceId: "ws-1", + modelString: "anthropic:claude-sonnet-5-5", + usage: { inputTokens: 11, cachedInputTokens: 0, outputTokens: 4 }, + }, + ]); + expect(liveDeltas).toEqual(["ws-1", "ws-1"]); + }); + + test("a Bearer key works for OpenAI, and streamed chat requests ask for usage", async () => { + const env = await proxy.envFor("ws-2", "local", []); + const res = await fetch(`${env.OPENAI_BASE_URL}/chat/completions`, { + method: "POST", + headers: { + authorization: `Bearer ${env.OPENAI_API_KEY}`, + "content-type": "application/json", + }, + body: JSON.stringify({ model: "gpt-6.1", stream: true, stream_options: { foo: 1 } }), + }); + expect(res.status).toBe(200); + await res.text(); + const [request] = upstream.seen; + expect(request.headers.authorization).toBe("Bearer real-openai-key"); + // providers.jsonc headers go upstream, but never replace the auth header. + expect(request.headers["x-gateway-team"]).toBe("qa"); + expect(JSON.parse(request.body)).toEqual({ + model: "gpt-6.1", + stream: true, + stream_options: { foo: 1, include_usage: true }, + }); + expect(recorded.map((r) => r.modelString)).toEqual(["openai:gpt-6.1"]); + }); + + test("an explicit include_usage choice is kept", async () => { + const env = await proxy.envFor("ws-2b", "local", []); + const body = JSON.stringify({ stream: true, stream_options: { include_usage: false } }); + const res = await fetch(`${env.OPENAI_BASE_URL}/chat/completions`, { + method: "POST", + headers: { authorization: `Bearer ${env.OPENAI_API_KEY}` }, + body, + }); + await res.text(); + expect(upstream.seen[0].body).toBe(body); + }); + + test("an origin-only OpenAI base URL gets /v1, like chat requests", async () => { + configs.openai = { ...configs.openai, baseUrl: upstream.baseUrl.replace(/\/v1$/, "") }; + const env = await proxy.envFor("ws-origin", "local", []); + const res = await fetch(`${env.OPENAI_BASE_URL}/chat/completions`, { + method: "POST", + headers: { authorization: `Bearer ${env.OPENAI_API_KEY}` }, + body: "{}", + }); + expect(res.status).toBe(200); + expect(upstream.seen.map((r) => r.url)).toEqual(["/v1/chat/completions"]); + }); + + test("a base URL query stays a query, merged with the request's", async () => { + configs.anthropic = { ...configs.anthropic, baseUrl: `${upstream.baseUrl}?token=abc` }; + const env = await proxy.envFor("ws-query", "local", []); + await fetch(`${env.ANTHROPIC_BASE_URL}/v1/messages?beta=true`, { + method: "POST", + headers: { "x-api-key": env.ANTHROPIC_API_KEY }, + body: "{}", + }); + expect(upstream.seen.map((r) => r.url)).toEqual(["/v1/messages?token=abc&beta=true"]); + }); + + test("a configured OpenAI-Project header goes upstream, a command's own does not", async () => { + configs.openai = { ...configs.openai, headers: { "OpenAI-Project": "proj_config" } }; + const env = await proxy.envFor("ws-project", "local", []); + await fetch(`${env.OPENAI_BASE_URL}/chat/completions`, { + method: "POST", + headers: { authorization: `Bearer ${env.OPENAI_API_KEY}`, "openai-project": "proj_child" }, + body: "{}", + }); + expect(upstream.seen[0].headers["openai-project"]).toBe("proj_config"); + }); + + test("refusals never reach the provider", async () => { + const env = await proxy.envFor("ws-3", "local", []); + const post = (url: string, key: string) => + fetch(url, { method: "POST", headers: { "x-api-key": key }, body: "{}" }); + + const unknownKey = await post(`${env.ANTHROPIC_BASE_URL}/v1/messages`, "xum-proxy-nope"); + expect(unknownKey.status).toBe(401); + expect(((await unknownKey.json()) as { error: { message: string } }).error.message).toContain( + "unknown key" + ); + + const notAllowed = await post(`${env.ANTHROPIC_BASE_URL}/v1/files`, env.ANTHROPIC_API_KEY); + expect(notAllowed.status).toBe(404); + + removed.add("ws-3"); + const revoked = await post(`${env.ANTHROPIC_BASE_URL}/v1/messages`, env.ANTHROPIC_API_KEY); + expect(revoked.status).toBe(401); + + expect(upstream.seen).toEqual([]); + expect(recorded).toEqual([]); + }); + + test("a provider redirect is refused, not followed", async () => { + const env = await proxy.envFor("ws-4", "local", []); + const res = await fetch(`${env.OPENAI_BASE_URL}/responses`, { + method: "POST", + headers: { authorization: `Bearer ${env.OPENAI_API_KEY}` }, + body: "{}", + }); + expect(res.status).toBe(502); + expect(upstream.seen.map((r) => r.url)).toEqual(["/v1/responses"]); + }); + + test("no live delta when the ledger write did not happen", async () => { + const env = await proxy.envFor("removed-ws", "local", []); + const res = await fetch(`${env.ANTHROPIC_BASE_URL}/v1/messages`, { + method: "POST", + headers: { "x-api-key": env.ANTHROPIC_API_KEY }, + body: "{}", + }); + await res.text(); + expect(recorded).toHaveLength(1); + expect(liveDeltas).toEqual([]); + }); + + test("turning the switch off refuses keys that commands already hold", async () => { + const env = await proxy.envFor("ws-off", "local", []); + enabled = false; + const res = await fetch(`${env.ANTHROPIC_BASE_URL}/v1/messages`, { + method: "POST", + headers: { "x-api-key": env.ANTHROPIC_API_KEY }, + body: "{}", + }); + expect(res.status).toBe(503); + expect(upstream.seen).toEqual([]); + }); + + test("revoking project trust refuses keys that commands already hold", async () => { + const env = await proxy.envFor("ws-trust", "local", []); + untrusted.add("ws-trust"); + const res = await fetch(`${env.ANTHROPIC_BASE_URL}/v1/messages`, { + method: "POST", + headers: { "x-api-key": env.ANTHROPIC_API_KEY }, + body: "{}", + }); + expect(res.status).toBe(403); + expect(upstream.seen).toEqual([]); + expect(recorded).toEqual([]); + }); + + test("envFor gives a pair only when it is safe and useful", async () => { + // A project secret that names any provider var turns off that provider's whole pair. + const withSecret = await proxy.envFor("ws-5", "local", ["ANTHROPIC_BASE_URL"]); + expect(Object.keys(withSecret).sort()).toEqual(["OPENAI_API_KEY", "OPENAI_BASE_URL"]); + // So does an OpenAI organization or project secret: the proxy would replace the header + // that the SDK builds from it with the Xum account's. + for (const name of ["OPENAI_ORG_ID", "OPENAI_PROJECT_ID"]) { + expect(Object.keys(await proxy.envFor("ws-5", "local", [name])).sort()).toEqual([ + "ANTHROPIC_API_KEY", + "ANTHROPIC_AUTH_TOKEN", + "ANTHROPIC_BASE_URL", + ]); + } + + // Runtimes that cannot reach the backend's loopback get nothing. + for (const runtime of ["ssh", "docker", "devcontainer"] as const) { + expect(await proxy.envFor("ws-5", runtime, [])).toEqual({}); + } + + // A disabled provider has nothing to pay with. + configs.openai = { ...configs.openai, enabled: false }; + expect(Object.keys(await proxy.envFor("ws-5", "local", []))).not.toContain("OPENAI_API_KEY"); + + // The Settings switch turns the whole feature off. + enabled = false; + expect(await proxy.envFor("ws-5", "local", [])).toEqual({}); + }); +}); diff --git a/src/node/services/bashAiProxy/bashAiProxyService.ts b/src/node/services/bashAiProxy/bashAiProxyService.ts new file mode 100644 index 00000000000..5963091aa34 --- /dev/null +++ b/src/node/services/bashAiProxy/bashAiProxyService.ts @@ -0,0 +1,582 @@ +/** + * Bash AI proxy: a loopback HTTP proxy for Anthropic and OpenAI calls made by processes that the + * bash tool starts (test harnesses, `make bug-bash`, scripts that call the SDKs). + * + * Why: those calls bypass the chat stream, so their spend never reached the workspace's Costs + * tab or Analytics. Xum now gives each bash command a base URL that points here plus a key that + * names its workspace. The proxy swaps that key for the provider key in the Xum settings, + * forwards the request, streams the answer back, and records the usage in that workspace. + * + * Contract: + * - Only the Local and Worktree runtimes get the env pair: other runtimes cannot reach the + * backend's 127.0.0.1, and Xum adds no tunnel. + * - A proxy key authorizes only these provider endpoints for one workspace. It is not a Xum API + * token. Keys and the port live in memory: after a restart old processes get ECONNREFUSED. + * A key stops working when its workspace is removed. + * - The upstream host comes from the Xum provider config only, never from the request, and the + * proxy never follows redirects, so the real key cannot reach another host. + * - A refused request (unknown key, path not allowed, no Xum key) fails with an error. The proxy + * never falls back to a direct call. + */ +import assert from "node:assert/strict"; +import { randomBytes } from "node:crypto"; +import * as http from "node:http"; +import type { AddressInfo } from "node:net"; + +import { EnvHttpProxyAgent, type Dispatcher } from "undici"; + +import { isProviderDisabledInConfig } from "@/common/utils/providers/isProviderDisabled"; +import { + normalizeAnthropicBaseURL, + normalizeOpenAICompatibleBaseURL, +} from "@/common/utils/providers/baseUrl"; +import type { RuntimeMode } from "@/common/types/runtime"; +import type { ChatUsageDisplay } from "@/common/utils/tokens/usageAggregator"; +import type { AiSdkUsageLike } from "@/common/utils/tokens/usageHelpers"; +import { log } from "@/node/services/log"; +import { + resolveProviderCredentials, + type ProviderConfigRaw, +} from "@/node/utils/providerRequirements"; + +import { UsageTap, type BashAiProxyProvider } from "./usageExtract"; + +/** Prefix of every proxy key, so a leaked value is easy to recognize. */ +export const BASH_AI_PROXY_KEY_PREFIX = "xum-proxy-"; + +/** Analytics source: rows land as `tool_name = headless:bash_proxy`. */ +export const BASH_AI_PROXY_ANALYTICS_SOURCE = "bash_proxy"; + +const MAX_REQUEST_BODY_BYTES = 100 * 1024 * 1024; + +interface ProxyRoute { + provider: BashAiProxyProvider; + /** Path prefix on the proxy listener. */ + prefix: string; + defaultUpstream: string; + /** Env names a project secret can set; any one of them turns the pair off for this provider. */ + envNames: readonly string[]; + /** Paths that bill tokens: a 2xx answer without usage is logged as uncounted. */ + billable: readonly string[]; + allowed: ReadonlyArray<{ method: string; path: string | RegExp }>; + env: (origin: string, key: string) => Record; +} + +const ROUTES: readonly ProxyRoute[] = [ + { + provider: "anthropic", + prefix: "/anthropic", + defaultUpstream: "https://api.anthropic.com/v1", + envNames: ["ANTHROPIC_API_KEY", "ANTHROPIC_AUTH_TOKEN", "ANTHROPIC_BASE_URL"], + billable: ["/v1/messages"], + allowed: [ + { method: "POST", path: "/v1/messages" }, + { method: "POST", path: "/v1/messages/count_tokens" }, + { method: "GET", path: /^\/v1\/models(\/[^/]+)?$/ }, + ], + // No /v1: the official SDK and Claude Code append /v1/..., the AI SDK appends /messages. + // handle() adds a missing /v1, so both conventions reach the same endpoint. + env: (origin, key) => ({ + ANTHROPIC_BASE_URL: `${origin}/anthropic`, + ANTHROPIC_API_KEY: key, + ANTHROPIC_AUTH_TOKEN: key, + }), + }, + { + provider: "openai", + prefix: "/openai", + defaultUpstream: "https://api.openai.com/v1", + // The OpenAI SDKs also read OPENAI_ORG_ID and OPENAI_PROJECT_ID into headers that the proxy + // replaces with the Xum account's, so a secret that sets one keeps the direct call. + envNames: [ + "OPENAI_API_KEY", + "OPENAI_BASE_URL", + "OPENAI_API_BASE", + "OPENAI_ORG_ID", + "OPENAI_PROJECT_ID", + ], + billable: ["/v1/responses", "/v1/chat/completions"], + allowed: [ + { method: "POST", path: "/v1/responses" }, + { method: "POST", path: "/v1/chat/completions" }, + { method: "GET", path: /^\/v1\/models(\/[^/]+)?$/ }, + ], + env: (origin, key) => ({ OPENAI_BASE_URL: `${origin}/openai/v1`, OPENAI_API_KEY: key }), + }, +]; + +// Account headers: a command must not pick the account it bills, but providers.jsonc `headers` +// may set them, as chat requests send them. +const ACCOUNT_HEADERS = ["cookie", "openai-organization", "openai-project"]; + +// Request headers that never go upstream: hop-by-hop headers, the proxy key, and headers that +// the proxy sets itself from the Xum config. +const DROPPED_REQUEST_HEADERS = new Set([ + "host", + "connection", + "keep-alive", + "proxy-authorization", + "proxy-connection", + "te", + "trailer", + "transfer-encoding", + "upgrade", + // Clients such as curl can send `Expect: 100-continue` for large bodies. Node already + // answered it, and undici's fetch rejects the header. + "expect", + "content-length", + "accept-encoding", + "authorization", + "x-api-key", + ...ACCOUNT_HEADERS, +]); + +// fetch() decodes compressed bodies, so length and encoding headers no longer describe the bytes. +const DROPPED_RESPONSE_HEADERS = new Set([ + "connection", + "keep-alive", + "transfer-encoding", + "content-length", + "content-encoding", +]); + +export interface RecordedUsage { + model: string; + usage: ChatUsageDisplay; +} + +export interface BashAiProxyServiceOptions { + /** The Settings switch (config bashAiProxyEnabled; absent = off). */ + isEnabled: () => boolean; + /** A key verifies only while its workspace exists, so removal revokes it. */ + workspaceExists: (workspaceId: string) => boolean; + /** + * Whether the workspace may run repo code with provider credentials (shared-execution trust, + * project automation allowed). Checked per request, so revoking trust also stops commands + * that already hold a key, like turning the switch off. + */ + isWorkspaceTrusted: (workspaceId: string) => Promise; + /** Raw providers.jsonc entry for one provider. */ + loadProviderConfig: (provider: BashAiProxyProvider) => ProviderConfigRaw; + /** Writes priced usage to the workspace ledger and analytics sidecar; undefined = not written. */ + recordUsage: ( + workspaceId: string, + modelString: string, + usage: AiSdkUsageLike, + providerMetadata: Record | undefined + ) => Promise; + /** Live update for the Costs tab after a successful write. */ + onUsageRecorded: (workspaceId: string, recorded: RecordedUsage) => void; +} + +type RequestInitWithDispatcher = RequestInit & { dispatcher?: Dispatcher }; + +// Same outbound behavior as chat requests (providerModelFactory): honor HTTP(S)_PROXY and never +// time out a long stream. One process-wide agent, like the chat path's. +const outboundDispatcher = new EnvHttpProxyAgent({ bodyTimeout: 0, headersTimeout: 0 }); + +class ProxyRefusal extends Error { + constructor( + readonly status: number, + message: string + ) { + super(message); + } +} + +export class BashAiProxyService { + private readonly keys = new Map(); // proxy key -> workspaceId + private readonly keyByWorkspace = new Map(); + private server: http.Server | undefined; + private startPromise: Promise | undefined; + private stopped = false; + + constructor(private readonly options: BashAiProxyServiceOptions) {} + + /** + * Env vars for one bash command in this workspace: {} when the switch is off, the runtime + * cannot reach loopback, the listener failed, or Xum has no key for a provider. A provider + * whose env names appear in `secretKeys` is skipped as a whole, so a command never gets a + * proxy URL paired with a real key (project secrets override xumEnv in the bash tool). + */ + async envFor( + workspaceId: string, + runtime: RuntimeMode, + secretKeys: readonly string[] + ): Promise> { + assert(workspaceId.length > 0, "envFor requires a workspaceId"); + if (!this.options.isEnabled()) return {}; + const port = await this.ensureStarted(); + if (port === undefined) return {}; + const origin = this.originFor(runtime, port); + if (origin === undefined) return {}; + const env: Record = {}; + for (const route of ROUTES) { + if (route.envNames.some((name) => secretKeys.includes(name))) continue; + if (this.upstreamFor(route) === undefined) continue; // nothing to pay with + Object.assign(env, route.env(origin, this.keyFor(workspaceId))); + } + return env; + } + + async stop(): Promise { + this.stopped = true; + const server = this.server; + this.server = undefined; + this.keys.clear(); + this.keyByWorkspace.clear(); + if (server) { + server.closeAllConnections(); + await new Promise((resolve) => server.close(() => resolve())); + } + } + + /** Where commands of this workspace reach the proxy, or undefined when they cannot. */ + private originFor(runtime: RuntimeMode, port: number): string | undefined { + // Other runtimes cannot reach the backend's 127.0.0.1. + return runtime === "local" || runtime === "worktree" ? `http://127.0.0.1:${port}` : undefined; + } + + private keyFor(workspaceId: string): string { + const existing = this.keyByWorkspace.get(workspaceId); + if (existing !== undefined) return existing; + const key = `${BASH_AI_PROXY_KEY_PREFIX}${randomBytes(32).toString("hex")}`; + this.keys.set(key, workspaceId); + this.keyByWorkspace.set(workspaceId, key); + return key; + } + + /** The workspace a request's key names, if the key is known and the workspace still exists. */ + private workspaceForKey(key: string | undefined): string | undefined { + const workspaceId = key === undefined ? undefined : this.keys.get(key); + return workspaceId !== undefined && this.options.workspaceExists(workspaceId) + ? workspaceId + : undefined; + } + + /** + * Starts the listener on the first envFor() call (the first agent turn in a Local or Worktree + * workspace), so a backend without such turns binds no port. + * A failed start logs and returns undefined (bash then runs without the pair); the next call + * tries again. + */ + private ensureStarted(): Promise { + if (this.stopped) return Promise.resolve(undefined); + this.startPromise ??= this.listen().catch((error: unknown) => { + log.warn("[bash-ai-proxy] listener failed to start; bash AI calls stay uncounted", { + error: error instanceof Error ? error.message : String(error), + }); + this.startPromise = undefined; + return undefined; + }); + return this.startPromise; + } + + private async listen(): Promise { + const server = http.createServer((req, res) => { + this.handle(req, res).catch((error: unknown) => { + // A client that hangs up aborts the upstream fetch: that is not a proxy failure. + (res.destroyed ? log.debug : log.warn)("[bash-ai-proxy] request failed", { + error: error instanceof Error ? error.message : String(error), + }); + if (!res.headersSent) sendError(res, 502, "Xum bash AI proxy: upstream request failed."); + else res.destroy(); + }); + }); + await listenOn(server, 0); + if (this.stopped) { + server.close(); + throw new Error("stopped while starting"); + } + this.server = server; + const port = (server.address() as AddressInfo).port; + assert(port > 0, "bash AI proxy listener must have a port"); + log.info("[bash-ai-proxy] listening", { port }); + return port; + } + + /** Upstream base URL (with /v1) and key from the Xum provider config, or undefined. */ + private upstreamFor( + route: ProxyRoute + ): + | { baseUrl: string; apiKey: string; organization?: string; headers: Record } + | undefined { + const config = this.options.loadProviderConfig(route.provider); + if (isProviderDisabledInConfig(config)) return undefined; + const creds = resolveProviderCredentials(route.provider, config); + if (!creds.isConfigured || !creds.apiKey) return undefined; + const configured = creds.baseUrl?.trim(); + const baseUrl = + route.provider === "anthropic" + ? normalizeAnthropicBaseURL(configured ?? route.defaultUpstream) + : normalizeOpenAICompatibleBaseURL(configured ?? route.defaultUpstream); + return { + baseUrl, + apiKey: creds.apiKey, + ...(creds.organization ? { organization: creds.organization } : {}), + // Custom headers from providers.jsonc (gateways can need them), as chat requests send them. + headers: readStringRecord((config as { headers?: unknown }).headers), + }; + } + + private async handle(req: http.IncomingMessage, res: http.ServerResponse): Promise { + try { + await this.forward(req, res); + } catch (error) { + if (!(error instanceof ProxyRefusal)) throw error; + // Drain the unread body so the client sees the answer instead of a reset socket. + req.resume(); + sendError(res, error.status, error.message); + } + } + + private async forward(req: http.IncomingMessage, res: http.ServerResponse): Promise { + const url = new URL(req.url ?? "/", "http://127.0.0.1"); + // Turning the switch off cuts access at once, for processes that already hold a key too. + if (!this.options.isEnabled()) { + throw new ProxyRefusal(503, "Xum bash AI proxy is turned off in Settings → Providers."); + } + const route = ROUTES.find( + (r) => url.pathname === r.prefix || url.pathname.startsWith(`${r.prefix}/`) + ); + if (!route) throw new ProxyRefusal(404, "Xum bash AI proxy: unknown route."); + + const workspaceId = this.workspaceForKey(readProxyKey(req.headers)); + if (workspaceId === undefined) { + throw new ProxyRefusal( + 401, + "Xum bash AI proxy: unknown key. The workspace was removed, or the variable holds another key (do not put a real provider key in it)." + ); + } + if (!(await this.options.isWorkspaceTrusted(workspaceId))) { + throw new ProxyRefusal( + 403, + "Xum bash AI proxy: the workspace's project is not trusted, so its commands cannot call providers through Xum." + ); + } + + let path = url.pathname.slice(route.prefix.length) || "/"; + if (path !== "/v1" && !path.startsWith("/v1/")) path = `/v1${path}`; + const method = (req.method ?? "GET").toUpperCase(); + const allowed = route.allowed.some( + (a) => + a.method === method && (typeof a.path === "string" ? a.path === path : a.path.test(path)) + ); + if (!allowed) { + throw new ProxyRefusal(404, `Xum bash AI proxy: ${method} ${path} is not allowed.`); + } + + const upstream = this.upstreamFor(route); + if (!upstream) { + throw new ProxyRefusal( + 503, + `Xum bash AI proxy: Xum has no ${route.provider} API key. Add one in Settings → Providers.` + ); + } + + let body = method === "GET" ? undefined : await readBody(req); + if (body && route.provider === "openai" && path === "/v1/chat/completions") { + body = withStreamUsage(body); + } + + const headers = new Headers(); + for (const [name, value] of Object.entries(req.headers)) { + if (value === undefined || DROPPED_REQUEST_HEADERS.has(name)) continue; + headers.set(name, Array.isArray(value) ? value.join(", ") : value); + } + if (route.provider === "anthropic") { + headers.set("x-api-key", upstream.apiKey); + } else { + headers.set("authorization", `Bearer ${upstream.apiKey}`); + if (upstream.organization) headers.set("openai-organization", upstream.organization); + } + for (const [name, value] of Object.entries(upstream.headers)) headers.set(name, value); + headers.set("accept-encoding", "identity"); + + // A client that hangs up stops the upstream request too. + const abort = new AbortController(); + res.once("close", () => abort.abort()); + + const init: RequestInitWithDispatcher = { + method, + headers, + body: body ? new Uint8Array(body) : undefined, + redirect: "manual", + signal: abort.signal, + dispatcher: outboundDispatcher, + }; + const response = await fetch(upstreamUrl(upstream.baseUrl, path, url.searchParams), init); + if (response.status >= 300 && response.status < 400) { + await response.body?.cancel(); + throw new ProxyRefusal( + 502, + "Xum bash AI proxy: the provider sent a redirect. Xum does not follow it." + ); + } + + const responseHeaders: Record = {}; + response.headers.forEach((value, name) => { + if (!DROPPED_RESPONSE_HEADERS.has(name)) responseHeaders[name] = value; + }); + res.writeHead(response.status, responseHeaders); + + const billable = response.ok && route.billable.includes(path); + const tap = billable + ? new UsageTap(route.provider, response.headers.get("content-type") ?? undefined) + : undefined; + try { + if (response.body) { + const reader = response.body.getReader(); + for (;;) { + const { done, value } = await reader.read(); + if (done) break; + tap?.push(value); + if (!res.write(value)) await waitForDrain(res); + } + } + res.end(); + } finally { + // Record what the response carried even if the client hung up mid-stream: the provider + // bills the tokens it already produced. + if (tap) await this.record(workspaceId, route.provider, path, tap); + } + } + + private async record( + workspaceId: string, + provider: BashAiProxyProvider, + path: string, + tap: UsageTap + ): Promise { + const extracted = tap.finish(); + if (!extracted) { + log.warn("[bash-ai-proxy] response had no usage; not counted", { workspaceId, path }); + return; + } + const recorded = await this.options.recordUsage( + workspaceId, + `${provider}:${extracted.model ?? "unknown"}`, + extracted.usage, + extracted.providerMetadata + ); + if (recorded) this.options.onUsageRecorded(workspaceId, recorded); + } +} + +function listenOn(server: http.Server, port: number): Promise { + return new Promise((resolve, reject) => { + const onError = (error: Error) => { + server.off("listening", onListening); + reject(error); + }; + const onListening = () => { + server.off("error", onError); + resolve(); + }; + server.once("error", onError); + server.once("listening", onListening); + server.listen(port, "127.0.0.1"); + }); +} + +/** + * The upstream URL for `path` (starts with /v1, which the base URL already ends with). The + * endpoint goes into the base URL's path and the queries merge: a base URL can carry its own + * query (a gateway token), and string concatenation would put the endpoint inside it. + */ +function upstreamUrl(baseUrl: string, path: string, search: URLSearchParams): URL { + assert(path.startsWith("/v1"), "proxy paths are normalized to /v1"); + const target = new URL(baseUrl); + target.pathname = `${target.pathname.replace(/\/+$/, "")}${path.slice(3)}`; + for (const [name, value] of search) target.searchParams.append(name, value); + target.hash = ""; + return target; +} + +function readStringRecord(value: unknown): Record { + if (typeof value !== "object" || value === null || Array.isArray(value)) return {}; + const out: Record = {}; + for (const [name, v] of Object.entries(value)) { + // Never let config headers replace the auth header that the proxy sets. + const lower = name.toLowerCase(); + const dropped = DROPPED_REQUEST_HEADERS.has(lower) && !ACCOUNT_HEADERS.includes(lower); + if (typeof v === "string" && !dropped) out[name] = v; + } + return out; +} + +function readProxyKey(headers: http.IncomingHttpHeaders): string | undefined { + const apiKey = headers["x-api-key"]; + if (typeof apiKey === "string" && apiKey.startsWith(BASH_AI_PROXY_KEY_PREFIX)) return apiKey; + const auth = headers.authorization; + if (typeof auth === "string") { + const match = /^Bearer\s+(\S+)$/i.exec(auth); + if (match?.[1]?.startsWith(BASH_AI_PROXY_KEY_PREFIX)) return match[1]; + } + return undefined; +} + +async function readBody(req: http.IncomingMessage): Promise { + const chunks: Buffer[] = []; + let size = 0; + for await (const chunk of req as AsyncIterable) { + size += chunk.byteLength; + if (size > MAX_REQUEST_BODY_BYTES) { + throw new ProxyRefusal(413, "Xum bash AI proxy: request body is too large."); + } + chunks.push(chunk); + } + return Buffer.concat(chunks); +} + +/** + * Streamed Chat Completions carry usage only with stream_options.include_usage. Xum sets it only + * when the client left it unset: the usage chunk has `choices: []`, and a client that turned it + * off on purpose may not handle that chunk. Such a stream stays uncounted (logged). + */ +function withStreamUsage(body: Buffer): Buffer { + let parsed: unknown; + try { + parsed = JSON.parse(body.toString("utf8")); + } catch { + return body; + } + if ( + typeof parsed !== "object" || + parsed === null || + (parsed as { stream?: unknown }).stream !== true + ) { + return body; + } + const request = parsed as { stream_options?: Record }; + if (request.stream_options?.include_usage !== undefined) return body; + request.stream_options = { ...(request.stream_options ?? {}), include_usage: true }; + return Buffer.from(JSON.stringify(request)); +} + +function waitForDrain(res: http.ServerResponse): Promise { + return new Promise((resolve) => { + const done = () => { + res.off("drain", done); + res.off("close", done); + resolve(); + }; + res.once("drain", done); + res.once("close", done); + }); +} + +function sendError(res: http.ServerResponse, status: number, message: string): void { + if (res.headersSent) { + res.destroy(); + return; + } + // Anthropic's error envelope; OpenAI SDKs also read `error.message` from it. + res.writeHead(status, { "content-type": "application/json" }); + res.end( + JSON.stringify({ + type: "error", + error: { type: status === 401 ? "authentication_error" : "xum_proxy_error", message }, + }) + ); +} diff --git a/src/node/services/bashAiProxy/usageExtract.test.ts b/src/node/services/bashAiProxy/usageExtract.test.ts new file mode 100644 index 00000000000..1e67f98c7c0 --- /dev/null +++ b/src/node/services/bashAiProxy/usageExtract.test.ts @@ -0,0 +1,160 @@ +import { describe, expect, test } from "bun:test"; + +import { createDisplayUsage } from "@/common/utils/tokens/displayUsage"; + +import { UsageTap, extractUsageFromJson, type BashAiProxyProvider } from "./usageExtract"; + +function sse(events: unknown[]): string { + return events.map((e) => `event: x\ndata: ${JSON.stringify(e)}\n\n`).join(""); +} + +/** Feeds `text` in awkward slices so lines and UTF-8 sequences split across chunks. */ +function tapAll(provider: BashAiProxyProvider, contentType: string, text: string) { + const tap = new UsageTap(provider, contentType); + const bytes = new TextEncoder().encode(text); + for (let i = 0; i < bytes.length; i += 7) tap.push(bytes.subarray(i, i + 7)); + return tap.finish(); +} + +// The invariant that matters: after pricing, every token bucket matches the vendor's own count +// once (no cache token is counted as plain input, no reasoning token as plain output). +function priced(provider: BashAiProxyProvider, text: string, contentType: string) { + const extracted = tapAll(provider, contentType, text); + expect(extracted).toBeDefined(); + const display = createDisplayUsage( + extracted!.usage, + `${provider}:${extracted!.model ?? "unknown"}`, + extracted!.providerMetadata + )!; + return { + model: extracted!.model, + input: display.input.tokens, + cached: display.cached.tokens, + cacheCreate: display.cacheCreate.tokens, + output: display.output.tokens, + reasoning: display.reasoning.tokens, + }; +} + +describe("UsageTap", () => { + test("Anthropic SSE combines message_start with the cumulative message_delta", () => { + const text = sse([ + { + type: "message_start", + message: { + model: "claude-sonnet-5-5", + usage: { + input_tokens: 100, + cache_read_input_tokens: 2000, + cache_creation_input_tokens: 300, + output_tokens: 1, + }, + }, + }, + { type: "content_block_delta", delta: { text: "héllo ✓" } }, + { type: "message_delta", usage: { output_tokens: 42 } }, + { type: "message_stop" }, + ]); + expect(priced("anthropic", text, "text/event-stream; charset=utf-8")).toEqual({ + model: "claude-sonnet-5-5", + input: 100, + cached: 2000, + cacheCreate: 300, + output: 42, + reasoning: 0, + }); + }); + + test("OpenAI Responses SSE reads response.completed", () => { + const text = sse([ + { type: "response.created", response: { model: "gpt-6.1", usage: null } }, + { + type: "response.completed", + response: { + model: "gpt-6.1", + usage: { + input_tokens: 1000, + input_tokens_details: { cached_tokens: 600 }, + output_tokens: 250, + output_tokens_details: { reasoning_tokens: 200 }, + }, + }, + }, + ]); + expect(priced("openai", text, "text/event-stream")).toEqual({ + model: "gpt-6.1", + input: 400, + cached: 600, + cacheCreate: 0, + output: 50, + reasoning: 200, + }); + }); + + test("OpenAI Chat Completions SSE reads the final usage chunk", () => { + const text = + sse([ + { model: "gpt-6.1-mini", choices: [{ delta: { content: "hi" } }], usage: null }, + { + model: "gpt-6.1-mini", + choices: [], + usage: { prompt_tokens: 30, completion_tokens: 5, prompt_tokens_details: {} }, + }, + ]) + "data: [DONE]\n\n"; + expect(priced("openai", text, "text/event-stream")).toEqual({ + model: "gpt-6.1-mini", + input: 30, + cached: 0, + cacheCreate: 0, + output: 5, + reasoning: 0, + }); + }); + + test("a JSON Anthropic body is priced like the SSE form", () => { + const body = JSON.stringify({ + type: "message", + model: "claude-opus-5-5", + usage: { input_tokens: 10, cache_read_input_tokens: 5, output_tokens: 7 }, + }); + expect(priced("anthropic", body, "application/json")).toEqual({ + model: "claude-opus-5-5", + input: 10, + cached: 5, + cacheCreate: 0, + output: 7, + reasoning: 0, + }); + }); + + test("bodies without usage report nothing", () => { + expect(tapAll("anthropic", "text/html", "bad gateway")).toBeUndefined(); + expect(tapAll("openai", "application/json", JSON.stringify({ data: [] }))).toBeUndefined(); + expect(tapAll("openai", "text/event-stream", "data: [DONE]\n\n")).toBeUndefined(); + }); + + test("a stream cut before message_delta still reports the input it saw", () => { + const text = sse([ + { + type: "message_start", + message: { model: "claude-sonnet-5-5", usage: { input_tokens: 9, output_tokens: 1 } }, + }, + ]).slice(0, -1); // no trailing newline: the last line arrives at finish() + expect(tapAll("anthropic", "text/event-stream", text)?.usage).toEqual({ + inputTokens: 9, + cachedInputTokens: 0, + outputTokens: 1, + }); + }); +}); + +describe("extractUsageFromJson", () => { + test("passes an OpenAI service tier to pricing metadata", () => { + const extracted = extractUsageFromJson("openai", { + model: "gpt-6.1", + service_tier: "flex", + usage: { input_tokens: 1, output_tokens: 1 }, + }); + expect(extracted?.providerMetadata).toEqual({ openai: { serviceTier: "flex" } }); + }); +}); diff --git a/src/node/services/bashAiProxy/usageExtract.ts b/src/node/services/bashAiProxy/usageExtract.ts new file mode 100644 index 00000000000..bf8f963ba1b --- /dev/null +++ b/src/node/services/bashAiProxy/usageExtract.ts @@ -0,0 +1,226 @@ +/** + * Reads token usage out of provider responses that the bash AI proxy forwards. + * + * The proxy sees raw wire responses (not AI SDK results), so this module maps each vendor's + * usage block to the flat AI SDK v6 shape that `createDisplayUsage` prices: `inputTokens` + * INCLUDES cache reads and cache writes, `outputTokens` INCLUDES reasoning. Cache writes travel + * in `providerMetadata.anthropic.cacheCreationInputTokens`, as for chat streams. + * + * Supported shapes (JSON bodies and SSE streams): + * - Anthropic Messages: `usage` on the message; SSE gives it in `message_start` and the + * cumulative totals in `message_delta`. + * - OpenAI Responses: `usage` on the response; SSE gives it in `response.completed` + * (also `response.incomplete` / `response.failed`). + * - OpenAI Chat Completions: `usage` on the body or on the last SSE chunk (the proxy forces + * `stream_options.include_usage` so streams carry it). + */ +import assert from "node:assert/strict"; + +import type { AiSdkUsageLike } from "@/common/utils/tokens/usageHelpers"; + +export type BashAiProxyProvider = "anthropic" | "openai"; + +export interface ExtractedUsage { + /** Model id as the provider reported it, without the provider prefix. */ + model: string | undefined; + usage: AiSdkUsageLike; + providerMetadata?: Record; +} + +/** A JSON body larger than this is forwarded but not parsed for usage (logged as uncounted). */ +const MAX_JSON_BODY_BYTES = 64 * 1024 * 1024; + +type JsonRecord = Record; + +function isRecord(value: unknown): value is JsonRecord { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function num(value: unknown): number | undefined { + return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : undefined; +} + +function str(value: unknown): string | undefined { + return typeof value === "string" && value.length > 0 ? value : undefined; +} + +function anthropicUsage(raw: JsonRecord, model: string | undefined): ExtractedUsage { + const input = num(raw.input_tokens) ?? 0; + const cacheRead = num(raw.cache_read_input_tokens) ?? 0; + const cacheWrite = num(raw.cache_creation_input_tokens) ?? 0; + return { + model, + // Anthropic's input_tokens excludes both cache buckets; the v6 shape includes them. + usage: { + inputTokens: input + cacheRead + cacheWrite, + cachedInputTokens: cacheRead, + outputTokens: num(raw.output_tokens) ?? 0, + }, + providerMetadata: { + // `usage` lets service-tier pricing read `speed`, as it does for chat streams. + anthropic: { + usage: raw, + ...(cacheWrite > 0 ? { cacheCreationInputTokens: cacheWrite } : {}), + }, + }, + }; +} + +function openAiUsage( + raw: JsonRecord, + model: string | undefined, + serviceTier: unknown +): ExtractedUsage { + const inputDetails = isRecord(raw.input_tokens_details) + ? raw.input_tokens_details + : isRecord(raw.prompt_tokens_details) + ? raw.prompt_tokens_details + : undefined; + const outputDetails = isRecord(raw.output_tokens_details) + ? raw.output_tokens_details + : isRecord(raw.completion_tokens_details) + ? raw.completion_tokens_details + : undefined; + const tier = str(serviceTier); + return { + model, + // OpenAI already reports input inclusive of cached tokens and output inclusive of reasoning. + usage: { + inputTokens: num(raw.input_tokens) ?? num(raw.prompt_tokens) ?? 0, + cachedInputTokens: num(inputDetails?.cached_tokens) ?? 0, + outputTokens: num(raw.output_tokens) ?? num(raw.completion_tokens) ?? 0, + reasoningTokens: num(outputDetails?.reasoning_tokens) ?? 0, + }, + ...(tier ? { providerMetadata: { openai: { serviceTier: tier } } } : {}), + }; +} + +/** Usage from one complete JSON response body, or undefined if it carries none. */ +export function extractUsageFromJson( + provider: BashAiProxyProvider, + body: unknown +): ExtractedUsage | undefined { + if (!isRecord(body) || !isRecord(body.usage)) return undefined; + const model = str(body.model); + return provider === "anthropic" + ? anthropicUsage(body.usage, model) + : openAiUsage(body.usage, model, body.service_tier); +} + +/** + * Observes a response body chunk by chunk while the proxy streams it to the client, then reports + * the usage it found. One instance per upstream response, so each response is recorded once. + */ +export class UsageTap { + private readonly decoder = new TextDecoder(); + private readonly isSse: boolean; + private sseBuffer = ""; + private readonly jsonChunks: Uint8Array[] = []; + private jsonBytes = 0; + private jsonOverflow = false; + private finished = false; + + // SSE state. Anthropic splits usage across message_start and message_delta. + private model: string | undefined; + private anthropicRaw: JsonRecord | undefined; + private openAiResult: ExtractedUsage | undefined; + + constructor( + private readonly provider: BashAiProxyProvider, + contentType: string | undefined + ) { + this.isSse = (contentType ?? "").toLowerCase().includes("text/event-stream"); + } + + push(chunk: Uint8Array): void { + assert(!this.finished, "UsageTap.push after finish"); + if (!this.isSse) { + this.jsonBytes += chunk.byteLength; + if (this.jsonBytes > MAX_JSON_BODY_BYTES) { + this.jsonOverflow = true; + this.jsonChunks.length = 0; + } else if (!this.jsonOverflow) { + this.jsonChunks.push(chunk); + } + return; + } + this.sseBuffer += this.decoder.decode(chunk, { stream: true }); + let newline = this.sseBuffer.indexOf("\n"); + while (newline !== -1) { + this.observeSseLine(this.sseBuffer.slice(0, newline)); + this.sseBuffer = this.sseBuffer.slice(newline + 1); + newline = this.sseBuffer.indexOf("\n"); + } + } + + /** The usage seen so far; a stream cut short still reports what it carried. */ + finish(): ExtractedUsage | undefined { + assert(!this.finished, "UsageTap.finish called twice"); + this.finished = true; + if (!this.isSse) { + if (this.jsonOverflow || this.jsonChunks.length === 0) return undefined; + try { + return extractUsageFromJson( + this.provider, + JSON.parse(Buffer.concat(this.jsonChunks).toString("utf8")) + ); + } catch { + return undefined; // Not JSON (an HTML error page, say): nothing to count. + } + } + this.sseBuffer += this.decoder.decode(); + if (this.sseBuffer.length > 0) this.observeSseLine(this.sseBuffer); + this.sseBuffer = ""; + if (this.provider === "anthropic") { + return this.anthropicRaw ? anthropicUsage(this.anthropicRaw, this.model) : undefined; + } + return this.openAiResult; + } + + private observeSseLine(rawLine: string): void { + const line = rawLine.endsWith("\r") ? rawLine.slice(0, -1) : rawLine; + if (!line.startsWith("data:")) return; + const data = line.slice(5).trim(); + if (data.length === 0 || data === "[DONE]") return; + let event: unknown; + try { + event = JSON.parse(data); + } catch { + return; + } + if (!isRecord(event)) return; + if (this.provider === "anthropic") { + this.observeAnthropicEvent(event); + } else { + this.observeOpenAiEvent(event); + } + } + + private observeAnthropicEvent(event: JsonRecord): void { + if (event.type === "message_start" && isRecord(event.message)) { + this.model = str(event.message.model) ?? this.model; + if (isRecord(event.message.usage)) this.anthropicRaw = { ...event.message.usage }; + return; + } + // message_delta usage is cumulative: each field present replaces the earlier value. + if (event.type === "message_delta" && isRecord(event.usage)) { + const merged: JsonRecord = { ...(this.anthropicRaw ?? {}) }; + for (const [key, value] of Object.entries(event.usage)) { + if (value !== null && value !== undefined) merged[key] = value; + } + this.anthropicRaw = merged; + } + } + + private observeOpenAiEvent(event: JsonRecord): void { + // Responses API: response.completed / .incomplete / .failed carry the final usage. + if (isRecord(event.response) && isRecord(event.response.usage)) { + this.openAiResult = extractUsageFromJson("openai", event.response); + return; + } + // Chat Completions: the final chunk has `usage` (null on the others). + if (isRecord(event.usage)) { + this.openAiResult = extractUsageFromJson("openai", event); + } + } +} diff --git a/src/node/services/di/layers/desktop.ts b/src/node/services/di/layers/desktop.ts index 6d2d8aed7d9..20ae0960109 100644 --- a/src/node/services/di/layers/desktop.ts +++ b/src/node/services/di/layers/desktop.ts @@ -31,6 +31,13 @@ import { setSshPromptService } from "@/node/runtime/sshConnectionPool"; import { createWorktreeArchiveHook } from "@/node/runtime/worktreeLifecycleHooks"; import { AgentPluginInstallService } from "@/node/services/agentPlugins/installService"; import { AgentStatusService } from "@/node/services/agentStatusService"; +import { + BASH_AI_PROXY_ANALYTICS_SOURCE, + BashAiProxyService, +} from "@/node/services/bashAiProxy/bashAiProxyService"; +import { isWorkspaceTrustedForSharedExecution } from "@/node/services/utils/workspaceTrust"; +import { projectAutomationDisabled } from "@/node/utils/projectAutomation"; +import type { ProviderConfigRaw } from "@/node/utils/providerRequirements"; import { AnalyticsService, type IngestWorkspaceMeta, @@ -62,6 +69,7 @@ import { Analytics, BackgroundProcessManagerTag, Backup, + BashAiProxy, BrowserBridgeServerTag, BrowserBridgeTokenManagerTag, BrowserControl, @@ -498,6 +506,7 @@ export const WorkersLive: Layer.Layer< | SessionUsage | Tokenizer | WindowTag + | ProvidersConfigStoreTag > = Layer.effectContext( Effect.gen(function* () { const config = yield* ConfigTag; @@ -576,7 +585,47 @@ export const WorkersLive: Layer.Layer< }, } ); + // Bash AI proxy: AI calls made by bash tool processes go through it so their spend lands in + // the workspace ledger (Costs tab) and the analytics sidecar, like status generation above. + const providersConfigStore = yield* ProvidersConfigStoreTag; + const bashAiProxy = new BashAiProxyService({ + workspaceExists: (workspaceId) => config.findWorkspace(workspaceId) !== null, + // The same rule as the bash tool's trust (TurnRequestBuilder sharedExecutionTrusted). + isWorkspaceTrusted: async (workspaceId) => { + const metadata = await config.getWorkspaceMetadataById(workspaceId); + return ( + metadata != null && + isWorkspaceTrustedForSharedExecution(metadata, config.loadConfigOrDefault().projects) && + !projectAutomationDisabled() + ); + }, + // Opt-in: with the vars set, agent CLIs such as `claude -p` bill the Xum API key instead + // of a subscription login, so Xum must not change that silently. + isEnabled: () => config.loadConfigOrDefault().bashAiProxyEnabled === true, + loadProviderConfig: (provider) => + (providersConfigStore.loadProvidersConfig()?.[provider] ?? {}) as ProviderConfigRaw, + recordUsage: async (workspaceId, modelString, usage, providerMetadata) => { + const recorded = await sessionUsageService.recordHeadlessUsage( + workspaceId, + modelString, + usage, + providerMetadata, + { analyticsSource: BASH_AI_PROXY_ANALYTICS_SOURCE } + ); + workspaceService.emit("analyticsIngest", { workspaceId }); + return recorded; + }, + onUsageRecorded: (workspaceId, recorded) => + workspaceService.emitChatEvent(workspaceId, { + type: "session-usage-delta", + workspaceId, + sourceWorkspaceId: workspaceId, + byModelDelta: { [recorded.model]: recorded.usage }, + timestamp: Date.now(), + }), + }); return Context.empty().pipe( + Context.add(BashAiProxy, bashAiProxy), Context.add(IdleCompaction, idleCompactionService), Context.add(Heartbeat, heartbeatService), Context.add(Timeline, timelineService), @@ -631,6 +680,8 @@ export const DesktopWiringLive: Layer.Layer< const experimentsService = yield* Experiments; turnRequestBuilderBindings.analyticsService = analyticsService; + const bashAiProxy = yield* BashAiProxy; + turnRequestBuilderBindings.bashAiProxy = bashAiProxy; projectService.setWorkspaceService(workspaceService); projectService.setWorkspaceMetadataRefresher(workspaceService); diff --git a/src/node/services/di/tags.ts b/src/node/services/di/tags.ts index 147ed05af57..f9b473166b7 100644 --- a/src/node/services/di/tags.ts +++ b/src/node/services/di/tags.ts @@ -21,6 +21,7 @@ import type { } from "@/node/config"; import type { AgentPluginInstallService } from "@/node/services/agentPlugins/installService"; import type { AgentStatusService } from "@/node/services/agentStatusService"; +import type { BashAiProxyService } from "@/node/services/bashAiProxy/bashAiProxyService"; import type { AIService } from "@/node/services/aiService"; import type { AutoModelRouter } from "@/node/services/autoModelRouter"; import type { AnalyticsService } from "@/node/services/analytics/analyticsService"; @@ -307,6 +308,9 @@ export class Refine extends Context.Service()("xum/Refine export class AgentStatus extends Context.Service()( "xum/AgentStatus" ) {} +export class BashAiProxy extends Context.Service()( + "xum/BashAiProxy" +) {} /** The process's config stores (`ConfigStores`), one tag per store. */ export type StoreTags = @@ -403,7 +407,7 @@ export type MiscDesktopTags = | WorkspaceLifecycleHooksTag | WorktreeArchiveSnapshot; export type OauthTags = McpOauth | MuxGatewayOauth | CodexOauth | CoderOauth | CopilotOauth; -export type WorkerTags = IdleCompaction | Heartbeat | Timeline | Refine | AgentStatus; +export type WorkerTags = IdleCompaction | Heartbeat | Timeline | Refine | AgentStatus | BashAiProxy; export type DesktopTags = | BrowserTags | DesktopBridgeTags diff --git a/src/node/services/serviceContainer.ts b/src/node/services/serviceContainer.ts index 005b202b0a9..52811b878aa 100644 --- a/src/node/services/serviceContainer.ts +++ b/src/node/services/serviceContainer.ts @@ -64,6 +64,7 @@ import type { AgentPluginInstallService } from "@/node/services/agentPlugins/ins import type { McpOauthService } from "@/node/services/mcpOauthService"; import type { HeartbeatService } from "@/node/services/heartbeatService"; import type { AgentStatusService } from "@/node/services/agentStatusService"; +import type { BashAiProxyService } from "@/node/services/bashAiProxy/bashAiProxyService"; import type { IdleCompactionService } from "@/node/services/idleCompactionService"; import type { IdleDispatcher } from "@/node/services/idleDispatcher"; import type { CoderService } from "@/node/services/coderService"; @@ -96,6 +97,7 @@ import { AgentBrowserSessionDiscovery, AgentPluginInstall, AgentStatus, + BashAiProxy, AI, Analytics, BackgroundProcessManagerTag, @@ -292,6 +294,7 @@ export class ServiceContainer { public readonly idleDispatcher: IdleDispatcher; public readonly heartbeatService: HeartbeatService; public readonly agentStatusService: AgentStatusService; + public readonly bashAiProxy: BashAiProxyService; // Shared between initializeCore() and runStartupHousekeeping() so the completion log still // reports every step and the total wall time from the start of core init. private startupStartedAt: number | undefined; @@ -406,6 +409,7 @@ export class ServiceContainer { this.idleDispatcher = get(IdleDispatcherTag); this.heartbeatService = get(Heartbeat); this.agentStatusService = get(AgentStatus); + this.bashAiProxy = get(BashAiProxy); assert( new Set(this.startupCoreSteps.map((step) => step.name)).size === this.startupCoreSteps.length, "startupCoreSteps names must be unique (they key stepDurationsMs)" @@ -807,6 +811,7 @@ export class ServiceContainer { await this.perfCaptures.dispose(); this.idleCompactionService.stop(); await this.browserBridgeServer.stop(); + await this.bashAiProxy.stop(); this.browserSessionStateHub.dispose(); this.browserBridgeTokenManager.dispose(); await this.timelineService.flush(); @@ -970,6 +975,7 @@ export class ServiceContainer { shutdownStep("perfFlightRecorder.stop", () => this.perfFlightRecorder.stop()); await shutdownStep("perfCaptures.dispose", () => this.perfCaptures.dispose()); await shutdownStep("browserBridgeServer.stop", () => this.browserBridgeServer.stop()); + await shutdownStep("bashAiProxy.stop", () => this.bashAiProxy.stop()); shutdownStep("browserSessionStateHub.dispose", () => this.browserSessionStateHub.dispose()); shutdownStep("browserBridgeTokenManager.dispose", () => this.browserBridgeTokenManager.dispose() diff --git a/src/node/services/turnRequestBuilder.ts b/src/node/services/turnRequestBuilder.ts index 7610e5a3942..b2461095ca0 100644 --- a/src/node/services/turnRequestBuilder.ts +++ b/src/node/services/turnRequestBuilder.ts @@ -97,6 +97,7 @@ import { extractChunkDeltaText } from "@/common/utils/ai/streamChunks"; import { createDisplayUsage } from "@/common/utils/tokens/displayUsage"; import { getTotalCost, sumUsageHistory } from "@/common/utils/tokens/usageAggregator"; import type { DesktopSessionManager } from "@/node/services/desktop/DesktopSessionManager"; +import type { BashAiProxyService } from "@/node/services/bashAiProxy/bashAiProxyService"; import type { ComputerUseService } from "@/node/services/computerUse/computerUseService"; import type { DevToolsService } from "@/node/services/devToolsService"; import type { ExperimentsService } from "@/node/services/experimentsService"; @@ -581,6 +582,8 @@ export interface TurnRequestBuilderBindings extends OauthServiceBindings { analyticsService?: { executeRawQuery(sql: string): Promise }; desktopSessionManager?: DesktopSessionManager; computerUseService?: ComputerUseService; + /** Routes AI calls from bash commands through Xum so their spend is accounted. */ + bashAiProxy?: Pick; } /** @@ -2174,6 +2177,32 @@ export class TurnRequestBuilder { costsUsd: sessionCostsUsd, scratchDir, }); + const projectSecretsRecord = await secretsToRecord(projectSecrets); + // AI calls from bash commands bypass the chat stream. Point their SDKs at the local proxy so + // the spend lands in this workspace's ledger (Costs tab and Analytics). envFor() skips any + // provider that project secrets configure: secrets override xumEnv in the bash tool, and a + // proxy URL must never pair with a real key. + // Best effort: a proxy problem leaves bash without the pair and never fails the turn. + // Untrusted projects and the project-automation kill switch blank provider keys in bash + // (gitNoRepoAutomationEnv), so repo code there must not spend through the proxy either. + // Add nothing rather than a proxy URL whose key the bash tool then blanks. + if (sharedExecutionTrusted) { + try { + Object.assign( + xumEnv, + await this.dependencies.bindings.bashAiProxy?.envFor( + workspaceId, + runtimeType, + Object.keys(projectSecretsRecord) + ) + ); + } catch (error) { + log.warn("[bash-ai-proxy] env setup failed; bash AI calls stay uncounted", { + workspaceId, + error: getErrorMessage(error), + }); + } + } const getWorkflowProjectTrusted = () => isWorkspaceProjectTrusted(this.dependencies.config, metadata); @@ -2369,7 +2398,7 @@ export class TurnRequestBuilder { cwd: workspacePath, runtime, projects: getProjects(metadata), - secrets: await secretsToRecord(projectSecrets), + secrets: projectSecretsRecord, xumEnv, runtimeTempDir, ...(advisorToolEligible