From b11a57f2cbb5021b358c77763593591dec74e3cb Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Sun, 4 Oct 2026 03:55:33 +0000 Subject: [PATCH 1/3] Initial plan From 933a23c19d37a42fd54cc1c74f05fe799ca4f132 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Sun, 4 Oct 2026 04:01:03 +0000 Subject: [PATCH 2/3] fix: report unsupported Cursor native token accounting Co-authored-by: lpcox <15877973+lpcox@users.noreply.github.com> --- containers/api-proxy/token-tracker-http.js | 22 +++ .../api-proxy/token-tracker.cursor.test.js | 130 ++++++++++++++++++ docs/api-proxy-sidecar.md | 26 ++++ 3 files changed, 178 insertions(+) create mode 100644 containers/api-proxy/token-tracker.cursor.test.js diff --git a/containers/api-proxy/token-tracker-http.js b/containers/api-proxy/token-tracker-http.js index 01abd76de..8f9440a17 100644 --- a/containers/api-proxy/token-tracker-http.js +++ b/containers/api-proxy/token-tracker-http.js @@ -391,6 +391,28 @@ function finalizeHttpTracking(state, proxyRes, opts) { const normalized = normalizeUsage(usage); if (!normalized) { + // The native Cursor stream is not an OpenAI completion just because it + // uses the OpenAI upstream route. No verified native usage decoder exists. + if (streaming && typeof reqPath === 'string' + && reqPath.split('?')[0] === '/agent.v1.AgentService/RunSSE') { + const accounting = { + provider, + path: '/agent.v1.AgentService/RunSSE', + status: proxyRes.statusCode, + streaming, + protocol: 'cursor-runsse', + reason: 'native_usage_contract_unverified', + }; + auditTrack('TRACK_END', { rid: requestId, result: 'unsupported_accounting', ...accounting }); + logRequest('warn', 'token_track_unsupported_accounting', { + request_id: requestId, + ...accounting, + message: 'Cursor native token usage could not be extracted: AWF has no verified native usage decoder. ' + + 'Token and AI-credit budgets cannot account for this response; missing usage does not mean zero cost.', + }); + if (typeof onSpanEnd === 'function') onSpanEnd(proxyRes.statusCode); + return; + } auditTrack('TRACK_END', { rid: requestId, result: 'no_usage', streaming, bytes: state.totalBytes, overflow: state.overflow, ct: state.contentType, ce: contentEncoding }); // Log at info level so failed extraction is visible in CI without debug mode logRequest('info', 'token_track_no_usage', { diff --git a/containers/api-proxy/token-tracker.cursor.test.js b/containers/api-proxy/token-tracker.cursor.test.js new file mode 100644 index 000000000..38a66026b --- /dev/null +++ b/containers/api-proxy/token-tracker.cursor.test.js @@ -0,0 +1,130 @@ +'use strict'; + +require('./test-helpers/token-tracker-setup'); + +jest.mock('./logging', () => ({ logRequest: jest.fn() })); +jest.mock('./token-persistence', () => ({ + ...jest.requireActual('./token-persistence'), + auditTrack: jest.fn(), + writeTokenUsage: jest.fn(), + incrementTokenMetrics: jest.fn(), +})); + +const { EventEmitter } = require('events'); +const { trackTokenUsage } = require('./token-tracker-http'); +const { auditTrack, writeTokenUsage, incrementTokenMetrics } = require('./token-persistence'); +const { logRequest } = require('./logging'); + +const RUN_SSE = '/agent.v1.AgentService/RunSSE'; + +function trackResponse({ path = RUN_SSE, contentType = 'text/event-stream', status = 200, body = '', ...opts } = {}) { + const response = new EventEmitter(); + response.headers = { 'content-type': contentType }; + response.statusCode = status; + const onUsage = jest.fn(); + const onSpanEnd = jest.fn(); + trackTokenUsage(response, { + requestId: 'shared-native-request', + provider: 'openai', + path, + startTime: Date.now(), + onUsage, + onSpanEnd, + ...opts, + }); + const bytes = Buffer.from(body); + response.emit('data', bytes.subarray(0, 7)); + response.emit('data', bytes.subarray(7)); + response.emit('end'); + response.emit('close'); + return { onUsage, onSpanEnd }; +} + +beforeEach(() => jest.clearAllMocks()); + +test('successful opaque native stream explicitly reports unsupported accounting without estimating tokens', () => { + // Synthetic opaque data exercises routing, not a captured Cursor usage contract. + const { onUsage, onSpanEnd } = trackResponse({ body: 'data: AAEC\n\n'.repeat(100) }); + expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', { + rid: 'shared-native-request', + result: 'unsupported_accounting', + provider: 'openai', + path: RUN_SSE, + status: 200, + streaming: true, + protocol: 'cursor-runsse', + reason: 'native_usage_contract_unverified', + }); + expect(logRequest).toHaveBeenCalledWith('warn', 'token_track_unsupported_accounting', + expect.objectContaining({ protocol: 'cursor-runsse', message: expect.stringContaining('does not mean zero cost') })); + expect(writeTokenUsage).not.toHaveBeenCalled(); + expect(incrementTokenMetrics).not.toHaveBeenCalled(); + expect(onUsage).not.toHaveBeenCalled(); + expect(onSpanEnd).toHaveBeenCalledTimes(1); + expect(onSpanEnd).toHaveBeenCalledWith(200); + expect(auditTrack.mock.calls.filter(([event]) => event === 'TRACK_END')).toHaveLength(1); +}); + +test('BidiAppend sharing a request ID cannot overwrite the native stream accounting state', () => { + trackResponse({ body: 'data: AAEC\n\n' }); + trackResponse({ + path: '/aiserver.v1.BidiService/BidiAppend', + contentType: 'application/proto', + body: '\x00\x01\x02', + }); + const ends = auditTrack.mock.calls.filter(([event]) => event === 'TRACK_END').map(([, record]) => record); + expect(ends).toHaveLength(2); + expect(ends[0]).toMatchObject({ rid: 'shared-native-request', streaming: true, result: 'unsupported_accounting', path: RUN_SSE }); + expect(ends[1]).toMatchObject({ rid: 'shared-native-request', streaming: false, result: 'no_usage' }); + expect(writeTokenUsage).not.toHaveBeenCalled(); +}); + +test('Connect-framed binary data advertised as text/event-stream is not treated as token usage', () => { + // Synthetic five-byte Connect envelopes, not an authoritative usage fixture. + const body = Buffer.from([0, 0, 0, 0, 2, 8, 1, 2, 0, 0, 0, 2, 123, 125]); + const { onUsage } = trackResponse({ body }); + expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', + expect.objectContaining({ result: 'unsupported_accounting', protocol: 'cursor-runsse' })); + expect(writeTokenUsage).not.toHaveBeenCalled(); + expect(onUsage).not.toHaveBeenCalled(); +}); + +test.each(['', 'data: [DONE]\n\n', 'data: {"unknownUsage":{"tokens":123}}\n\n'])( + 'missing native usage reports unsupported accounting (%j)', (body) => { + trackResponse({ body, path: `${RUN_SSE}?request=example` }); + expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', + expect.objectContaining({ result: 'unsupported_accounting', path: RUN_SSE })); + expect(writeTokenUsage).not.toHaveBeenCalled(); + }, +); + +test.each([ + { path: '/v1/chat/completions' }, + { contentType: 'application/proto' }, + { path: `${RUN_SSE}Other` }, +])('unrelated missing usage retains no_usage (%j)', (opts) => { + trackResponse(opts); + expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', expect.objectContaining({ result: 'no_usage' })); + expect(logRequest).not.toHaveBeenCalledWith('warn', 'token_track_unsupported_accounting', expect.anything()); +}); + +test('unsuccessful native responses retain skip_status', () => { + trackResponse({ status: 401 }); + expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', expect.objectContaining({ result: 'skip_status', status: 401 })); + expect(writeTokenUsage).not.toHaveBeenCalled(); +}); + +test.each([ + ['openai', [{ model: 'gpt-4o', usage: { prompt_tokens: 12, completion_tokens: 7 } }], 'gpt-4o'], + ['anthropic', [ + { type: 'message_start', message: { model: 'claude-sonnet-4', usage: { input_tokens: 12 } } }, + { type: 'message_delta', usage: { output_tokens: 7 } }, + ], 'claude-sonnet-4'], + ['gemini', [{ modelVersion: 'gemini-2.5-pro', usageMetadata: { promptTokenCount: 12, candidatesTokenCount: 7 } }], 'gemini-2.5-pro'], +])('recognized %s usage still reaches accounting even on the native route', (provider, events, model) => { + const body = events.map((event) => `data: ${JSON.stringify(event)}\n\n`).join(''); + const { onUsage } = trackResponse({ provider, body }); + expect(onUsage).toHaveBeenCalledWith(expect.objectContaining({ input_tokens: 12, output_tokens: 7 }), model); + expect(writeTokenUsage).toHaveBeenCalledWith(expect.objectContaining({ provider, model, input_tokens: 12, output_tokens: 7 })); + expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', expect.objectContaining({ result: 'ok' })); +}); diff --git a/docs/api-proxy-sidecar.md b/docs/api-proxy-sidecar.md index 35c2a02f0..8103d0151 100644 --- a/docs/api-proxy-sidecar.md +++ b/docs/api-proxy-sidecar.md @@ -1511,6 +1511,32 @@ into the api-proxy container, so no extra configuration is needed. ## Limitations +- **Cursor native token accounting is unsupported when no recognized usage is observable**: + routing `apiProxy.targets.openai.host` to `api2.cursor.sh` can successfully + proxy `/agent.v1.AgentService/RunSSE` without producing token records. + The native protocol is not an OpenAI completion protocol merely because it + uses the OpenAI route or advertises `text/event-stream`. + Third-party [interoperability implementations](https://github.com/leookun/cursor-byok) + describe optional turn-ended counts and Connect-framed protobuf messages, + but these do not verify authoritative counts or executed-model metadata for + Cursor CLI `2026.07.20-8cc9c0b`. The reported smoke run did not capture the raw + response. Cursor's [official SDK contract](https://github.com/cursor/sdk-bridge/blob/main/proto/sdk/v1/sdk_agent_service.proto) + exposes cloud usage through a different service, not native RunSSE; AWF does + not assume those contracts are interchangeable. + Until a native usage contract is verified, a 2xx RunSSE stream without + recognized usage emits `TRACK_END` with `result: "unsupported_accounting"`, + `protocol: "cursor-runsse"`, `reason: "native_usage_contract_unverified"`, + `provider`, `path`, `status`, and `streaming: true` in the always-on + `token-tracker-audit.jsonl`, plus a `token_track_unsupported_accounting` + warning. Consumers must inspect these records rather than interpret an empty + `token-usage.jsonl` as zero cost or an emitter failure. Match the native + `path` and `streaming: true`, not just `rid`: BidiAppend calls can reuse the + request ID and have their own `no_usage` result. + No zero-token record or byte-derived estimate is written. Effective-token + and AI-credit budgets cannot account for these responses, so their totals + are incomplete and must not be relied on as spending limits for this + protocol. Recognized OpenAI, Anthropic, and Gemini usage continues through + normal accounting. Inference routing and authentication are unchanged. - Keys must be set as environment variables (not file-based) - No request/response logging (by design, for security) - **AWS Bedrock OIDC signs HTTP requests only**: WebSocket upgrades are rejected, and the signing target is restricted to the exact regional Bedrock Runtime hostname. See [OIDC Authentication > AWS Bedrock](#aws-bedrock). From c560ca5fac45d4294b0689472b9bacccab300c1b Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Sun, 4 Oct 2026 15:56:02 +0000 Subject: [PATCH 3/3] fix: report unsupported accounting for compressed Cursor streams Co-authored-by: lpcox <15877973+lpcox@users.noreply.github.com> --- containers/api-proxy/token-tracker-http.js | 49 +++++++++++++------ .../api-proxy/token-tracker.cursor.test.js | 30 +++++++++++- 2 files changed, 64 insertions(+), 15 deletions(-) diff --git a/containers/api-proxy/token-tracker-http.js b/containers/api-proxy/token-tracker-http.js index 8f9440a17..b27176451 100644 --- a/containers/api-proxy/token-tracker-http.js +++ b/containers/api-proxy/token-tracker-http.js @@ -345,6 +345,25 @@ function buildAndWriteTokenRecord(normalized, { requestId, provider, model, reqP }); } +function reportUnsupportedCursorAccounting(requestId, provider, status, reason, contentEncoding) { + const accounting = { + provider, + path: '/agent.v1.AgentService/RunSSE', + status, + streaming: true, + protocol: 'cursor-runsse', + reason, + ...(contentEncoding ? { content_encoding: contentEncoding } : {}), + }; + auditTrack('TRACK_END', { rid: requestId, result: 'unsupported_accounting', ...accounting }); + logRequest('warn', 'token_track_unsupported_accounting', { + request_id: requestId, + ...accounting, + message: 'Cursor native token usage could not be extracted: AWF has no verified native usage decoder. ' + + 'Token and AI-credit budgets cannot account for this response; missing usage does not mean zero cost.', + }); +} + /** * Finalize token tracking for an HTTP response. * @@ -395,21 +414,12 @@ function finalizeHttpTracking(state, proxyRes, opts) { // uses the OpenAI upstream route. No verified native usage decoder exists. if (streaming && typeof reqPath === 'string' && reqPath.split('?')[0] === '/agent.v1.AgentService/RunSSE') { - const accounting = { + reportUnsupportedCursorAccounting( + requestId, provider, - path: '/agent.v1.AgentService/RunSSE', - status: proxyRes.statusCode, - streaming, - protocol: 'cursor-runsse', - reason: 'native_usage_contract_unverified', - }; - auditTrack('TRACK_END', { rid: requestId, result: 'unsupported_accounting', ...accounting }); - logRequest('warn', 'token_track_unsupported_accounting', { - request_id: requestId, - ...accounting, - message: 'Cursor native token usage could not be extracted: AWF has no verified native usage decoder. ' + - 'Token and AI-credit budgets cannot account for this response; missing usage does not mean zero cost.', - }); + proxyRes.statusCode, + 'native_usage_contract_unverified', + ); if (typeof onSpanEnd === 'function') onSpanEnd(proxyRes.statusCode); return; } @@ -529,6 +539,17 @@ function trackTokenUsage(proxyRes, opts) { 'The Accept-Encoding sanitizer should have prevented this — check if the client bypassed the proxy header rewrite.', }); diag('HTTP_TRACK_UNSUPPORTED_ENCODING', { request_id: requestId, provider, path: reqPath, content_encoding: contentEncoding }); + if (proxyRes.statusCode >= 200 && proxyRes.statusCode < 300 + && streaming && typeof reqPath === 'string' + && reqPath.split('?')[0] === '/agent.v1.AgentService/RunSSE') { + reportUnsupportedCursorAccounting( + requestId, + provider, + proxyRes.statusCode, + `unsupported_content_encoding_${contentEncoding}`, + contentEncoding, + ); + } if (typeof opts.onSpanEnd === 'function') opts.onSpanEnd(proxyRes.statusCode); return; } diff --git a/containers/api-proxy/token-tracker.cursor.test.js b/containers/api-proxy/token-tracker.cursor.test.js index 38a66026b..05fe76e39 100644 --- a/containers/api-proxy/token-tracker.cursor.test.js +++ b/containers/api-proxy/token-tracker.cursor.test.js @@ -17,9 +17,10 @@ const { logRequest } = require('./logging'); const RUN_SSE = '/agent.v1.AgentService/RunSSE'; -function trackResponse({ path = RUN_SSE, contentType = 'text/event-stream', status = 200, body = '', ...opts } = {}) { +function trackResponse({ path = RUN_SSE, contentType = 'text/event-stream', contentEncoding, status = 200, body = '', ...opts } = {}) { const response = new EventEmitter(); response.headers = { 'content-type': contentType }; + if (contentEncoding) response.headers['content-encoding'] = contentEncoding; response.statusCode = status; const onUsage = jest.fn(); const onSpanEnd = jest.fn(); @@ -108,6 +109,33 @@ test.each([ expect(logRequest).not.toHaveBeenCalledWith('warn', 'token_track_unsupported_accounting', expect.anything()); }); +test('successful native streams with unsupported encoding report unsupported accounting without a token record', () => { + trackResponse({ contentEncoding: 'zstd', body: 'opaque compressed bytes' }); + expect(auditTrack).toHaveBeenCalledWith('TRACK_SKIP_ENCODING', { + rid: 'shared-native-request', + ce: 'zstd', + }); + expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', { + rid: 'shared-native-request', + result: 'unsupported_accounting', + provider: 'openai', + path: RUN_SSE, + status: 200, + streaming: true, + protocol: 'cursor-runsse', + reason: 'unsupported_content_encoding_zstd', + content_encoding: 'zstd', + }); + expect(logRequest).toHaveBeenCalledWith('warn', 'token_track_unsupported_accounting', + expect.objectContaining({ + protocol: 'cursor-runsse', + reason: 'unsupported_content_encoding_zstd', + content_encoding: 'zstd', + })); + expect(writeTokenUsage).not.toHaveBeenCalled(); + expect(incrementTokenMetrics).not.toHaveBeenCalled(); +}); + test('unsuccessful native responses retain skip_status', () => { trackResponse({ status: 401 }); expect(auditTrack).toHaveBeenLastCalledWith('TRACK_END', expect.objectContaining({ result: 'skip_status', status: 401 }));