From c6e3db5397c7d10da4d7ef7801072763f4623063 Mon Sep 17 00:00:00 2001 From: Vivek shukla Date: Wed, 5 Aug 2026 02:08:26 +0530 Subject: [PATCH] feat: add Cursor CLI agent harness and experiments Enable Composer 2.5 and Grok 4.5 evals via cursor-agent with stream-json parsing, MCP/skills support, and benchmark/no-skills experiment variants. Co-authored-by: Cursor --- .env.example | 3 + README.md | 2 +- apps/web/src/lib/format.ts | 1 + experiments/cursor-composer-2.5-no-skills.ts | 22 ++ experiments/cursor-composer-2.5.ts | 19 ++ experiments/cursor-grok-4.5-high-no-skills.ts | 22 ++ experiments/cursor-grok-4.5-high.ts | 19 ++ .../fixtures/stream-json-docs-sample.jsonl | 10 + .../stream-json-harbor-tool-shapes.jsonl | 5 + .../stream-json-live-composer-2.5.jsonl | 8 + .../stream-json-live-mcp-composer-2.5.jsonl | 21 ++ packages/core/src/agents/cursor/index.test.ts | 33 ++ packages/core/src/agents/cursor/index.ts | 40 +++ .../core/src/agents/cursor/parser.test.ts | 149 +++++++++ packages/core/src/agents/cursor/parser.ts | 302 ++++++++++++++++++ .../core/src/agents/cursor/runner.test.ts | 202 ++++++++++++ packages/core/src/agents/cursor/runner.ts | 180 +++++++++++ packages/core/src/agents/engine.ts | 2 + packages/core/src/agents/registry.ts | 2 + packages/core/src/eval-metadata.ts | 2 + packages/core/src/index.ts | 3 +- 21 files changed, 1045 insertions(+), 2 deletions(-) create mode 100644 experiments/cursor-composer-2.5-no-skills.ts create mode 100644 experiments/cursor-composer-2.5.ts create mode 100644 experiments/cursor-grok-4.5-high-no-skills.ts create mode 100644 experiments/cursor-grok-4.5-high.ts create mode 100644 packages/core/src/agents/cursor/fixtures/stream-json-docs-sample.jsonl create mode 100644 packages/core/src/agents/cursor/fixtures/stream-json-harbor-tool-shapes.jsonl create mode 100644 packages/core/src/agents/cursor/fixtures/stream-json-live-composer-2.5.jsonl create mode 100644 packages/core/src/agents/cursor/fixtures/stream-json-live-mcp-composer-2.5.jsonl create mode 100644 packages/core/src/agents/cursor/index.test.ts create mode 100644 packages/core/src/agents/cursor/index.ts create mode 100644 packages/core/src/agents/cursor/parser.test.ts create mode 100644 packages/core/src/agents/cursor/parser.ts create mode 100644 packages/core/src/agents/cursor/runner.test.ts create mode 100644 packages/core/src/agents/cursor/runner.ts diff --git a/.env.example b/.env.example index 53fb4854..0e36c76b 100644 --- a/.env.example +++ b/.env.example @@ -8,3 +8,6 @@ OPENAI_API_KEY= # Vercel AI Gateway. The opencode harness routes through this. AI_GATEWAY_API_KEY= + +# Cursor CLI harness (composer-2.5 / cursor-grok-4.5-high experiments). +CURSOR_API_KEY= diff --git a/README.md b/README.md index 31527982..70df1ba9 100644 --- a/README.md +++ b/README.md @@ -23,7 +23,7 @@ pnpm install cp .env.example .env ``` -Agent-backed runs require the relevant provider key in `.env` (e.g. `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`) +Agent-backed runs require the relevant provider key in `.env` (e.g. `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `CURSOR_API_KEY`). Rubric/judge checks also need `OPENAI_API_KEY` regardless of harness. ## Concepts diff --git a/apps/web/src/lib/format.ts b/apps/web/src/lib/format.ts index 12fc4d22..2685d26b 100644 --- a/apps/web/src/lib/format.ts +++ b/apps/web/src/lib/format.ts @@ -6,6 +6,7 @@ export const AGENT_LABELS = { "ai-sdk": "AI SDK", "claude-code": "Claude Code", codex: "Codex", + cursor: "Cursor", opencode: "OpenCode", } satisfies Record diff --git a/experiments/cursor-composer-2.5-no-skills.ts b/experiments/cursor-composer-2.5-no-skills.ts new file mode 100644 index 00000000..6b94dc20 --- /dev/null +++ b/experiments/cursor-composer-2.5-no-skills.ts @@ -0,0 +1,22 @@ +import { + cursorAgent, + defineExperiment, + platformLiteRuntime, + supabaseMcpServer, +} from '@supabase-evals/core'; +import { localStackRuntime } from '@supabase-evals/sandbox'; + +// Same as cursor-composer-2.5 but with no skills, to measure skills' impact. +export default defineExperiment({ + suite: ['no-skills', 'regression'], + agent: cursorAgent({ + model: 'composer-2.5', + }), + runtime: platformLiteRuntime({ + mcpServers: [supabaseMcpServer()], + }), + localStack: localStackRuntime(), + skills: [], + // Evals that override `skills: []` already run under the baseline experiment. Skip them from running again here. + skipEval: (ev) => ev.metadata.skills?.length === 0, +}); diff --git a/experiments/cursor-composer-2.5.ts b/experiments/cursor-composer-2.5.ts new file mode 100644 index 00000000..ca7f261f --- /dev/null +++ b/experiments/cursor-composer-2.5.ts @@ -0,0 +1,19 @@ +import { + cursorAgent, + defineExperiment, + platformLiteRuntime, + supabaseMcpServer, +} from '@supabase-evals/core'; +import { localStackRuntime } from '@supabase-evals/sandbox'; + +export default defineExperiment({ + suite: ['benchmark', 'regression'], + agent: cursorAgent({ + model: 'composer-2.5', + }), + runtime: platformLiteRuntime({ + mcpServers: [supabaseMcpServer()], + }), + localStack: localStackRuntime(), + skills: ['supabase', 'supabase-postgres-best-practices'], +}); diff --git a/experiments/cursor-grok-4.5-high-no-skills.ts b/experiments/cursor-grok-4.5-high-no-skills.ts new file mode 100644 index 00000000..88fd1b6d --- /dev/null +++ b/experiments/cursor-grok-4.5-high-no-skills.ts @@ -0,0 +1,22 @@ +import { + cursorAgent, + defineExperiment, + platformLiteRuntime, + supabaseMcpServer, +} from '@supabase-evals/core'; +import { localStackRuntime } from '@supabase-evals/sandbox'; + +// Same as cursor-grok-4.5-high but with no skills, to measure skills' impact. +export default defineExperiment({ + suite: ['no-skills', 'regression'], + agent: cursorAgent({ + model: 'cursor-grok-4.5-high', + }), + runtime: platformLiteRuntime({ + mcpServers: [supabaseMcpServer()], + }), + localStack: localStackRuntime(), + skills: [], + // Evals that override `skills: []` already run under the baseline experiment. Skip them from running again here. + skipEval: (ev) => ev.metadata.skills?.length === 0, +}); diff --git a/experiments/cursor-grok-4.5-high.ts b/experiments/cursor-grok-4.5-high.ts new file mode 100644 index 00000000..952b8a56 --- /dev/null +++ b/experiments/cursor-grok-4.5-high.ts @@ -0,0 +1,19 @@ +import { + cursorAgent, + defineExperiment, + platformLiteRuntime, + supabaseMcpServer, +} from '@supabase-evals/core'; +import { localStackRuntime } from '@supabase-evals/sandbox'; + +export default defineExperiment({ + suite: ['benchmark', 'regression'], + agent: cursorAgent({ + model: 'cursor-grok-4.5-high', + }), + runtime: platformLiteRuntime({ + mcpServers: [supabaseMcpServer()], + }), + localStack: localStackRuntime(), + skills: ['supabase', 'supabase-postgres-best-practices'], +}); diff --git a/packages/core/src/agents/cursor/fixtures/stream-json-docs-sample.jsonl b/packages/core/src/agents/cursor/fixtures/stream-json-docs-sample.jsonl new file mode 100644 index 00000000..2b8515ae --- /dev/null +++ b/packages/core/src/agents/cursor/fixtures/stream-json-docs-sample.jsonl @@ -0,0 +1,10 @@ +{"type":"system","subtype":"init","apiKeySource":"login","cwd":"/Users/user/project","session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff","model":"Claude 4 Sonnet","permissionMode":"default"} +{"type":"user","message":{"role":"user","content":[{"type":"text","text":"Read README.md and create a summary"}]},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"assistant","message":{"role":"assistant","content":[{"type":"text","text":"I'll read the README.md file"}]},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"tool_call","subtype":"started","call_id":"toolu_vrtx_01NnjaR886UcE8whekg2MGJd","tool_call":{"readToolCall":{"args":{"path":"README.md"}}},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"tool_call","subtype":"completed","call_id":"toolu_vrtx_01NnjaR886UcE8whekg2MGJd","tool_call":{"readToolCall":{"args":{"path":"README.md"},"result":{"success":{"content":"# Project\n\nThis is a sample project...","isEmpty":false,"exceededLimit":false,"totalLines":54,"totalChars":1254}}}},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"assistant","message":{"role":"assistant","content":[{"type":"text","text":"Based on the README, I'll create a summary"}]},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"tool_call","subtype":"started","call_id":"toolu_vrtx_01Q3VHVnWFSKygaRPT7WDxrv","tool_call":{"writeToolCall":{"args":{"path":"summary.txt","fileText":"# README Summary\n\nThis project contains...","toolCallId":"toolu_vrtx_01Q3VHVnWFSKygaRPT7WDxrv"}}},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"tool_call","subtype":"completed","call_id":"toolu_vrtx_01Q3VHVnWFSKygaRPT7WDxrv","tool_call":{"writeToolCall":{"args":{"path":"summary.txt","fileText":"# README Summary\n\nThis project contains...","toolCallId":"toolu_vrtx_01Q3VHVnWFSKygaRPT7WDxrv"},"result":{"success":{"path":"/Users/user/project/summary.txt","linesCreated":19,"fileSize":942}}}},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"assistant","message":{"role":"assistant","content":[{"type":"text","text":"Done! I've created the summary in summary.txt"}]},"session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff"} +{"type":"result","subtype":"success","duration_ms":5234,"duration_api_ms":5234,"is_error":false,"result":"I'll read the README.md fileBased on the README, I'll create a summaryDone! I've created the summary in summary.txt","session_id":"c6b62c6f-7ead-4fd6-9922-e952131177ff","request_id":"10e11780-df2f-45dc-a1ff-4540af32e9c0"} diff --git a/packages/core/src/agents/cursor/fixtures/stream-json-harbor-tool-shapes.jsonl b/packages/core/src/agents/cursor/fixtures/stream-json-harbor-tool-shapes.jsonl new file mode 100644 index 00000000..f5e4674d --- /dev/null +++ b/packages/core/src/agents/cursor/fixtures/stream-json-harbor-tool-shapes.jsonl @@ -0,0 +1,5 @@ +{"type":"system","subtype":"init","apiKeySource":"env","cwd":"/workspace","session_id":"session-1","model":"composer-2.5","permissionMode":"default"} +{"type":"assistant","session_id":"session-1","model_call_id":"model-call-1","timestamp_ms":1700000000000,"message":{"role":"assistant","content":[{"type":"text","text":"Using tools now."}]}} +{"type":"tool_call","subtype":"completed","call_id":"call-1","session_id":"session-1","model_call_id":"model-call-1","timestamp_ms":1700000000000,"tool_call":{"shellToolCall":{"args":{"command":"ls"},"result":{"success":{"stdout":"README.md"}}}}} +{"type":"tool_call","subtype":"completed","call_id":"call-2","session_id":"session-1","model_call_id":"model-call-1","timestamp_ms":1700000001000,"tool_call":{"write_file":{"args":{"path":"/tmp/out.txt"},"result":"file created successfully"}}} +{"type":"result","subtype":"success","duration_ms":1000,"duration_api_ms":1000,"is_error":false,"result":"Using tools now.","session_id":"session-1"} diff --git a/packages/core/src/agents/cursor/fixtures/stream-json-live-composer-2.5.jsonl b/packages/core/src/agents/cursor/fixtures/stream-json-live-composer-2.5.jsonl new file mode 100644 index 00000000..b8c92519 --- /dev/null +++ b/packages/core/src/agents/cursor/fixtures/stream-json-live-composer-2.5.jsonl @@ -0,0 +1,8 @@ +{"type":"system","subtype":"init","apiKeySource":"env","cwd":"/home/vivek/Desktop/projects/xulux/evals/evalprojects/docs/modules/m0-discovery/tmp-workspace","session_id":"cd97ac09-d300-408d-9f66-385fdb57f261","model":"Composer 2.5","permissionMode":"default"} +{"type":"user","message":{"role":"user","content":[{"type":"text","text":"Reply with exactly: PONG. Do not use tools."}]},"session_id":"cd97ac09-d300-408d-9f66-385fdb57f261"} +{"type":"thinking","subtype":"delta","text":"The response will be","session_id":"cd97ac09-d300-408d-9f66-385fdb57f261","timestamp_ms":1785848255137} +{"type":"thinking","subtype":"delta","text":" exactly \"PONG\" without","session_id":"cd97ac09-d300-408d-9f66-385fdb57f261","timestamp_ms":1785848255142} +{"type":"thinking","subtype":"delta","text":" using tools.","session_id":"cd97ac09-d300-408d-9f66-385fdb57f261","timestamp_ms":1785848255143} +{"type":"thinking","subtype":"completed","session_id":"cd97ac09-d300-408d-9f66-385fdb57f261","timestamp_ms":1785848255144} +{"type":"assistant","message":{"role":"assistant","content":[{"type":"text","text":"PONG"}]},"session_id":"cd97ac09-d300-408d-9f66-385fdb57f261"} +{"type":"result","subtype":"success","duration_ms":7708,"duration_api_ms":7708,"is_error":false,"result":"PONG","session_id":"cd97ac09-d300-408d-9f66-385fdb57f261","request_id":"23d47179-318d-4c01-9dd6-8d1fbe54e8b0","usage":{"inputTokens":5544,"outputTokens":28,"cacheReadTokens":7002,"cacheWriteTokens":0}} diff --git a/packages/core/src/agents/cursor/fixtures/stream-json-live-mcp-composer-2.5.jsonl b/packages/core/src/agents/cursor/fixtures/stream-json-live-mcp-composer-2.5.jsonl new file mode 100644 index 00000000..d7737cf5 --- /dev/null +++ b/packages/core/src/agents/cursor/fixtures/stream-json-live-mcp-composer-2.5.jsonl @@ -0,0 +1,21 @@ +{"type":"system","subtype":"init","apiKeySource":"env","cwd":"/home/vivek/Desktop/projects/xulux/evals/evalprojects/docs/modules/m0-discovery/tmp-workspace","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","model":"Composer 2.5","permissionMode":"default"} +{"type":"user","message":{"role":"user","content":[{"type":"text","text":"You have an MCP server m0-ping with tool m0_ping. Call m0_ping once, then reply with the tool result text only."}]},"session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58"} +{"type":"thinking","subtype":"delta","text":"I will call the m0_ping","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345075} +{"type":"thinking","subtype":"delta","text":" tool once and return","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345080} +{"type":"thinking","subtype":"delta","text":" only its result text.","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345084} +{"type":"thinking","subtype":"delta","text":"\n\nFirst I will fetch","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345086} +{"type":"thinking","subtype":"delta","text":" the MCP tool schema.","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345090} +{"type":"thinking","subtype":"completed","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345093} +{"type":"tool_call","subtype":"started","call_id":"tool_835acca9-b527-4742-b102-f346c24fc89","tool_call":{"getMcpToolsToolCall":{"args":{"server":"m0-ping","toolName":"m0_ping","toolCallId":"tool_835acca9-b527-4742-b102-f346c24fc89"}},"hookAdditionalContexts":[],"toolCallId":"tool_835acca9-b527-4742-b102-f346c24fc89","startedAtMs":"1785848344974"},"model_call_id":"55387b0b-56af-40ec-b2b1-c277b120d0cd-0-iosm","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345103} +{"type":"tool_call","subtype":"completed","call_id":"tool_835acca9-b527-4742-b102-f346c24fc89","tool_call":{"getMcpToolsToolCall":{"args":{"server":"m0-ping","toolName":"m0_ping","toolCallId":"tool_835acca9-b527-4742-b102-f346c24fc89"},"result":{"success":{"content":"{\n \"mode\": \"single_tool\",\n \"server\": \"m0-ping\",\n \"serverStatus\": \"ready\",\n \"tool\": {\n \"tool\": \"m0_ping\",\n \"description\": \"Return pong for Module 0 MCP probe\",\n \"inputSchema\": {\n \"type\": \"object\",\n \"properties\": {\n \"note\": {\n \"type\": \"string\"\n }\n },\n \"additionalProperties\": false\n }\n }\n}"}}},"hookAdditionalContexts":[],"toolCallId":"tool_835acca9-b527-4742-b102-f346c24fc89","startedAtMs":"1785848344974","completedAtMs":"1785848345191"},"model_call_id":"55387b0b-56af-40ec-b2b1-c277b120d0cd-0-iosm","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848345229} +{"type":"thinking","subtype":"delta","text":"Calling m0_ping now.","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848348100} +{"type":"thinking","subtype":"completed","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848348103} +{"type":"tool_call","subtype":"started","call_id":"tool_35ba9454-cd3d-4cd9-89fd-3ee3afc9263","tool_call":{"mcpToolCall":{"args":{"name":"m0-ping-m0_ping","args":{},"toolCallId":"tool_35ba9454-cd3d-4cd9-89fd-3ee3afc9263","providerIdentifier":"m0-ping","toolName":"m0_ping","smartModeApprovalOnly":false,"skipApproval":false,"serverIdentifier":"m0-ping"},"description":"Call m0_ping probe tool once"},"hookAdditionalContexts":[],"toolCallId":"tool_35ba9454-cd3d-4cd9-89fd-3ee3afc9263","startedAtMs":"1785848348059"},"model_call_id":"55387b0b-56af-40ec-b2b1-c277b120d0cd-1-8ucj","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848348105} +{"type":"tool_call","subtype":"completed","call_id":"tool_35ba9454-cd3d-4cd9-89fd-3ee3afc9263","tool_call":{"mcpToolCall":{"args":{"name":"m0-ping-m0_ping","args":{},"toolCallId":"tool_35ba9454-cd3d-4cd9-89fd-3ee3afc9263","providerIdentifier":"m0-ping","toolName":"m0_ping","smartModeApprovalOnly":false,"skipApproval":false,"serverIdentifier":"m0-ping"},"result":{"success":{"content":[{"text":{"text":"PONG_FROM_MCP"}}],"isError":false}},"description":"Call m0_ping probe tool once"},"hookAdditionalContexts":[],"toolCallId":"tool_35ba9454-cd3d-4cd9-89fd-3ee3afc9263","startedAtMs":"1785848348059","completedAtMs":"1785848348236"},"model_call_id":"55387b0b-56af-40ec-b2b1-c277b120d0cd-1-8ucj","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848348279} +{"type":"thinking","subtype":"delta","text":"I will call the m0_ping","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848351924} +{"type":"thinking","subtype":"delta","text":" tool once and return","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848351928} +{"type":"thinking","subtype":"delta","text":" only the result text.","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848351929} +{"type":"thinking","subtype":"delta","text":"\n\nThe result is \"PONG_FROM_MCP\".","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848351931} +{"type":"thinking","subtype":"completed","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","timestamp_ms":1785848351932} +{"type":"assistant","message":{"role":"assistant","content":[{"type":"text","text":"PONG_FROM_MCP"}]},"session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58"} +{"type":"result","subtype":"success","duration_ms":13623,"duration_api_ms":13623,"is_error":false,"result":"PONG_FROM_MCP","session_id":"8c78e182-9de5-49a1-b412-5e9689b0be58","request_id":"55387b0b-56af-40ec-b2b1-c277b120d0cd","usage":{"inputTokens":4998,"outputTokens":177,"cacheReadTokens":36298,"cacheWriteTokens":0}} diff --git a/packages/core/src/agents/cursor/index.test.ts b/packages/core/src/agents/cursor/index.test.ts new file mode 100644 index 00000000..cdc10a87 --- /dev/null +++ b/packages/core/src/agents/cursor/index.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest'; +import { createParser, supportedParsers } from '../registry.js'; +import { cursorAgent, cursorDefinition } from './index.js'; +import { cursorParser } from './parser.js'; + +describe('cursorAgent wiring', () => { + it('registers the cursor parser in the agent registry', () => { + expect(supportedParsers()).toContain('cursor'); + expect(createParser('cursor')).toBe(cursorParser); + expect(cursorDefinition.parser).toBe(cursorParser); + }); + + it('builds a harness with cursor metadata defaults', () => { + const harness = cursorAgent(); + expect(harness.id).toBe('cursor'); + expect(harness.modelId).toBe('composer-2.5'); + expect(harness.metadata).toEqual({ + agent: 'cursor', + modelProvider: 'cursor', + modelId: 'composer-2.5', + }); + expect(harness.runsInSandbox).toBe(true); + }); + + it('accepts model and reasoningEffort overrides', () => { + const harness = cursorAgent({ + model: 'cursor-grok-4.5-high', + reasoningEffort: 'high', + }); + expect(harness.modelId).toBe('cursor-grok-4.5-high'); + expect(harness.metadata.reasoningEffort).toBe('high'); + }); +}); diff --git a/packages/core/src/agents/cursor/index.ts b/packages/core/src/agents/cursor/index.ts new file mode 100644 index 00000000..5614aee8 --- /dev/null +++ b/packages/core/src/agents/cursor/index.ts @@ -0,0 +1,40 @@ +/** + * Cursor agent. Owns everything Cursor-specific: it wires its own runner + + * parser into the public `cursorAgent` factory (via the generic `createCliAgent` + * engine) and exports the registry definition the harness uses to parse Cursor + * transcripts. + */ + +import type { AgentHarness } from '../../index.js'; +import type { ReasoningEffortLevel } from '../../eval-metadata.js'; +import { createCliAgent } from '../engine.js'; +import type { AgentDefinition } from '../types.js'; +import { cursorParser } from './parser.js'; +import { cursorRunner, type CursorModel } from './runner.js'; + +/** Cursor CLI agent as an `AgentHarness`. */ +export function cursorAgent( + options: { + /** Cursor model id from `cursor-agent --list-models` (e.g. composer-2.5). */ + model?: CursorModel; + /** + * Reasoning effort applied as Cursor's `model[effort=…]` syntax. + * Omit to use Cursor's default for the model. + */ + reasoningEffort?: ReasoningEffortLevel; + /** Override the pinned CLI tarball version. */ + cliVersion?: string; + } = {} +): AgentHarness { + return createCliAgent(cursorRunner, cursorParser, { + model: options.model ?? cursorRunner.defaultModel, + reasoningEffort: options.reasoningEffort, + cliVersion: options.cliVersion, + }); +} + +/** Runner + parser pairing for the agent registry (id comes from `runner.id`). */ +export const cursorDefinition: AgentDefinition = { + runner: cursorRunner, + parser: cursorParser, +}; diff --git a/packages/core/src/agents/cursor/parser.test.ts b/packages/core/src/agents/cursor/parser.test.ts new file mode 100644 index 00000000..5ebb6c2b --- /dev/null +++ b/packages/core/src/agents/cursor/parser.test.ts @@ -0,0 +1,149 @@ +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { describe, expect, it } from 'vitest'; +import { adaptTranscript } from '../../parsers/adapt.js'; +import { cursorParser } from './parser.js'; + +const FIXTURES = join(dirname(fileURLToPath(import.meta.url)), 'fixtures'); + +function loadFixture(name: string): string { + return readFileSync(join(FIXTURES, name), 'utf8'); +} + +describe('cursorParser', () => { + it('parses a live Composer ask session (thinking + assistant + result)', () => { + const raw = loadFixture('stream-json-live-composer-2.5.jsonl'); + const { events, errors } = cursorParser.parseTranscript(raw); + expect(errors).toEqual([]); + + expect(events.some((e) => e.type === 'message' && e.role === 'user')).toBe( + true + ); + expect( + events.some((e) => e.type === 'message' && e.role === 'assistant') + ).toBe(true); + expect(events.some((e) => e.type === 'thinking')).toBe(true); + + const thinking = events.find((e) => e.type === 'thinking'); + expect(thinking?.content).toContain('PONG'); + + const assistant = events.filter( + (e) => e.type === 'message' && e.role === 'assistant' + ); + // Result line repeats "PONG" — must not double-emit. + expect(assistant.map((e) => e.content)).toEqual(['PONG']); + }); + + it('parses live MCP tool use into tool_call + tool_result', () => { + const raw = loadFixture('stream-json-live-mcp-composer-2.5.jsonl'); + const { events, errors } = cursorParser.parseTranscript(raw); + expect(errors).toEqual([]); + + const toolCalls = events.filter((e) => e.type === 'tool_call'); + const toolResults = events.filter((e) => e.type === 'tool_result'); + expect(toolCalls.length).toBeGreaterThanOrEqual(1); + expect(toolResults.length).toBe(toolCalls.length); + + const mcpCall = toolCalls.find((e) => e.tool?.originalName === 'm0_ping'); + expect(mcpCall?.tool?.name).toBe('tool_use'); + expect(mcpCall?.tool?.id).toBeTruthy(); + + const mcpResult = toolResults.find((e) => e.tool?.id === mcpCall?.tool?.id); + expect(mcpResult?.tool?.success).toBe(true); + expect(JSON.stringify(mcpResult?.tool?.result)).toContain('PONG_FROM_MCP'); + }); + + it('normalizes docs-sample read/write tool calls', () => { + const raw = loadFixture('stream-json-docs-sample.jsonl'); + const { events, errors } = cursorParser.parseTranscript(raw); + expect(errors).toEqual([]); + + const toolCalls = events.filter((e) => e.type === 'tool_call'); + expect(toolCalls.map((e) => e.tool?.name)).toEqual([ + 'file_read', + 'file_write', + ]); + expect(toolCalls[0].tool?.path).toBe('README.md'); + expect(toolCalls[1].tool?.path).toBe('summary.txt'); + // started events are skipped — only completed pairs + expect(events.filter((e) => e.type === 'tool_result')).toHaveLength(2); + }); + + it('normalizes Harbor shellToolCall shapes', () => { + const raw = loadFixture('stream-json-harbor-tool-shapes.jsonl'); + const { events, errors } = cursorParser.parseTranscript(raw); + expect(errors).toEqual([]); + + const shell = events.find( + (e) => e.type === 'tool_call' && e.tool?.name === 'shell' + ); + expect(shell?.tool?.command).toBe('ls'); + expect(shell?.tool?.originalName).toBe('shellToolCall'); + }); + + it('skips started tool_call events', () => { + const transcript = [ + JSON.stringify({ + type: 'tool_call', + subtype: 'started', + call_id: 'c1', + tool_call: { readToolCall: { args: { path: 'a.txt' } } }, + }), + JSON.stringify({ + type: 'tool_call', + subtype: 'completed', + call_id: 'c1', + tool_call: { + readToolCall: { + args: { path: 'a.txt' }, + result: { success: { content: 'hi' } }, + }, + }, + }), + ].join('\n'); + + const { events } = cursorParser.parseTranscript(transcript); + expect(events.filter((e) => e.type === 'tool_call')).toHaveLength(1); + expect(events.filter((e) => e.type === 'tool_result')).toHaveLength(1); + }); + + it('never throws on malformed lines and records parse errors', () => { + const { events, errors } = cursorParser.parseTranscript( + 'not json\n' + JSON.stringify({ type: 'system', subtype: 'init' }) + ); + expect(events).toEqual([]); + expect(errors.length).toBe(1); + }); + + it('emits an error event when the result line is_error', () => { + const transcript = JSON.stringify({ + type: 'result', + subtype: 'error', + is_error: true, + result: 'something broke', + }); + const { events } = cursorParser.parseTranscript(transcript); + expect(events.filter((e) => e.type === 'error')).toEqual([ + expect.objectContaining({ content: 'something broke' }), + ]); + }); +}); + +describe('adaptTranscript (cursor fixtures)', () => { + it('derives report and tool calls from the live MCP session', () => { + const raw = loadFixture('stream-json-live-mcp-composer-2.5.jsonl'); + const { events } = cursorParser.parseTranscript(raw); + const adapted = adaptTranscript(events); + + expect(adapted.agentReport).toBe('PONG_FROM_MCP'); + expect(adapted.toolCalls.length).toBeGreaterThanOrEqual(1); + expect( + adapted.toolCalls.some( + (t) => + t.endpoint.includes('m0_ping') || + JSON.stringify(t).includes('m0_ping') + ) + ).toBe(true); + }); +}); diff --git a/packages/core/src/agents/cursor/parser.ts b/packages/core/src/agents/cursor/parser.ts new file mode 100644 index 00000000..9615e8a4 --- /dev/null +++ b/packages/core/src/agents/cursor/parser.ts @@ -0,0 +1,302 @@ +/** + * Cursor transcript parser. + * + * Parses the stream-json NDJSON that `cursor-agent --print --output-format=stream-json` + * writes to stdout (see ../runner.ts). Event shapes follow Cursor's CLI docs and + * Harbor's cursor-cli trajectory mapper; we emit supabase canonical + * `TranscriptEvent`s (not ATIF). + */ + +import type { + ParsedTranscript, + TranscriptEvent, +} from '../../transcript/types.js'; +import type { AgentTranscriptParser } from '../../parsers/types.js'; +import { isRecord, parseJsonlRecords } from '../../json.js'; +import { + normalizeToolName, + type AgentToolMap, +} from '../../parsers/shared/normalize.js'; +import { + extractArgs, + extractLoadedSkillsFromText, + type ArgFieldMap, +} from '../../parsers/shared/extract.js'; + +/** Cursor stream-json tool keys → canonical names. */ +const CURSOR_TOOLS: AgentToolMap = { + tools: { + readToolCall: 'file_read', + writeToolCall: 'file_write', + shellToolCall: 'shell', + // Live MCP probe (M0) + mcpToolCall: 'tool_use', + getMcpToolsToolCall: 'tool_use', + // Harbor / older shapes + write_file: 'file_write', + mark_done: 'unknown', + }, +}; + +const CURSOR_ARG_FIELDS: ArgFieldMap = { + path: ['path', 'file_path'], + command: ['command'], + url: ['url'], +}; + +/** Non-tool keys that appear beside tool payloads on completed tool_call events. */ +const TOOL_CALL_META = new Set([ + 'hookAdditionalContexts', + 'toolCallId', + 'startedAtMs', + 'completedAtMs', +]); + +function timestampFromRecord( + data: Record +): string | undefined { + if (typeof data.timestamp === 'string') return data.timestamp; + if (typeof data.timestamp_ms === 'number') { + return new Date(data.timestamp_ms).toISOString(); + } + return undefined; +} + +function messageText(data: Record): string | undefined { + const message = data.message; + if (!isRecord(message)) return undefined; + if (typeof message.content === 'string') return message.content; + if (!Array.isArray(message.content)) return undefined; + const texts = message.content + .filter( + (b): b is Record => isRecord(b) && b.type === 'text' + ) + .map((b) => (typeof b.text === 'string' ? b.text : '')) + .filter(Boolean); + return texts.length > 0 ? texts.join('\n') : undefined; +} + +function enrich(event: TranscriptEvent): TranscriptEvent { + if (event.type !== 'tool_call' || !event.tool) return event; + const { path, command, url } = extractArgs( + event.tool.args ?? {}, + CURSOR_ARG_FIELDS + ); + if (path) event.tool.path = path; + if (command) event.tool.command = command; + if (url) event.tool.url = url; + const loaded: string[] = []; + if (path) loaded.push(...extractLoadedSkillsFromText(path)); + if (command) loaded.push(...extractLoadedSkillsFromText(command)); + if (loaded.length > 0) event.tool.loadedSkills = loaded; + return event; +} + +function originalNameFor( + toolKey: string, + payload: Record +): string { + if (toolKey === 'mcpToolCall') { + const args = isRecord(payload.args) ? payload.args : {}; + if (typeof args.toolName === 'string') return args.toolName; + if (typeof args.name === 'string') return args.name; + } + return toolKey; +} + +function toolArgs(payload: Record): Record { + if (isRecord(payload.args)) return payload.args; + return {}; +} + +/** + * Normalize Cursor's varied result payloads into a value + success flag. + * Built-ins often use `{ success: {...} }` / `{ error: {...} }`; MCP may nest + * `{ success: { content, isError } }`. + */ +function normalizeToolResult(result: unknown): { + result: unknown; + success?: boolean; +} { + if (result === undefined) return { result: undefined, success: true }; + if (!isRecord(result)) return { result, success: true }; + + if ('success' in result) { + const successBody = result.success; + if (isRecord(successBody) && typeof successBody.isError === 'boolean') { + return { result: successBody, success: !successBody.isError }; + } + return { result: successBody, success: true }; + } + if ('error' in result) { + return { result: result.error, success: false }; + } + if (typeof result.isError === 'boolean') { + return { result, success: !result.isError }; + } + return { result, success: true }; +} + +function toolPayloads( + toolCall: Record +): Array<{ key: string; payload: Record }> { + const out: Array<{ key: string; payload: Record }> = []; + for (const [key, value] of Object.entries(toolCall)) { + if (TOOL_CALL_META.has(key)) continue; + if (!isRecord(value)) continue; + if (!('args' in value) && !('result' in value)) continue; + out.push({ key, payload: value }); + } + return out; +} + +function completedToolEvents(data: Record): TranscriptEvent[] { + const toolCall = data.tool_call; + if (!isRecord(toolCall)) return []; + + const timestamp = timestampFromRecord(data); + const callId = typeof data.call_id === 'string' ? data.call_id : undefined; + const events: TranscriptEvent[] = []; + + for (const { key, payload } of toolPayloads(toolCall)) { + const originalName = originalNameFor(key, payload); + const args = toolArgs(payload); + const id = + callId ?? + (typeof payload.toolCallId === 'string' ? payload.toolCallId : undefined); + + events.push( + enrich({ + timestamp, + type: 'tool_call', + tool: { + name: normalizeToolName(key, CURSOR_TOOLS), + originalName, + id, + args, + }, + raw: data, + }) + ); + + if ('result' in payload) { + const normalized = normalizeToolResult(payload.result); + events.push({ + timestamp, + type: 'tool_result', + tool: { + name: normalizeToolName(key, CURSOR_TOOLS), + originalName, + id, + result: normalized.result, + success: normalized.success, + }, + raw: data, + }); + } + } + return events; +} + +export const cursorParser: AgentTranscriptParser = { + parseTranscript(raw: string): ParsedTranscript { + const { records, errors } = parseJsonlRecords(raw); + const events: TranscriptEvent[] = []; + let thinkingBuf: string[] = []; + let lastAssistantText: string | undefined; + + const flushThinking = (): void => { + const content = thinkingBuf.join(''); + thinkingBuf = []; + if (!content.trim()) return; + events.push({ type: 'thinking', content }); + }; + + for (const record of records) { + try { + const type = record.type; + + if (type === 'thinking') { + if (record.subtype === 'delta' && typeof record.text === 'string') { + thinkingBuf.push(record.text); + } else if (record.subtype === 'completed') { + flushThinking(); + } + continue; + } + + // Any non-thinking event ends the current thinking span. + flushThinking(); + + if (type === 'system' || type === 'interaction_query') { + continue; + } + + if (type === 'user') { + const content = messageText(record); + if (content) { + events.push({ + timestamp: timestampFromRecord(record), + type: 'message', + role: 'user', + content, + raw: record, + }); + } + continue; + } + + if (type === 'assistant') { + const content = messageText(record); + if (content) { + events.push({ + timestamp: timestampFromRecord(record), + type: 'message', + role: 'assistant', + content, + raw: record, + }); + lastAssistantText = content; + } + continue; + } + + if (type === 'tool_call') { + if (record.subtype === 'started') continue; + if (record.subtype === 'completed') { + events.push(...completedToolEvents(record)); + } + continue; + } + + if (type === 'result') { + const resultText = + typeof record.result === 'string' ? record.result : undefined; + if (record.is_error === true) { + events.push({ + timestamp: timestampFromRecord(record), + type: 'error', + content: resultText ?? 'cursor agent error', + raw: record, + }); + } else if (resultText && resultText !== lastAssistantText) { + events.push({ + timestamp: timestampFromRecord(record), + type: 'message', + role: 'assistant', + content: resultText, + raw: record, + }); + lastAssistantText = resultText; + } + continue; + } + } catch (e) { + errors.push(e instanceof Error ? e.message : String(e)); + } + } + + flushThinking(); + return { events, errors }; + }, +}; diff --git a/packages/core/src/agents/cursor/runner.test.ts b/packages/core/src/agents/cursor/runner.test.ts new file mode 100644 index 00000000..ea6026d6 --- /dev/null +++ b/packages/core/src/agents/cursor/runner.test.ts @@ -0,0 +1,202 @@ +import { describe, expect, it } from 'vitest'; +import type { CommandResult } from '../../index.js'; +import { + buildCursorMcpConfig, + cursorModelArg, + cursorRunner, +} from './runner.js'; + +const ok: CommandResult = { ok: true, exitCode: 0, stdout: '', stderr: '' }; +const timedOut: CommandResult = { + ok: false, + exitCode: 124, + stdout: '', + stderr: '[command timed out after 540s and was terminated]', +}; +const failed: CommandResult = { + ok: false, + exitCode: 1, + stdout: '', + stderr: 'boom', +}; + +function streamJson(subtype: string, isError = false): string { + return [ + JSON.stringify({ + type: 'system', + subtype: 'init', + session_id: 's1', + model: 'composer-2.5', + }), + JSON.stringify({ + type: 'result', + subtype, + is_error: isError, + result: 'all done', + session_id: 's1', + }), + ].join('\n'); +} + +describe('cursorModelArg', () => { + it('returns the bare model when effort is omitted', () => { + expect(cursorModelArg('composer-2.5')).toBe('composer-2.5'); + }); + + it('applies Cursor effort bracket syntax', () => { + expect(cursorModelArg('composer-2.5', 'high')).toBe( + 'composer-2.5[effort=high]' + ); + }); + + it('rejects combining effort with an already-parameterized model', () => { + expect(() => + cursorModelArg('composer-2.5[fast=false]', 'high') + ).toThrowError(/already has bracket params/); + }); +}); + +describe('buildCursorMcpConfig', () => { + it("builds Cursor's ~/.cursor/mcp.json stdio shape", () => { + const config = JSON.parse( + buildCursorMcpConfig({ + supabase: { + command: 'npx', + args: ['-y', 'srv'], + env: { TOKEN: 't' }, + }, + docs: { command: 'docs-server' }, + }) + ); + expect(config.mcpServers).toEqual({ + supabase: { + command: 'npx', + args: ['-y', 'srv'], + env: { TOKEN: 't' }, + }, + docs: { command: 'docs-server' }, + }); + }); +}); + +describe('cursorRunner.deriveStopReason', () => { + const derive = cursorRunner.deriveStopReason!; + + it('maps a successful result event to a normal stop', () => { + expect(derive(streamJson('success'), ok)).toBe('stop'); + }); + + it('surfaces non-success result subtypes verbatim', () => { + expect(derive(streamJson('error_max_turns', true), ok)).toBe( + 'error_max_turns' + ); + }); + + it('falls back to the process result when there is no result event', () => { + expect(derive(undefined, timedOut)).toBe('timeout'); + expect(derive('not json\n', failed)).toBe('error_exit_1'); + expect(derive(undefined, ok)).toBe('stop'); + }); +}); + +type CapturedExec = { + commands: string[]; + runCommand: string; + runEnv: Record | undefined; + mcpConfig: Record | undefined; +}; + +async function captureExec(opts: { + mcp?: boolean; + reasoningEffort?: string; + skillsPresent?: boolean; +}): Promise { + const commands: string[] = []; + let runCommand = ''; + let runEnv: Record | undefined; + let mcpConfig: Record | undefined; + + await cursorRunner.exec({ + sandbox: { + workspace: '/workspace', + exec: async (cmd, options) => { + commands.push(cmd); + const write = /^printf %s '([^']+)'/.exec(cmd); + if (write) { + const decoded = Buffer.from(write[1], 'base64').toString('utf8'); + if (decoded.includes('mcpServers')) { + mcpConfig = JSON.parse(decoded); + } + } else if (cmd.includes('cursor-agent') && cmd.includes('--print')) { + runCommand = cmd; + runEnv = options?.env; + } else if (opts.skillsPresent && cmd.includes('.claude/skills')) { + // pretend the skills dir exists by no-op success + } + return ok; + }, + readFile: async () => '', + }, + model: 'composer-2.5', + apiKey: 'cursor-key', + systemPromptPath: '"$HOME/.eval/system-prompt.txt"', + userPromptPath: '"$HOME/.eval/user-prompt.txt"', + mcpServers: opts.mcp + ? { supabase: { command: 'npx', args: ['-y', '@supabase/mcp'] } } + : {}, + reasoningEffort: opts.reasoningEffort, + timeoutSec: 1, + }); + + return { commands, runCommand, runEnv, mcpConfig }; +} + +describe('cursorRunner.exec', () => { + it('runs headless with locked flags, workspace, model, and API key', async () => { + const { runCommand, runEnv } = await captureExec({}); + expect(runCommand).toContain('--print'); + expect(runCommand).toContain('--output-format=stream-json'); + expect(runCommand).toContain('--yolo'); + expect(runCommand).toContain('--trust'); + expect(runCommand).toContain('--approve-mcps'); + expect(runCommand).toContain("--model='composer-2.5'"); + expect(runCommand).toContain("--workspace='/workspace'"); + expect(runCommand).toContain('cursor-prompt.txt'); + expect(runEnv).toEqual({ CURSOR_API_KEY: 'cursor-key' }); + }); + + it('writes MCP config when servers are present', async () => { + const { mcpConfig, commands } = await captureExec({ mcp: true }); + expect(mcpConfig?.mcpServers).toHaveProperty('supabase'); + expect(commands.some((c) => c.includes('.cursor/mcp.json'))).toBe(true); + }); + + it('does not write MCP config when no servers are configured', async () => { + const { mcpConfig } = await captureExec({ mcp: false }); + expect(mcpConfig).toBeUndefined(); + }); + + it('applies reasoning effort as a model bracket param', async () => { + const { runCommand } = await captureExec({ reasoningEffort: 'high' }); + expect(runCommand).toContain("--model='composer-2.5[effort=high]'"); + }); + + it('always attempts the .claude/skills → ~/.cursor/skills bridge', async () => { + const { commands } = await captureExec({}); + expect( + commands.some( + (c) => c.includes('.claude/skills') && c.includes('.cursor/skills') + ) + ).toBe(true); + }); +}); + +describe('cursorRunner metadata', () => { + it('advertises cursor provider and pinned defaults', () => { + expect(cursorRunner.id).toBe('cursor'); + expect(cursorRunner.modelProvider).toBe('cursor'); + expect(cursorRunner.apiKeyEnvVar).toBe('CURSOR_API_KEY'); + expect(cursorRunner.defaultModel).toBe('composer-2.5'); + expect(cursorRunner.defaultCliVersion).toBe('2026.07.23-e383d2b'); + }); +}); diff --git a/packages/core/src/agents/cursor/runner.ts b/packages/core/src/agents/cursor/runner.ts new file mode 100644 index 00000000..d2a4b4de --- /dev/null +++ b/packages/core/src/agents/cursor/runner.ts @@ -0,0 +1,180 @@ +/** + * Cursor runner. Headless via `cursor-agent --print --output-format stream-json` + * (same programmatic path Harbor uses; see M0 discovery). Install is a pinned + * official tarball — not npm. See ./parser.ts (M2) for the transcript. + */ + +import type { McpServerConfig } from '../../index.js'; +import { parseJsonlRecords } from '../../json.js'; +import type { AgentRunner } from '../types.js'; +import { + SCRATCH, + processStopReason, + shellQuote, + writeSandboxFile, +} from '../shared.js'; + +/** Cursor CLI model id from `cursor-agent --list-models` (e.g. composer-2.5). */ +export type CursorModel = string; + +/** Pinned CLI build — bump deliberately and re-check the parser. */ +export const DEFAULT_CURSOR_CLI_VERSION = '2026.07.23-e383d2b'; + +export const DEFAULT_CURSOR_MODEL: CursorModel = 'composer-2.5'; + +const CURSOR_BIN = '"$HOME/.local/bin/cursor-agent"'; +const MCP_CONFIG_PATH = '"$HOME/.cursor/mcp.json"'; +const PROMPT_PATH = '"$HOME/.eval/cursor-prompt.txt"'; +const SKILLS_SOURCE = '.claude/skills'; +const SKILLS_DEST = '"$HOME/.cursor/skills"'; + +/** + * Harbor / Cursor MCP config: `{ mcpServers: { name: { command, args?, env? } } }`. + * Supabase harness only supplies stdio MCP servers today. + */ +export function buildCursorMcpConfig( + servers: Record +): string { + const mcpServers: Record = {}; + for (const [name, server] of Object.entries(servers)) { + mcpServers[name] = { + command: server.command, + ...(server.args ? { args: server.args } : {}), + ...(server.env ? { env: server.env } : {}), + }; + } + return JSON.stringify({ mcpServers }, null, 2); +} + +/** Apply Cursor's `model[effort=…]` syntax when the experiment pins effort. */ +export function cursorModelArg( + model: string, + reasoningEffort?: string +): string { + if (!reasoningEffort) return model; + if (model.includes('[')) { + throw new Error( + `Model '${model}' already has bracket params; omit reasoningEffort ` + + 'or remove them from the model id' + ); + } + return `${model}[effort=${reasoningEffort}]`; +} + +function tarballUrl(version: string): string { + return `https://downloads.cursor.com/lab/${version}/linux/x64/agent-cli-package.tar.gz`; +} + +export const cursorRunner: AgentRunner = { + id: 'cursor', + displayName: 'Cursor', + apiKeyEnvVar: 'CURSOR_API_KEY', + modelProvider: 'cursor', + // Not an npm package — install() fetches the official tarball. Kept so the + // AgentRunner interface stays uniform with peer CLI agents. + cliPackage: 'cursor-agent', + defaultCliVersion: DEFAULT_CURSOR_CLI_VERSION, + defaultModel: DEFAULT_CURSOR_MODEL, + + async install(sandbox, version) { + const url = tarballUrl(version); + const versionDir = `"$HOME/.local/share/cursor-agent/versions/${version}"`; + const result = await sandbox.exec( + [ + 'set -euo pipefail', + `mkdir -p ${versionDir} "$HOME/.local/bin"`, + `curl -fsSL ${shellQuote(url)} -o /tmp/cursor-agent-${version}.tar.gz`, + // Archive root is dist-package/; strip so cursor-agent lands in versionDir. + `tar -xzf /tmp/cursor-agent-${version}.tar.gz -C ${versionDir} --strip-components=1`, + `ln -sfn ${versionDir}/cursor-agent "$HOME/.local/bin/cursor-agent"`, + `ln -sfn "$HOME/.local/bin/cursor-agent" "$HOME/.local/bin/agent"`, + 'export PATH="$HOME/.local/bin:$PATH"', + 'cursor-agent --version', + ].join(' && '), + { timeoutMs: 300_000 } + ); + if (!result.ok) { + throw new Error( + `Cursor CLI install failed: ${result.stderr || result.stdout}` + ); + } + }, + + async exec({ + sandbox, + model, + apiKey, + systemPromptPath, + userPromptPath, + mcpServers, + reasoningEffort, + timeoutSec, + }) { + await sandbox.exec(`mkdir -p ${SCRATCH} "$HOME/.cursor"`); + + // Bridge supabase's .claude/skills install into Cursor's skill roots. + await sandbox.exec( + `if [ -d ${shellQuote(SKILLS_SOURCE)} ]; then ` + + `mkdir -p ${SKILLS_DEST} && ` + + `cp -a ${shellQuote(SKILLS_SOURCE)}/. ${SKILLS_DEST}/; ` + + 'fi' + ); + + if (Object.keys(mcpServers).length > 0) { + await writeSandboxFile( + sandbox, + MCP_CONFIG_PATH, + buildCursorMcpConfig(mcpServers) + ); + } + + // Cursor has no append-system-prompt-file; stage one combined prompt (Codex style). + const stagePrompt = await sandbox.exec( + `{ cat ${systemPromptPath}; printf '\\n\\n'; cat ${userPromptPath}; } > ${PROMPT_PATH}` + ); + if (!stagePrompt.ok) { + throw new Error( + `failed to stage Cursor prompt: ${stagePrompt.stderr || stagePrompt.stdout}` + ); + } + + const modelFlag = shellQuote(cursorModelArg(model, reasoningEffort)); + const workspace = shellQuote(sandbox.workspace); + const flags = [ + '--print', + '--output-format=stream-json', + '--yolo', + '--trust', + '--approve-mcps', + `--model=${modelFlag}`, + `--workspace=${workspace}`, + ].join(' '); + + const command = await sandbox.exec( + `export PATH="$HOME/.local/bin:$PATH"; ` + + `${CURSOR_BIN} ${flags} -- "$(cat ${PROMPT_PATH})"`, + { + timeoutMs: timeoutSec * 1000, + env: { CURSOR_API_KEY: apiKey }, + } + ); + return { command, raw: command.stdout }; + }, + + deriveStopReason(raw, command) { + const result = lastResultEvent(raw); + const subtype = + typeof result?.subtype === 'string' ? result.subtype : undefined; + if (subtype === 'success') return 'stop'; + if (subtype) return subtype; + return processStopReason(command); + }, +}; + +function lastResultEvent( + raw: string | undefined +): Record | undefined { + if (!raw) return undefined; + const { records } = parseJsonlRecords(raw); + return [...records].reverse().find((r) => r.type === 'result'); +} diff --git a/packages/core/src/agents/engine.ts b/packages/core/src/agents/engine.ts index de61c1fa..771d363e 100644 --- a/packages/core/src/agents/engine.ts +++ b/packages/core/src/agents/engine.ts @@ -40,6 +40,8 @@ function modelProviderForAgent(id: AgentRunner['id']): ModelProvider { return 'anthropic'; case 'codex': return 'openai'; + case 'cursor': + return 'cursor'; case 'opencode': throw new Error( 'opencode is multi-provider; its runner sets `modelProvider` from the model id' diff --git a/packages/core/src/agents/registry.ts b/packages/core/src/agents/registry.ts index 58db31b8..3409f482 100644 --- a/packages/core/src/agents/registry.ts +++ b/packages/core/src/agents/registry.ts @@ -13,11 +13,13 @@ import type { AgentHarnessId } from '../eval-metadata.js'; import type { AgentTranscriptParser } from '../parsers/types.js'; import { claudeCodeDefinition } from './claude-code/index.js'; import { codexDefinition } from './codex/index.js'; +import { cursorDefinition } from './cursor/index.js'; import { opencodeDefinition } from './opencode/index.js'; const AGENTS: AgentDefinition[] = [ claudeCodeDefinition, codexDefinition, + cursorDefinition, opencodeDefinition, ]; diff --git a/packages/core/src/eval-metadata.ts b/packages/core/src/eval-metadata.ts index 66c3f8bf..07f28914 100644 --- a/packages/core/src/eval-metadata.ts +++ b/packages/core/src/eval-metadata.ts @@ -56,6 +56,7 @@ export const agentHarnessIdSchema = z.enum([ 'ai-sdk', 'claude-code', 'codex', + 'cursor', 'opencode', ]); export type AgentHarnessId = z.infer; @@ -64,6 +65,7 @@ export const modelProviderSchema = z.enum([ 'anthropic', 'openai', 'moonshotai', + 'cursor', ]); export type ModelProvider = z.infer; diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 90a424d9..9414a4a2 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -100,10 +100,11 @@ export { rehydrateTruncatedDocsResults, } from './docs-results.js'; export type { DocsResultSandbox } from './docs-results.js'; -// CLI agent harnesses (Claude Code, Codex, OpenCode, and the framework for adding more). +// CLI agent harnesses (Claude Code, Codex, Cursor, OpenCode, and the framework for adding more). export { createCliAgent } from './agents/engine.js'; export { claudeCodeAgent } from './agents/claude-code/index.js'; export { codexAgent } from './agents/codex/index.js'; +export { cursorAgent } from './agents/cursor/index.js'; export { opencodeAgent } from './agents/opencode/index.js'; export type { AgentMetadata,