diff --git a/bun.lock b/bun.lock index 0fb063997..3dfe8a040 100644 --- a/bun.lock +++ b/bun.lock @@ -7,11 +7,7 @@ "devDependencies": { "@biomejs/biome": "2.5.7", "@types/node": "^25.5.0", - "@types/react": "19.2.10", - "@types/react-dom": "19.2.3", "knip": "^5.80.0", - "react": "19.2.4", - "react-dom": "19.2.4", "typescript": "^5.9.2", }, }, @@ -103,7 +99,6 @@ "@radix-ui/react-popover": "^1.1.15", "@radix-ui/react-scroll-area": "^1.2.10", "@radix-ui/react-slot": "^1.0.2", - "@radix-ui/react-tabs": "^1.1.13", "@radix-ui/react-tooltip": "^1.2.8", "@xterm/addon-fit": "^0.11.0", "@xterm/addon-web-links": "^0.12.0", @@ -118,7 +113,6 @@ "remark-gfm": "^4.0.1", "tailwind-merge": "^2.2.0", "tailwindcss-animate": "^1.0.7", - "typebox": "1.3.7", }, "devDependencies": { "@biomejs/biome": "2.5.7", @@ -486,8 +480,6 @@ "@radix-ui/react-arrow": ["@radix-ui/react-arrow@1.1.7", "", { "dependencies": { "@radix-ui/react-primitive": "2.1.3" }, "optionalDependencies": { "@types/react": "19.2.10", "@types/react-dom": "19.2.3" }, "peerDependencies": { "react": "19.2.4", "react-dom": "19.2.4" } }, "sha512-F+M1tLhO+mlQaOWspE8Wstg+z6PwxwRd8oQ8IXceWz92kfAmalTRf0EjrouQeo7QssEPfCn05B4Ihs1K9WQ/7w=="], - "@radix-ui/react-collection": ["@radix-ui/react-collection@1.1.7", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.2", "@radix-ui/react-context": "1.1.2", "@radix-ui/react-primitive": "2.1.3", "@radix-ui/react-slot": "1.2.3" }, "optionalDependencies": { "@types/react": "19.2.10", "@types/react-dom": "19.2.3" }, "peerDependencies": { "react": "19.2.4", "react-dom": "19.2.4" } }, "sha512-Fh9rGN0MoI4ZFUNyfFVNU4y9LUz93u9/0K+yLgA2bwRojxM8JU1DyvvMBabnZPBgMWREAJvU2jjVzq+LrFUglw=="], - "@radix-ui/react-compose-refs": ["@radix-ui/react-compose-refs@1.1.2", "", { "optionalDependencies": { "@types/react": "19.2.10" }, "peerDependencies": { "react": "19.2.4" } }, "sha512-z4eqJvfiNnFMHIIvXP3CY57y2WJs5g2v3X0zm9mEJkrkNv4rDxu+sg9Jh8EkXyeqBkB7SOcboo9dMVqhyrACIg=="], "@radix-ui/react-context": ["@radix-ui/react-context@1.1.2", "", { "optionalDependencies": { "@types/react": "19.2.10" }, "peerDependencies": { "react": "19.2.4" } }, "sha512-jCi/QKUM2r1Ju5a3J64TH2A5SpKAgh0LpknyqdQ4m6DCV0xJ2HG1xARRwNGPQfi1SLdLWZ1OJz6F4OMBBNiGJA=="], @@ -514,14 +506,10 @@ "@radix-ui/react-primitive": ["@radix-ui/react-primitive@2.1.3", "", { "dependencies": { "@radix-ui/react-slot": "1.2.3" }, "optionalDependencies": { "@types/react": "19.2.10", "@types/react-dom": "19.2.3" }, "peerDependencies": { "react": "19.2.4", "react-dom": "19.2.4" } }, "sha512-m9gTwRkhy2lvCPe6QJp4d3G1TYEUHn/FzJUtq9MjH46an1wJU+GdoGC5VLof8RX8Ft/DlpshApkhswDLZzHIcQ=="], - "@radix-ui/react-roving-focus": ["@radix-ui/react-roving-focus@1.1.11", "", { "dependencies": { "@radix-ui/primitive": "1.1.3", "@radix-ui/react-collection": "1.1.7", "@radix-ui/react-compose-refs": "1.1.2", "@radix-ui/react-context": "1.1.2", "@radix-ui/react-direction": "1.1.1", "@radix-ui/react-id": "1.1.1", "@radix-ui/react-primitive": "2.1.3", "@radix-ui/react-use-callback-ref": "1.1.1", "@radix-ui/react-use-controllable-state": "1.2.2" }, "optionalDependencies": { "@types/react": "19.2.10", "@types/react-dom": "19.2.3" }, "peerDependencies": { "react": "19.2.4", "react-dom": "19.2.4" } }, "sha512-7A6S9jSgm/S+7MdtNDSb+IU859vQqJ/QAtcYQcfFC6W8RS4IxIZDldLR0xqCFZ6DCyrQLjLPsxtTNch5jVA4lA=="], - "@radix-ui/react-scroll-area": ["@radix-ui/react-scroll-area@1.2.10", "", { "dependencies": { "@radix-ui/number": "1.1.1", "@radix-ui/primitive": "1.1.3", "@radix-ui/react-compose-refs": "1.1.2", "@radix-ui/react-context": "1.1.2", "@radix-ui/react-direction": "1.1.1", "@radix-ui/react-presence": "1.1.5", "@radix-ui/react-primitive": "2.1.3", "@radix-ui/react-use-callback-ref": "1.1.1", "@radix-ui/react-use-layout-effect": "1.1.1" }, "optionalDependencies": { "@types/react": "19.2.10", "@types/react-dom": "19.2.3" }, "peerDependencies": { "react": "19.2.4", "react-dom": "19.2.4" } }, "sha512-tAXIa1g3sM5CGpVT0uIbUx/U3Gs5N8T52IICuCtObaos1S8fzsrPXG5WObkQN3S6NVl6wKgPhAIiBGbWnvc97A=="], "@radix-ui/react-slot": ["@radix-ui/react-slot@1.2.4", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.2" }, "optionalDependencies": { "@types/react": "19.2.10" }, "peerDependencies": { "react": "19.2.4" } }, "sha512-Jl+bCv8HxKnlTLVrcDE8zTMJ09R9/ukw4qBs/oZClOfoQk/cOTbDn+NceXfV7j09YPVQUryJPHurafcSg6EVKA=="], - "@radix-ui/react-tabs": ["@radix-ui/react-tabs@1.1.13", "", { "dependencies": { "@radix-ui/primitive": "1.1.3", "@radix-ui/react-context": "1.1.2", "@radix-ui/react-direction": "1.1.1", "@radix-ui/react-id": "1.1.1", "@radix-ui/react-presence": "1.1.5", "@radix-ui/react-primitive": "2.1.3", "@radix-ui/react-roving-focus": "1.1.11", "@radix-ui/react-use-controllable-state": "1.2.2" }, "optionalDependencies": { "@types/react": "19.2.10", "@types/react-dom": "19.2.3" }, "peerDependencies": { "react": "19.2.4", "react-dom": "19.2.4" } }, "sha512-7xdcatg7/U+7+Udyoj2zodtI9H/IIopqo+YOIcZOq1nJwXWBZ9p8xiu5llXlekDbZkca79a/fozEYQXIA4sW6A=="], - "@radix-ui/react-tooltip": ["@radix-ui/react-tooltip@1.2.8", "", { "dependencies": { "@radix-ui/primitive": "1.1.3", "@radix-ui/react-compose-refs": "1.1.2", "@radix-ui/react-context": "1.1.2", "@radix-ui/react-dismissable-layer": "1.1.11", "@radix-ui/react-id": "1.1.1", "@radix-ui/react-popper": "1.2.8", "@radix-ui/react-portal": "1.1.9", "@radix-ui/react-presence": "1.1.5", "@radix-ui/react-primitive": "2.1.3", "@radix-ui/react-slot": "1.2.3", "@radix-ui/react-use-controllable-state": "1.2.2", "@radix-ui/react-visually-hidden": "1.2.3" }, "optionalDependencies": { "@types/react": "19.2.10", "@types/react-dom": "19.2.3" }, "peerDependencies": { "react": "19.2.4", "react-dom": "19.2.4" } }, "sha512-tY7sVt1yL9ozIxvmbtN5qtmH2krXcBCfjEiCgKGLqunJHvgvZG2Pcl2oQ3kbcZARb1BGEHdkLzcYGO8ynVlieg=="], "@radix-ui/react-use-callback-ref": ["@radix-ui/react-use-callback-ref@1.1.1", "", { "optionalDependencies": { "@types/react": "19.2.10" }, "peerDependencies": { "react": "19.2.4" } }, "sha512-FkBMwD+qbGQeMu1cOHnuGB6x4yzPjho8ap5WtbEJ26umhgqVXbhekKUQO+hZEL1vU92a3wHwdp0HAcqAUF5iDg=="], @@ -2190,8 +2178,6 @@ "@isaacs/cliui/wrap-ansi": ["wrap-ansi@8.1.0", "", { "dependencies": { "ansi-styles": "6.2.3", "string-width": "5.1.2", "strip-ansi": "7.1.2" } }, "sha512-si7QWI6zUMq56bESFvagtmzMdGOtoxfR+Sez11Mobfc7tm+VkUckk9bW2UeffTGVUbOksxmSw0AA2gs8g71NCQ=="], - "@radix-ui/react-collection/@radix-ui/react-slot": ["@radix-ui/react-slot@1.2.3", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.2" }, "optionalDependencies": { "@types/react": "19.2.10" }, "peerDependencies": { "react": "19.2.4" } }, "sha512-aeNmHnBxbi2St0au6VBVC7JXFlhLlOnvIIlePNniyUNAClzmtAUEY8/pBiK3iHjufOlwA+c20/8jngo7xcrg8A=="], - "@radix-ui/react-dialog/@radix-ui/react-slot": ["@radix-ui/react-slot@1.2.3", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.2" }, "optionalDependencies": { "@types/react": "19.2.10" }, "peerDependencies": { "react": "19.2.4" } }, "sha512-aeNmHnBxbi2St0au6VBVC7JXFlhLlOnvIIlePNniyUNAClzmtAUEY8/pBiK3iHjufOlwA+c20/8jngo7xcrg8A=="], "@radix-ui/react-popover/@radix-ui/react-slot": ["@radix-ui/react-slot@1.2.3", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.2" }, "optionalDependencies": { "@types/react": "19.2.10" }, "peerDependencies": { "react": "19.2.4" } }, "sha512-aeNmHnBxbi2St0au6VBVC7JXFlhLlOnvIIlePNniyUNAClzmtAUEY8/pBiK3iHjufOlwA+c20/8jngo7xcrg8A=="], @@ -2246,8 +2232,6 @@ "blade-vscode/@types/node": ["@types/node@22.19.3", "", { "dependencies": { "undici-types": "6.21.0" } }, "sha512-1N9SBnWYOJTrNZCdh/yJE+t910Y128BoyY+zBLWhL3r0TYzlTmFdXrPwHL9DyFZmlEXNQQolTZh3KHV31QDhyA=="], - "blade-web/typebox": ["typebox@1.3.7", "", {}, "sha512-meKuifc33Pccx0O6PdIzYMq3Og8zvP4TIi/a+Bw3AEMZMxOD0+RHGQvpglEe6Zdy3wZ8nqn/j95h8LUZLk/6Hg=="], - "blade-web/vitest": ["vitest@3.2.7", "", { "dependencies": { "@types/chai": "^5.2.2", "@vitest/expect": "3.2.7", "@vitest/mocker": "3.2.7", "@vitest/pretty-format": "^3.2.7", "@vitest/runner": "3.2.7", "@vitest/snapshot": "3.2.7", "@vitest/spy": "3.2.7", "@vitest/utils": "3.2.7", "chai": "^5.2.0", "debug": "^4.4.1", "expect-type": "^1.2.1", "magic-string": "^0.30.17", "pathe": "^2.0.3", "picomatch": "^4.0.2", "std-env": "^3.9.0", "tinybench": "^2.9.0", "tinyexec": "^0.3.2", "tinyglobby": "^0.2.14", "tinypool": "^1.1.1", "tinyrainbow": "^2.0.0", "vite": "^5.0.0 || ^6.0.0 || ^7.0.0-0", "vite-node": "3.2.4", "why-is-node-running": "^2.3.0" }, "peerDependencies": { "@edge-runtime/vm": "*", "@types/debug": "^4.1.12", "@types/node": "^18.0.0 || ^20.0.0 || >=22.0.0", "@vitest/browser": "3.2.7", "@vitest/ui": "3.2.7", "happy-dom": "*", "jsdom": "*" }, "optionalPeers": ["@edge-runtime/vm", "@types/debug", "@types/node", "@vitest/browser", "@vitest/ui", "happy-dom", "jsdom"], "bin": { "vitest": "./vitest.mjs" } }, "sha512-KrxIJ62Fd89gfysR4WotlgZABiz2dqFPgqGzX7s+CwsqLFomRH7777ZcrOD6+WVAh7khPQP41A+BKbpcJFrdEg=="], "body-parser/iconv-lite": ["iconv-lite@0.7.1", "", { "dependencies": { "safer-buffer": "2.1.2" } }, "sha512-2Tth85cXwGFHfvRgZWszZSvdo+0Xsqmw8k8ZwxScfcBneNUraK+dxRxRm24nszx80Y0TVio8kKLt5sLE7ZCLlw=="], diff --git a/docs/design/codebase-simplification.md b/docs/design/codebase-simplification.md new file mode 100644 index 000000000..a4549e7a3 --- /dev/null +++ b/docs/design/codebase-simplification.md @@ -0,0 +1,105 @@ +# Codebase Simplification + +## Objective + +Reduce tracked TypeScript/TSX/JavaScript LOC by at least 30% while preserving +the supported runtime surfaces and making ownership easier to follow. + +Baseline at `8bcd2696`: + +- Tracked TypeScript/TSX/JavaScript LOC: 593,930 +- Target maximum: 415,751 LOC +- Required net reduction: 178,179 LOC + +The metric excludes generated output, dependencies, Markdown, JSON, and +lockfiles. It compares the same tracked extensions at the baseline and +candidate commits. + +## Guardrails + +- Preserve CLI, Web, ACP, MCP, browser, LSP, task, team, goal, and provider + behavior unless a surface is unused or explicitly deprecated. +- Keep deterministic state-machine tests for durable storage, permissions, + compaction, scheduling, tool execution, and protocol boundaries. +- Keep a focused paid qualification matrix for behavior that cannot be proven + without a real Provider or production surface. +- Keep one authoritative state machine per behavior. Surfaces project state; + they do not reimplement it. +- Remove deprecated compatibility paths instead of retaining aliases. +- Prefer data-driven routing and named boundaries over nested conditionals. +- Keep each independently verifiable reduction in its own commit. +- Do not use git worktrees. + +## Implemented Changes + +### Runtime + +- Removed unused subsystems, compatibility APIs, global registries, and + management surfaces. +- Unified Session metadata updates, local/remote fork projection, event + projection, compaction telemetry, and inputless resume state. +- Replaced repeated shell and tool policy branches with declarative metadata. +- Split Session run ownership out of the Hono controller. + +### Test Architecture + +- Centralized Provider, ACP, PTY, Web, Agent loop, and Session fixture + lifecycles before deleting their repeated consumers. +- Removed source-string gates and tests whose subject was another test harness. +- Replaced broad cross-layer matrices with focused state-machine tests. +- Reduced paid release qualification to nine high-value production paths. + +## Resulting Boundaries + +### Session Run Lifecycle + +- `server/routes/session.ts` owns HTTP/SSE routing, projection residency, and + controller shutdown. +- `server/routes/sessionRunState.ts` owns active/recent run registration, + cancellation, pending permission lookup, and mutable Session task projection. +- `server/routes/sessionRunExecutor.ts` owns Agent creation, loop event + projection, pending-resume evidence, terminal task state, and resource + release. + +The executor depends on the run-state module. Neither extracted module depends +on the Hono controller. + +### Test Pyramid + +The default deterministic suite remains the primary regression authority. The +production real-API release matrix is limited to: + +1. Production Agent edit and verification +2. Structured output +3. Durable interaction recovery +4. Cross-surface release coding +5. Agent Team task coordination +6. Cross-Provider fallback +7. Goal completion +8. Native Browser tools +9. ACP remote filesystem + +Provider admission, retry, compaction, queueing, Session identity, event +projection, and resource cleanup remain covered by deterministic unit and +integration tests instead of repeated paid surface grids. + +## Result + +Representative commits: + +| Phase | Commits | +| --- | --- | +| Dead code and compatibility | `8b2d4ea0` through `6e0e4aa6` | +| Runtime unification | `d5373627`, `e7007472`, `953e6a67`, `5c0b67ca` | +| Harness consolidation | `b6a20834` through `877db51d` | +| Qualification focus | `89ba9afd` | +| Regression matrix reduction | `5a36aaf7` | +| Session decomposition | `c214641e` | +| Runtime and static data compaction | `7b07d031` | +| Final focused regression coverage | `a609f0a1` | + +At `a609f0a1`, tracked TypeScript/TSX/JavaScript is 414,908 lines: + +- Net reduction: 179,022 lines +- Reduction from baseline: 30.1419% +- Margin beyond the required reduction: 843 lines diff --git a/docs/en/reference/workspace-agent-resources.md b/docs/en/reference/workspace-agent-resources.md index fa38ce6c6..0c6b43cb2 100644 --- a/docs/en/reference/workspace-agent-resources.md +++ b/docs/en/reference/workspace-agent-resources.md @@ -65,8 +65,10 @@ Deterministic tests use two real temporary directories to load native and plugin Real API qualification includes: -- GPT dual `SessionRuntime` concurrently calling respective plugin commands; -- GPT calling respective plugin commands in dual-cwd Sessions within the same ACP connection; +- DeepSeek Flash dual `SessionRuntime` instances concurrently call their respective + plugin commands; +- DeepSeek Flash calls the respective plugin commands in dual-cwd Sessions within the + same ACP connection; - DeepSeek Flash/Pro completing Read/Edit/Bash via production CLI `--agents -> Task`; - Production Web GUI binding and trusting two projects A/B, calling `plugin-a:reveal` and `plugin-b:reveal` respectively in independent worktrees, maintaining independent projections after switching back, and a fresh tab having zero console errors. diff --git a/docs/en/testing/qualification.md b/docs/en/testing/qualification.md index 93a4bda66..fb5895f94 100644 --- a/docs/en/testing/qualification.md +++ b/docs/en/testing/qualification.md @@ -1,559 +1,155 @@ -# Blade Code Testing Methodology and Production Qualification +# Blade Code Testing and Production Qualification -Blade Code separates deterministic regression and paid model verification into two gates. Both gates must pass before a feature patch can be marked production-ready. +Blade Code separates deterministic regression checks from paid model +verification. Both gates must pass before a feature patch is production-ready. ## Controlled Coding Benchmark -`bun run benchmark:repo -- --model ` runs `controlled-coding-v2`: -read-only diagnosis, single-file repair, and cross-module API migration in three small -fixed projects. This is not a large real-repository benchmark; passing DeepSeek -Flash/Pro across these three tasks does not establish overall parity with other coding -agents. Run `bun run build:cli` first. The benchmark uses production dist, Node, npm, -and configured API-key model authentication, without installing dependencies or creating -worktrees. - -Each task has its own temporary project, HOME, and Session storage, with MCP, LSP, -hooks, and plugins disabled. Diagnosis requires a successful file read, exact structured -findings, and unchanged file inventory/content. Repairs require the exact allowed file -changes and a successful `npm test` after the last edit. The host then copies source bytes -to a separate directory, runs boundary examples, and compares returned values. Changed -tests/package.json, added files/directories, symlinks, missing tool evidence, success -claims, and `exit(0)` alone cannot pass. Verification has a five-second budget; each -Agent has 240 seconds, 16 turns, a 4 MiB output limit, and zero additional transport retries. - -Temporary-directory isolation is not an OS security sandbox: candidate code and tools -still run as the current user. Do not use it for hostile code; finite checks do not prove -correctness for every input. Subprocess environments contain only required settings, and -scores retain no credentials, model prose, or raw tool output. Results default to -`.blade/benchmarks/controlled-coding-v2-history.json`, recording source SHA-256, changed -paths, individual checks, cumulative request tokens, successful reads, and Agent time. -Use `--history-path` to change the destination. Legacy v1 keyword scores are neither -overwritten nor merged. Any failed task makes the command exit nonzero. History writes -are not concurrent-safe; use separate paths for parallel runs. - -A separate cross-surface migration qualification covers DeepSeek Flash/Pro across -production Chromium, Vite development Chromium, raw PTY TUI, and ACP (eight cells). -Each cell must modify both source files, preserve tests and package.json, and execute -exactly `npm test` after the edit results commit. Host verification uses durable -result-commit order, not invocation order, and checks copied source in a separate -verification directory. GUI assertions match each visible tool ID; ACP terminal updates -must match durable tool IDs; TUI waits for exact durable final output before exiting. -These controlled integration checks do not establish large-repository success rates or -native desktop Computer Use coverage. +`bun run benchmark:repo -- --model ` runs +`controlled-coding-v2`: read-only diagnosis, single-file repair, and a +cross-module API migration in fixed small projects. It uses the production +distribution, Node, npm, and configured API-key authentication. It does not +install dependencies or create worktrees. + +Each task receives an isolated project, HOME, and Session store, with MCP, LSP, +hooks, and plugins disabled. The host verifies real file reads, limited path +changes, `npm test` after the final edit, boundary examples in an independent +directory, and returned values. Changing tests or `package.json`, adding +unrequested files, omitting tool evidence, claiming success, or returning only +`exit(0)` cannot pass. + +Results default to `.blade/benchmarks/controlled-coding-v2-history.json`. This +is not a large-repository benchmark and does not establish overall parity with +other coding agents. ## Local Gate -Execute from the repository root: +Run from the repository root: ```bash bun run qualify:local ``` -The command runs 14 checks in a fixed order: +The command executes 14 checks in order: -1. `type-check` -2. `format:check` -3. `lint` +1. TypeScript type checking +2. format checking +3. lint 4. unit tests 5. integration tests -6. CLI integration tests -7. headless/runtime core regression +6. CLI tests +7. Headless/runtime core regressions 8. E2E -9. snapshot +9. snapshots 10. security tests -11. current source build +11. production build 12. Web tests -13. Web type-check -14. performance regression - -Each step runs in an independent subprocess. A non-zero exit from the first step immediately halts execution; subsequent steps are not counted as passed. This gate does not access paid models and does not depend on `~/.blade/config.json`. +13. Web type checking +14. performance regressions -Build and test commands from `scripts/test.js` do not spawn a child or watchdog when their cancellation signal is already aborted. Cancellation during execution still waits for the owned process tree to exit, without changing timeout budgets. +The first non-zero exit stops the gate. This gate does not access paid models. -Vitest setup creates a unique temporary `BLADE_STORAGE_ROOT` for each test-file lifecycle, and deletes it synchronously during teardown. A `BLADE_STORAGE_ROOT` explicitly passed by the caller is always managed by the caller; the test harness does not delete it. +`test:headless-core` explicitly runs `headless-boundaries.test.ts` and +`headless-event-contract.test.ts`. The latter directly verifies +`HEADLESS_EVENT_VERSION`, `createHeadlessJsonlEvent`, and +`HeadlessJsonlEventSchema`. Every explicit test inventory is checked before +Vitest starts; a missing path fails immediately. -The real-api setup loads credential configuration, the model catalog, and the application store only when `REAL_API_TEST=1`. With paid tests disabled, isolated storage is still created and reclaimed, and keyless regressions still run. Local keyless real-api files load in parallel within the existing four-worker ceiling, retaining process and file isolation. Paid matrices and CI remain serial with one worker. The test inventory, retry rules, and process timeout budgets are unchanged. +V8 coverage runs separately through `bun run test:coverage`. It covers unit, +integration, CLI, E2E, snapshot, security, and keyless real-api fixtures while +excluding the wall-clock performance project. -The GitHub `Quality Gate` re-runs the full-repo format check and CLI lint before build, and the workflow source contract enforces the install → format → lint → build order. root, CLI, and Web use the same exact Biome version to avoid workspace binary resolution differences causing divergent results between the local gate and CI. +## Real API Gate -V8 coverage is executed separately via `bun run --filter blade-code test:coverage`. The coverage orchestration covers unit, integration, CLI, E2E, snapshot, security, and real-api fixtures that do not require credentials, but explicitly excludes the wall-clock `performance` project; instrumentation and parallel project load make startup latency non-comparable. Performance regression remains a required item in `qualify:local` after the production build. +Real API tests must use `packages/cli/dist/blade.js` freshly built from the +current source. Provider credentials may be injected by a secret manager or +stored in `~/.blade/real-api-credentials.json`. That file must be a regular +file owned by the current user, mode `0600`, and at most 64 KiB. Symlinks, +loose permissions, and unknown fields fail closed. -## Real API Gate +Do not put credential values in inline `KEY=value` commands, shell history, or +evidence documents. Logs retain only variable names, model IDs, counts, +timings, and redacted host evidence. -The real API gate must use `packages/cli/dist/blade.js` freshly built from the current source. Provider credentials can be injected into the test subprocess environment by a secret manager, or placed in a test-specific `~/.blade/real-api-credentials.json`. The latter must be a regular file owned by the current user with permissions `0600`; symlinks, loose permissions, unknown fields, and files exceeding 64 KiB all fail closed. Do not write real values as inline `KEY=value`, leave `export` in shell history, or copy them into evidence documents. Command records only retain variable names, model IDs, and presence/absence; they do not record variable values. After environment preparation, execute: +Install Chromium after the first checkout or a Playwright version change: ```bash -bun run --filter blade-code browser:install # First time or after Playwright version changes -bun run --filter blade-code browser:check -bun run qualify:production +bun run --filter blade-code browser:install ``` -`browser:check` does not access the network or implicitly download anything; it only verifies that the locked Playwright version's Chromium executable is runnable and can launch/close. Production qualification runs 16 checks in fixed order: the 14 local gates, keyless Chromium preflight, and finally paid real APIs; the preflight must fail before any Provider request is launched. Browser pages only access the loopback Blade server; API keys never enter page context. - -The Provider request timeout and silent-stream idle timeout for each capability trajectory must both be shorter than the Vitest test timeout, leaving a deterministic cleanup window for abort, runtime dispose, temporary directory deletion, and global config restoration. Trajectories verifying permission recovery disable Provider retry; retry/backoff are only tested in dedicated fault-injection trajectories to avoid old `finally` blocks running concurrently with Vitest retry after the test framework times out. Write qualification for permission recovery uses a unique absolute `file_path` and a Write-only tool whitelist; text-only answers cannot substitute for real file side effects. - -This command runs a fixed release-blocking real matrix: real DeepSeek Flash/Pro Headless bugfix, GPT Web structured output, Claude ACP structured output, DeepSeek headless structured output, Web/ACP/TUI code review, durable interaction recovery, permission recovery, ACP model switch, ACP durable fork Write+Bash capability routing, transparent 503 retry proxy, durable 413 compaction proxy, real mid-stream stall proxy, assistant response fsync fail-stop/cold retry, turn-final receipt exactly-once recovery, foreground/background shell hard-crash recovery, and DeepSeek Flash Runtime/Web/ACP host-authoritative Goal completion verification. ACP fork fixedly requires DeepSeek Flash/Pro, and executes the same paired SDK trajectory for Claude, GPT, and domestic models configured in the current qualification environment; Clients that do not declare terminal capability must use Session-bound local terminal, and terminal failures after declaration must still fail closed. -Agent Team task-list coordination fixedly runs DeepSeek Flash/Pro; each model launches four real background teammates sharing `taskListId`, requiring each teammate to actually call `TaskCreate` once, and verifying that final task IDs are sequential and unique, subjects are fully preserved, JSON snapshots are parseable, cross-process locks are released, and same-process keyed coordination has returned to zero. Plain-text claims of task creation, only checking HTTP 200, mocking ToolExecutor, or only verifying single-process managers cannot substitute for this trajectory. -The fresh PASS authority for Goal completion is simultaneously bound to the current host run, mutation revision, and completion candidate identity consisting of Goal ID, attempt, and requested-at timestamp. When a model resubmits the same idempotent `UpdateGoal complete`, the verifier Session ID, verdict, evidence digest, and finalization snapshot already recorded by that host run must be preserved; candidate changes, workspace mutations, or process restarts can invalidate receipts, but repeated candidates cannot consume additional verification retry budget. -Goal premature-stop recovery fixedly runs the DeepSeek, Claude, GPT, and domestic models configured in the qualification environment. Each model must first produce one tool-free `self_deferral`, then receive the host recovery continuation without user input and complete a real file write, read-back verification, and independent Goal verifier. The trajectory uses an explicit token budget so a failing fixture cannot form an unbounded loop. -The Goal verifier feedback trajectory must first make a real verifier reject a missing artifact, then prove that the executing Agent recovers from the concrete persisted and sanitized gap and completes the work. Deterministic state-machine tests cover second-match strategy escalation and third-match automatic blocking for an identical feedback fingerprint. -The Goal execution-host failure trajectory fixedly runs DeepSeek Flash/Pro across Headless, real ACP stdio, raw PTY TUI, and production Chromium Web. In each cell the real model emits three Bash tool calls, the production Bash adapter produces three typed timeouts, and GoalStore atomically blocks on the third logical turn without starting a fourth continuation. Ordinary non-zero exits must not enter this streak. Web reload, ACP metadata, TUI status, and Headless JSONL must agree with the same durable Goal snapshot. See [Goal Execution-Host Failure Guard Qualification Evidence](./goal-execution-host-failure-evidence.md). -The durable Goal turn lineage trajectory fixedly runs DeepSeek Flash/Pro across Headless, real ACP stdio, raw PTY TUI, and production Chromium Web. Every cell must complete three real upstream model tool decisions and exactly six downstream Provider requests while forming a continuous root/current/parent chain; framework retry and model retry are both zero. The Goal sidecar, durable `turn_started`, ACP metadata, Headless JSONL, TUI status, and Web DOM attributes after reload must agree, and lineage must not enter Provider prompts. See [Durable Goal Turn Lineage Qualification Evidence](./goal-turn-lineage-evidence.md). -The complete `test:real-api` additionally includes the GPT Prompt Cache efficiency trajectory: first wait for a real cache read, then replace all stable prompt blocks, and require the runtime to output `system_prompt_changed` attribution. This trajectory simultaneously verifies adaptive token thresholds; it must not use mock usage, fixed cache counters, or only compare cumulative hit rates. Due to unstable GPT channel latency and availability, it is not part of the release-blocking matrix. -Browser Panel qualification uses a real DeepSeek Flash Session and production Chromium. The same Web page must first complete a real Provider turn, then open Browser from the right-side Preview. Preview loads two independent loopback fixtures and verifies iframe sandbox/no-referrer, back, forward, reload, and system-browser opening. Test must use an independent server Chromium, return a PNG and ARIA refs, perform a real form fill and click, and expose console diagnostics. Neither surface may escape the mobile full-screen modal. URL validation rejects non-HTTP(S), embedded credentials, and Blade's own origin in Preview. Console/page/request faults, Provider credentials, server/browser/port residue, and temporary roots must all be zero. This test is fixed in `realApiQualification` and cannot be replaced by jsdom-only coverage. -Bounded foreground output fixedly runs DeepSeek Flash/Pro × Headless/production Chromium Web/raw PTY TUI/real ACP SDK terminal in eight cells; per-cell Provider deadline is 180 seconds, test timeout 240 seconds, complete realApiQualification watchdog is 90 minutes, and the release matrix fixes framework `retry=0`. -Each cell additionally verifies surface egress: Headless waits for `write(false) -> drain`, ACP has at most one `sessionUpdate()` in-flight, raw PTY pauses the reader then continues rendering, and Web resumes the same tool/final state via durable cursor after reload during runtime. raw PTY must latch already-observed final markers, stdout/stderr retained tail, and truncation notices; resize redraws rotating the bounded terminal window must not retroactively erase established evidence; it must also re-observe the truncation notice from new PTY data after resize, and cannot pass on historical hits from before resize. -The model's entire final response must strictly equal the per-cell marker; ACP failure diagnostics may only retain bounded, redacted final text previews, and cannot use relaxed markers or framework retry to mask model deviation. -Root-turn crash auto-resume additionally fixedly runs DeepSeek Flash/Pro × Headless/raw PTY TUI/production Chromium Web/real ACP `session/load` in eight cells, and all entrypoints must not depend on additional wake-up prompts. Final response tokens must be distinguishable from resume prompts, marker files, and Read output, must not appear in full in prompts, and must not manufacture irrelevant Provider spelling ambiguity via repeated word segments; when Web has observed a resume prefix but complete tokens do not match, it must immediately output bounded, redacted assistant text, and cannot degenerate into a fixed 180-second locator timeout. raw PTY must observe acknowledgement by exact inbox message ID, followed by `turn_completed` for the same turn, and cannot use terminal history hits or fixed wait windows in place of durable terminal. ACP multi-Provider control final answer windows must be later than the 180-second runtime hard timeout, leaving remaining test window for connection teardown and temporary directory cleanup; ACP fork parent Read and child Write/Bash are two sequential prompt stages, each sending standard `session/cancel` with a host 180-second deadline and waiting for prompt convergence; the outer budget is fixed at 420 seconds, covering only the two stage deadlines plus a 60-second cleanup margin. It must not be extended via Provider or framework retry. Edit+rewind, Goal finalization crash handoff use the same Flash/Pro × four-entrypoint eight-cell matrix; the recovery phase must have zero Provider requests, followed by real API follow-up completion from the same surface. PTY follow-up user input must not contain the complete expected response marker; Provider request bodies must first parse JSON `messages` before verifying prompts, to avoid input echo masquerading as assistant completion. Completed-subagent adoption and background-subagent completion wake-up also each fixedly run Flash/Pro × Headless/raw PTY/production Chromium Web/real ACP eight-cell matrices. -Bounded coordinated shutdown additionally fixedly runs the same Flash/Pro × four-entrypoint eight-cell matrix; each cell sends production `SIGTERM` after real foreground Bash enters the host-visible PID barrier, requiring exactly-one cancelled abort, same-Session recovery, delayed side-effect control, and full resource reclamation. -High-cost provider/capability soaks such as open multi-file migration, compaction, process trees, concurrent owners, and crash-tails are run separately via: +Run production qualification with: ```bash -bun run test:real-api -``` - -Domestic model channels are optional soak providers and do not enter the default release-blocking matrix. To explicitly include them, set `REAL_API_INCLUDE_OPTIONAL_PROVIDERS=1`; insufficient balance or shared channel throttling cannot lower the required qualification standards for DeepSeek, Claude, or GPT. The extended required matrix described by capability below describes the complete soak contract, and does not indicate that every patch must synchronously block release. - -The credentials file format is as follows; `baseURL` and per-model `model` may be omitted: - -```json -{ - "version": 1, - "providers": { - "deepseek": { - "apiKey": "...", - "baseURL": "https://api.deepseek.com", - "models": ["deepseek-v4-flash", "deepseek-v4-pro"] - }, - "claude": { - "apiKey": "...", - "baseURL": "https://gateway.example.com", - "model": "claude-opus-4-8" - }, - "gpt": { - "apiKey": "...", - "baseURL": "https://gateway.example.com", - "model": "gpt-5.5" - }, - "domestic": { - "apiKey": "...", - "baseURL": "https://gateway.example.com", - "model": "qwen3.8-max" - } - } -} +bun run qualify:production ``` -Other files can be pointed to via `BLADE_REAL_API_CREDENTIALS_FILE`. Explicit Provider environment variables take precedence over same-named fields in the file; as long as any explicit API key exists in the command environment and no credentials file is explicitly specified, that environment variable set becomes the complete allowlist, and the default credentials file or personal models are not implicitly merged. - -`qualify:production` fail-closed validates before launching any test subprocess: - -- The DeepSeek key must come from explicit environment, restricted credentials file, or `~/.blade/auth.json`; -- `DEEPSEEK_MODELS` must simultaneously contain `deepseek-v4-flash` and `deepseek-v4-pro`; -- When `DEEPSEEK_BASE_URL` is not provided, `https://api.deepseek.com` is used; -- `DEEPSEEK_MODEL` defaults to the first model in the list for single-model trajectories; -- Claude, GPT, and domestic qualification configurations are projected as independent `modelProviders` channels; each channel uses its own provider-level endpoint and dedicated process credential slot, and same-protocol channels do not share keys; -- API keys are not written to project configuration, source code, command arguments, logs, JSONL, or snapshots. - -The release-blocking matrix fixedly includes DeepSeek Flash and Pro, and cross-provider Web/ACP structured output is verified via Claude and GPT. Explicitly enabled Claude, GPT, and domestic models in the full soak must also pass basic chat, streaming, usage, finish, tool calling, and the four production entrypoints: Runtime, TUI, Web, ACP. Receiving only text or HTTP `200` does not count as passing. Cross-end fork trajectories must have pi-ai resolve credentials from custom channels, and must not bypass via model-level `apiKey` parameters. Channel health qualification also sends real probes of at most 8 tokens to GPT, domestic, and Claude respectively from Web routes, TUI `/doctor`, and ACP callbacks; results must use canonical failure projection and must not contain model raw text, raw errors, or API keys. - -Provider Retry qualification must use a transparent local proxy to return `429` or `503` with `Retry-After` on the first real model request, and forward the second request unchanged to the real Provider. A single user turn must complete real tool modifications and project tests without relying on test framework reruns; Headless JSONL must sequentially show sanitized `provider_retry` `scheduled`, `attempt`, `recovered`, producing only one final content and one tool side effect. Deterministic transport tests separately cover jitter/cap, HTTP-date, `retry-after-ms`, backoff cancellation, quota/context fail-fast, no replay after partial output, exhausted and fallback; Provider response body, headers, and keys must not enter LoopEvent, SSE, ACP, JSONL, or transcript. Production Web GUI must display retry attempt and bounded waiting in the same Session StatusBar, then complete without refresh, fresh tab restores final result, and browser console has no application error. TUI, when conditions permit, verifies loading state and Esc cancellation via Computer Use; ACP must project via `session_info_update` metadata and not pollute assistant body text. - -Bounded foreground Provider recovery additionally fixedly runs DeepSeek Flash/Pro × Headless, real ACP stdio + child-backed terminal, raw PTY TUI, and production Chromium Web GUI in an eight-cell matrix. The transparent proxy must return replay-safe `503` for the first four model requests, and only forward the fifth request unchanged to the real Provider; therefore the old default of 2 retries cannot pass. Each cell must, within the same root turn: - -- Sequentially project attempts `1,2,3,4` and recovered `4/12`; -- Carry `mode=bounded_foreground`, total budget, and non-negative elapsed/remaining; -- Execute only one Edit and one Bash, with the host rerunning fixture tests; -- Maintain that Provider payload, transcript, and final workspace show no duplicate side effects; -- TUI displays bounded recovery and Esc, ACP only sends metadata, Web StatusBar shows recovery in real time and clears after completion, and reload preserves final result; -- Headless, ACP, raw PTY TUI, and Web must observe the same generation/revision-fenced - `provider_recovery` projection. Web reload/reconnect must receive the authoritative - snapshot before readiness, and a late live revision without a revision-zero anchor - must not revive state after terminal clear; -- Both the accessible banner above the Web composer and the TUI loading/status - surfaces must present the effective circuit/retry state, an absolute-deadline - countdown, and the existing Stop/Esc cancellation path; -- typed `model_fallback` must carry exact sanitized source/target identity and a closed - trigger, without clearing or overwriting the authoritative Runtime snapshot; -- Reclaim Provider proxy/socket, ACP terminal/process, PTY, browser/page, SSE, server, port, temporary HOME/storage/workspace; -- Must not write Provider keys, private failure bodies, or raw errors into JSONL, SSE, ACP, DOM, terminal capture, transcript, or test evidence. - -Deterministic tests must also cover recovery deadline winning within backoff and in-flight streams, caller abort, hard cap of 12 additional attempts, fallback shared clock, timer cleanup, and replay prohibition after any text, reasoning, tool call, usage, or finish chunk. Release evidence for the feature matrix must use `retry=0`. - -Shared Provider circuit reuses the same eight-cell matrix, and configures Open time to the valid `2000ms` (production default `10000ms`). After the 4th `503` completes, the transparent proxy records a monotonic boundary; subsequent requests must not arrive within the Open window; after expiry only one HalfOpen probe is allowed. Each entrypoint must project `opened -> waiting -> probe -> closed` according to the same typed schema; TUI circuit status takes priority over normal retry text, and Web/ACP must not write metadata into assistant body text or durable transcript. - -Web and ACP Flash/Pro cells must additionally launch a second Session within the same process. Session B submits after Session A has Opened, with zero Provider requests within the Open window; both Sessions together have exactly one initial probe, and after closing each obtains independent real Provider results. Headless/TUI do not host same-process multi-Session, and their cells only verify single-Session state machines and surface projection. All eight cells and dual-Session controls use `retry=0`, and reclaim secondary Sessions, SSE, terminals, browser/page, proxy, and temporary directories. - -Provider request admission qualification fixedly runs DeepSeek Flash/Pro × Headless, real ACP, raw PTY TUI, and production Chromium Web in eight cells, with test configuration using valid `providerRequestConcurrency=1` and `providerRequestAdmissionMs=120000`. The transparent proxy first stalls one admitted real Provider request at a host barrier; the second same-domain request must first project `queued`, and the proxy can only observe one request before release. After release, the second request projects `admitted` in order and obtains an independent real Provider result, and the proxy's maximum same-domain in-flight must be exactly 1. - -Web/ACP cells use same-process dual-Session controls. Headless/TUI cells use a production background Task child holding the permit, with the parent queued on the next round; both child and parent must access the real Provider, allowing the proxy to generate synthetic tool calls only for deterministic Task triggering. The eight cells must also verify root owner inheritance, zero framework retry, attempts not including queue wait, metadata not entering assistant/transcript, Web reload/ACP load clearing transient state, TUI Esc visibility, and full proxy/browser/PTY/process/profile reclamation. - -Weighted Provider admission additionally fixedly runs DeepSeek Flash/Pro × Headless, real ACP, raw PTY TUI, and production Chromium Web in an eight-cell negative matrix. With test configuration at valid `providerRequestPendingBytes=65536`, first let one large-context parent or Session A request go immediately active, then let the same-process background child or Session B large context enter waiting; the latter must project `rejected/queue_full/pending_bytes`, and the transparent proxy observes zero additional Provider traffic. -Headless/TUI uses parent holding active permit, background child rejected, proving the same fact respectively through schema-valid child JSONL or "failed sidecar + visible TUI failure"; Web/ACP uses dual root Sessions. The final queue-full turn retains failed abort but confirms this turn's input, and reload, SSE reconnect, or ACP load must not replay Session B markers to the Provider. Web must submit from the real composer and maintain zero console/page errors. TUI failure must be directly projected by the host's bounded terminal summary, and cannot rely on model retelling; raw PTY must also observe hidden completion acked before finishing, with `turn_completed` existing after its submit seq, to avoid overlap between teardown and held proxy requests. -All eight cells use `retry=0`, while continuing to run the aforementioned normal queued→admitted matrix to prevent the byte policy from narrowing normal long tasks to only failing closed. - -Weighted top-level task admission qualification uses the legal minimum `maxQueuedTaskBytes=65536`. DeepSeek Flash/Pro × production Chromium Task Home and real ACP stdio four-cell negative targets must first stall task A's real Provider request at a host barrier, then submit task B containing a unique marker and a large segment of non-ASCII text. B must be rejected with `pending_bytes` before Provider traffic: Web returns typed HTTP 429, displays inline error, and catalog/reload has no ghost task; ACP persists and projects retryable `capacity/pending_bytes`, with assistant body text not containing admission metadata. Subsequently normal small task C must enter queue position 1, and obtain independent real Provider results after A releases. Proxy request bodies must not contain B markers. The Task Home composer being able to accept input does not mean dispatch is available; Chromium must first observe explicit `data-blade-task-dispatch-ready=true`, a state jointly constrained by workspace and model readiness. Readiness timeout must output structured status diagnostics, and must not blindly wait for submit disabled or bypass via reload/retry. - -Non-interference controls fixedly run Flash/Pro Headless `--task-isolation local` coding tasks and Flash/Pro raw PTY root turns; all complete under the same minimum byte limit. Existing production Web task dispatch Flash/Pro worktree coding/FIFO trajectories continue to be release-blocking. Targets and controls all use `retry=0`, and reclaim rejected inbox/Runtime/worktree, accepted tasks, browser, ACP process/terminal, proxy/socket, port, HOME/storage/workspace. Chromium only allows expected `/tasks` 429 resource errors; other console/page/request faults must be zero. - -Bounded Session Runtime residency qualification fixedly uses `maxResidentSessionRuntimes=1` and `sessionRuntimeIdleMs=30000`. DeepSeek Flash/Pro × production Chromium Web GUI two cells first hold Session A's real Provider request, then Session B verifies typed HTTP 429 `resident_runtimes`, zero rejected marker Provider traffic, idle LRU slot reuse, and Session A durable cold follow-up. The browser only allows expected 429 resource errors; other console/page/request/SSE faults must be zero. - -Real ACP stdio two cells must advertise and invoke standard `session/close`: when active Session A occupies the only slot, new Session B receives bounded JSON-RPC capacity error before task/worktree/Runtime/Provider side effects; closing A cancels and settles prompts, confirms cancelled inbox, releases slot, then B completes a real turn, and after closing B, loading A completes real follow-up from durable transcript. Flash/Pro Headless and raw PTY each run one single-root coding control cell, proving resident limits do not interfere with non-multiplexed Runtimes. Targets and controls totaling eight cells all use `retry=0`, and check full reclamation of Runtime reservations, ACP/TUI processes, Provider proxy, browser/profile, HOME/storage/workspace. - -Deterministic keyed coordination reclamation qualification requires all keyed mutex registries to retain only active/queued operations. Deterministic gates must cover success, synchronous exceptions, async rejection, same-key FIFO, cross-key parallelism, new generations after reclamation, 10,000 historical keys, and 4,096 queued operations/256 keys; the seven owners: durable interaction, Goal, OAuth credential, Config, Worktree, Web message submission, and task delivery, must all precisely return to zero after high-cardinality churn. - -Real API targets fixedly run DeepSeek Flash/Pro × production Chromium Web GUI and real ACP in four cells. Web must complete multiple Sessions sequentially through the real composer, same-Session follow-up and reload, asserting message/task coordination is zero after each settle; old SSE `ERR_ABORTED` caused by navigation must be separately classified, other console/page/request faults are zero. ACP must complete multiple Sessions, standard `session/close`, `session/load` follow-up on the same multiplexed connection, and assert interaction/Goal/OAuth coordination and Runtime residency return to zero. Statistics may only be read via in-process test seams, and are prohibited from entering HTTP, ACP, CLI, transcript, or persisted schema. Four cells and existing Headless/raw PTY non-interference controls all use framework `retry=0`. - -Provider Stall qualification must have the transparent SSE proxy first forward real model content, then pause subsequent events before the hard idle timeout. Headless JSONL must output sanitized `detected → recovered` at the same stall count, mark `output_started=true`, then complete the real reply; the proxy must prove only one Provider request was received, and no retries, overlapping `iterator.next()`, duplicate content, or tool side effects may occur. Deterministic transport tests separately cover stall before first event, mid-stream stall, hard timeout after warning, caller abort, deadline reset, and only one pending read after warning. Production Web GUI must display stall duration/hard deadline in StatusBar, complete without refresh after recovery, and console has no application error; TUI must display stall state, hard deadline, and Esc cancel entrypoint; ACP and Headless only project metadata, not polluting assistant body text or durable transcript. - -Provider total-attempt deadline qualification must first forward real DeepSeek Flash content through a transparent SSE proxy, then delay completion beyond `timeout` but before `streamIdleTimeout`. Headless must produce a typed error at the 45-second total deadline, retain delivered content, have zero retries, a single Provider request, and synchronously abort the proxy's upstream fetch. Production Chromium Web must display the same `[data-blade-session-error]`, proving assistant partial content is visible, credentials do not enter DOM, and console/page faults are zero; terminal durable resync only allows the old `/sessions/:id/events` EventSource to produce exactly one `net::ERR_ABORTED`, other request faults must be zero. browser, server, proxy, HOME/storage/workspace must all be reclaimed. - -Reactive Compaction qualification must have the transparent proxy return one `413 context_length_exceeded` on the first real turn, then forward subsequent compression summaries and recovery requests unchanged to the real Provider. The runtime must emit paired compaction lifecycle within a zero-output replay boundary, first submitting a JSONL checkpoint containing exact replacement messages, then retrying the same turn; no `provider_retry`, duplicate tool side effects, or infinite compaction loops may be produced. A second independent Runtime must recover previous markers solely from checkpoint model projection, while complete transcripts remain available for UI display. Production Web GUI must display "Context limit exceeded, recovering...", completing the final reply without refresh; fresh tab restores visible history and continues answering checkpoint markers, with browser console zero application errors. Real raw PTY TUI must display "Compacting" and Esc entrypoint and complete the same marker; ACP, Server SSE, and Headless JSONL only expose reason/strategy/outcome/token metadata, not leaking Provider error bodies. - -Tool concurrency qualification requires GPT to simultaneously call two loaded tools in the same production stream. The two tools wait for each other within their execution functions, and can only release when both have entered the shared gate; therefore simply reducing total elapsed time or executing sequentially cannot pass. Deterministic tests separately cover exclusive FIFO, same-path file locks, abort, fallback epoch, Web multi-card refresh reconstruction, TUI keyed progress, and ACP independent tool-call IDs. - -Bounded fair tool admission additionally fixedly runs DeepSeek Flash/Pro × Headless, real ACP stdio + PTY terminal, raw PTY TUI, and production Chromium Web GUI in an eight-cell matrix. The model must issue four real foreground Bash calls in a single response; after all four canonical calls are submitted, the host proves that per-Session initially only two start, third/fourth wait, each release advances only one successor, and durable call/result maintains Provider order. An independent dual-Session Chromium trajectory requires that when Session A occupies two execute slots and queues a third, Session B completes first using remaining global slots; both Sessions retain final states after reload. All cells simultaneously verify queue progress, typed overload metadata, process tree/lease/port/browser/PTY/ACP/temporary root reclamation, and Provider credential absence. - -Bounded foreground command handoff fixedly runs DeepSeek Flash/Pro × Headless, real ACP stdio + child-backed terminal, raw PTY TUI, and production Chromium Web GUI in an eight-cell matrix. The model must start the same host-barrier Bash with `run_in_background=false`; after the 1-second test configuration budget expires, durable result and surface must first publish `auto_backgrounded=true`, typed reason/budget, and the same `shell_id`. While the child process is still active, the model completes an independent Read, and only then does the host release the barrier; subsequently exactly one TaskOutput obtains both output markers before and after handoff. Each cell proves: - -- The command starts only once, local PID or ACP terminal child identity does not change; -- Foreground lease is atomically replaced by background lease, ACP does not release early or fallback locally; -- Headless typed JSONL, TUI, Web SSE/DOM, and ACP updates all see the handoff; -- After TaskOutput final state, process/terminal, foreground/background lease, port, browser, PTY, SSE, temporary root, and Provider credentials are all zeroed. - -Durable token-budget handoff fixedly runs DeepSeek Flash/Pro × Headless, raw PTY TUI, production Chromium Web, and real ACP SDK in an eight-cell matrix, with fixed order and not filtered through release surface. Each cell forwards real task Provider output through a loopback transparent proxy; the proxy rewrites usage counters for the first two task responses, returns one controlled context overflow for the first compaction request, and requires the second request to use a strictly smaller payload. Pro then receives one `503` before the third real summary request is transparently forwarded. Flash instead receives a second `503` on the third compaction request, forcing token-targeted deterministic fallback. The 70%/80% thresholds come from the production model catalog's context window and output reserve, not test hardcoding. The proxy fixes the corresponding request's `promptTokens` at one token below the boundary; only a complete pre-request projection that includes real `completionTokens` and post-response tool/control deltas can cross it, so the old prompt-only implementation cannot pass. Each cell proves: - -- The handoff marker is persisted before the second task Provider request, current epoch has exactly one `handoff-message-` durable identity, pre-compaction requests have at most one marker, compaction and subsequent requests have zero; -- The threshold checkpoint records `preTokenSource=provider_plus_estimate` and `estimatedPendingTokens > 0`; Headless JSONL, Web SSE, and ACP metadata project the same fields, while the TUI and Web context meters use complete Provider totals rather than prompt-only usage; -- The latest replacement checkpoint is located after the marker, replacement/effective suffix does not contain the marker, the seven-segment continuation ledger precisely preserves mutation, failed verification, and pending action sentinels, the checkpoint records `sampleAttempts: 3`, `inputReductions: 1`, and at least one nonzero omission count across `messagesOmitted` and `filesOmitted`; Pro has no failure reason, while Flash records `transient_exhausted`, proves `postTokens <= fallbackTargetTokens` and `fallbackTargetTokens > 0`, deduplicates the original active-task message, and carries both fallback message counts. The complete final marker is emitted only by the passing verification command and the final assistant must echo it verbatim; the durable order of real Bash fail, Write, Bash pass is correct; -- Headless cold projection, PTY resume, ACP `session/load`, and Web post-completion reload must not add new Provider requests; internal events/tags/identities/reminders do not enter terminal bytes, ACP updates/terminal, HTTP history, SSE, Zustand, DOM, or HTML; -- Web reloads during real execution, and carries page-owned EventSource evidence before each navigation; PTY byte stream and raw Session JSONL are terminal authority; ACP uses real paired NDJSON codec and child-backed terminal; -- browser/page/SSE, PTY, ACP terminal/process, server/proxy/port, Session/foreground lease, temporary HOME/storage/workspace are all reclaimed, evidence and logs contain no Provider credentials. - -This release-blocking trajectory has framework retry set to 0 when `REAL_API_RELEASE_MATRIX=1`. Desktop computer-use can only serve as non-blocking visual observation; it cannot prove JSONL, Provider request order, or marker non-fan-out, therefore it is not the authority for this runtime contract. - -Compaction rich-media elision generates summaries directly through configured DeepSeek Flash/Pro, Claude, and GPT Providers. Each request starts from text plus an inline data-URL image and a remote image URL; before forwarding, the loopback proxy must prove that both raw image payloads are absent while the fixed image placeholders and textual evidence remain. The test also requires the exact `imagesOmitted` count, unchanged canonical source messages, Provider concurrency of one, and evidence that retains neither credentials nor omitted media markers. Deterministic tests cover the same sanitizer, fallback metadata, and Headless JSONL, Server SSE, and ACP projection. - -The compaction effectiveness guard uses the same real DeepSeek Flash/Pro, Claude, and GPT matrix, with every cell exceeding the 5,000-token enforcement threshold. A normal real summary must keep the complete replacement at or below 80% of the original compactable message body and at or below 50% of the model context window. A deterministic negative control injects a non-empty oversized summary and requires the host to reject that LLM candidate, retain its billable usage, persist `insufficient_reduction` in the durable fallback checkpoint, and project the stable classification through TUI, Headless JSONL, Server SSE, and ACP metadata. Tests must compare the actual replacement context rather than only the model's summary text. - -Token-targeted deterministic fallback qualification must cover newest-first selection by tokens rather than message count, atomic assistant tool-call/result units, head-and-tail truncation of an oversized boundary message, exclusion of reasoning and image payloads from the replacement, immutable canonical source messages, and `postTokens <= fallbackTargetTokens`. The target is the smallest of `max(5,000, 80% of the original message body)`, 50% of the model context window, and 50,000 tokens, raised only when exact continuation records and the active-task checkpoint themselves require more space. A real DeepSeek Flash control must first produce a real summary, then deterministically enter fallback through the complete-replacement 50% headroom guard, and verify persistence and cross-surface projection of `fallbackTargetTokens`, `fallbackMessagesOmitted`, and `fallbackMessagesTruncated`. - -Fresh independent verification qualification requires the main model to actually complete non-trivial implementation of three files, and at the end of the first attempt the runtime forcibly starts a new built-in `verification` subagent. The Verifier must be in an independent child Session, run the project's configured real tests, and return exactly one structured PASS; FAIL/PARTIAL, missing verdict, continuing to write after PASS, or exhausting retries must not successfully complete. Deterministic tests separately cover backend/API/infrastructure single-file triggering, documentation/fixture exclusion, reserved agent, YOLO read-only Bash, out-of-bounds cwd/env/background/redirect rejection, and durable mutation revision recovery. The local verifier's sandbox must also prove workspace is not writable, network is closed, user home/Blade storage is not readable, and provider keys/Session env do not enter the child process. Production Web GUI must display a unique verification card and PASS badge, three changed files, and the final marker; after clearing browser state and restarting the server, the same card must still be restored, and the page must not expose internal completion reminders or application console errors. ACP must display the same verdict through standard Task tool-call updates. - -Native Read-Only Code Review qualification must cover `uncommitted`, `base`, and `commit` targets, tracked/untracked SHA-256 digests, 500 files/8 MiB boundaries, precise changed-line verification, stale, abort, process-restart interrupted, fork/rewind, and single-active-review. Built-in `review` and `verification` share audit authority: write tools, background commands, env override, out-of-bounds cwd, and network must be rejected; in sandbox workspace is not writable, HOME/Blade storage/provider keys are not readable, while Git reads targets normally through isolated config. - -Real GPT must find the same authorization bypass through production Web routes, ACP `/review`, and TUI runtime hooks respectively, return structured P0/P1 findings within targets, and be proven by the host that file bytes and Git status are completely identical before and after review. Production DeepSeek GUI must launch from the Task Home "Review" template, displaying independent running state, read-only tool progress, and completed reports in real time without manual refresh; fresh tab must also restore priority, relative path, line, and confidence. Structured report chrome must support both Chinese and English, and browser console must have no application errors. - -Hook trust qualification requires GPT to actually issue tool calls through production streams. The same PreToolUse command must have zero side effects in `untrusted` state, and can only execute after the current SHA-256 digest is explicitly trusted. Deterministic security tests separately cover `0600` atomic storage, symlink/owner/mode fail-closed, Git worktree common root, config changes automatically entering `modified`, cross-workspace config isolation, stale reviewed digests returning `409`, and managed Function hooks not entering project summaries. Production Web GUI must verify review, trust, modified, re-trust, revoke, Escape focus recovery, and fresh-tab console. - -Workspace Trust qualification must simultaneously place malicious model endpoints, stdio MCP markers, `permissions.allow: Bash(*)`, BASH_ENV, project instructions, and side-effect-bearing package `type-check` scripts in an isolated repository. Untrusted trajectories must filter all project layers, complete real GPT SessionRuntime turns through user channels, and prove that malicious HTTP endpoint request counts and MCP/package-script markers are both 0. Only trusted local YOLO Sessions allow post-edit verification, and only use declared `type-check` scripts; ACP, `default`/`autoEdit`, scriptless, and untrusted paths must have zero execution. Deterministic tests separately cover `0700/0600` storage, symlink/owner/mode fail-closed, parent directory inheritance and subdirectory deny, Git worktree identity, TUI launch review, Web/ACP management entrypoints, verification process cancellation/reclamation, and discovery gates for plugins/commands/skills/agents/instructions. Production Web GUI must only display package script names rather than command bodies, and complete real model Writes respectively in untrusted Default and trusted Auto Edit modes; neither must produce verification markers, and fresh tab must not show application console errors. - -ApplyPatch qualification cannot only verify final file content. Deterministic tests must inject mid-release rename failures and prove all source/destination recovery and stage/backup zeroing; they must also cover complete grammar, zero side effects on context mismatch, symlink escape, same-path three-way queuing, multi-path deadlocks, Add/Delete/Move Snapshot overall rewind, Hook any-path matching, LSP didClose/didSave, and read-back compensation rollback for ACP remote ambiguous write failures. They must also manually construct `preparing` and `committed` crash journals, proving Session startup executes rollback and cleanup-only respectively, and verifying that two independent calls are serialized under the 0600 workspace lock. +This command runs the 14 local checks, a keyless Chromium preflight, and only +then paid Provider tests. A preflight failure cannot produce Provider traffic. -Real GPT must first Read two existing files, then call ApplyPatch only once to simultaneously update both files and add a third file; must not fall back to Edit, Write, or Bash, and no transaction files may remain in workspace. Production DeepSeek Web GUI must reproduce the same trajectory in Auto Edit local tasks, display three changed files and per-file diffs, and maintain zero application console errors in fresh tab. +### Release-Blocking Matrix -LSP qualification must use real stdio JSON-RPC subprocesses; pure mock transports are not accepted. Deterministic tests cover initialize, didOpen/didChange/didSave, publishDiagnostics, all semantic queries, same-named dual-Session process and environment isolation, ACP zero local processes, ContentModified retries, request abort, bounded restart on crash, and dispose PID reclamation. Project LSP commands must enter Workspace Trust review, and args/env must not be projected to Web. +`test:real-api:qualification` is controlled by a fixed allowlist in +`scripts/test-config.js`. It contains nine test files: -Real GPT must first activate deferred LSP schema through ToolSearch, then actually call hover; the next turn's Write must receive `FAKE1001` diagnostics from the same connection, and PID returns to zero after Session destruction. Production DeepSeek Web GUI must complete Security trust, Auto Edit local task, ToolSearch → LSP → Write diagnostic trajectory, prove at task final state that LSP PID has exited, and fresh tab has no console errors. +1. `agent-trajectory.test.ts`: production Agent read, edit, and test +2. `structured-output-trajectory.test.ts`: structured output +3. `durable-interaction-recovery-trajectory.test.ts`: durable recovery +4. `release-coding-trajectory.test.ts`: cross-surface coding migration +5. `task-list-team-trajectory.test.ts`: Agent Team coordination +6. `cross-provider-fallback-trajectory.test.ts`: cross-Provider fallback +7. `goal-mode-trajectory.test.ts`: Goal creation, execution, and completion +8. `browser-tool-trajectory.test.ts`: native Browser Tool +9. `acp-remote-filesystem-trajectory.test.ts`: ACP remote filesystem -MCP Session isolation qualification uses two trusted projects: the process starts from project A and lets Store contain A's stdio server, while production SessionRuntime is created for project B. Before real GPT turns, it must complete handshake with B's MCP, markers may only appear in B's cwd, and A's marker must remain non-existent. Deterministic tests must also prove builtin tools do not read global MCP registry, ACP/CLI source priority, `--strict-mcp-config`, and connecting/error client reclamation. Production Web GUI must approve B from the Security panel, dispatch real API tasks, see target replies, and prove stdio subprocesses return to zero after task final state. +The release matrix sets `REAL_API_TEST=1` and +`REAL_API_RELEASE_MATRIX=1`, fixes Vitest retry at zero, and requires DeepSeek +Flash and Pro. Cross-Provider cells that require Claude or GPT fail closed when +their credentials are unavailable; they do not degrade to mocks. -MCP Elicitation qualification must use real stdio MCP transport, covering Form, URL, `notifications/elicitation/complete`, no interaction surface, illegal responses, tool abort, and overlapping calls. Form responses must be validated against the original requested schema; Elicitation/ElicitationResult Hooks cannot bypass schema; events and Session transcripts must not save user-filled content. ACP must fail closed on required free text or multi-select that it cannot express. Real GPT must first activate deferred MCP tools via ToolSearch, then consume profiles existing only in elicitation results and continue Write. Production DeepSeek Web GUI must complete MCP tool approval, structured forms, task attention, final replies, fresh-tab zero application console errors, and prove stdio PIDs are reclaimed after task final state. +These trajectories require real Provider requests and host-observable effects: +file content, durable events, tool results, browser state, ACP updates, or test +process results. Model claims, HTTP `200`, mocked ToolExecutors, and jsdom-only +coverage are not substitutes. -MCP Roots/Sampling qualification must have real stdio servers actively call `roots/list` and `sampling/createMessage`. Deterministic tests cover canonical URIs, worktree execution root, ACP empty roots, capability negotiation, configured limits, text/image, unsupported content, request counts, overlapping calls, and parent abort. Sampling is not declared by default; after explicit opt-in each call must still be one-shot approved, YOLO cannot bypass, and TUI/Web/ACP must not display false persistent authorization. Real GPT must complete ToolSearch → MCP tool → nested sampling → Write; production DeepSeek Web GUI must display request preview and token limits, complete final reply, restore in fresh tab, and prove PIDs, ports, and temporary directories are reclaimed. +### General Real API Inventory -MCP Call Lifecycle qualification must use real stdio servers to verify progress tokens, sequential progress, parent abort, idle heartbeat, hard total timeout, and disconnect PID reclamation. Illegal, regressive, or excessive progress must not enter Loop; progress is only projected as transient `tool_progress` to TUI/Web/headless/ACP and subagents, and must not enter model transcripts. Real GPT must complete ToolSearch → progressive MCP → Write; production DeepSeek Web GUI must directly display progress messages and percentages in default collapsed tool groups, and complete final replies and fresh-tab console qualification. +`bun run test:real-api` also uses an explicit inventory rather than scanning +the directory. The current inventory contains the nine files above plus: -MCP Tool Result qualification must use real stdio servers to return text, structured content, image/audio, resource text/blob, resource links, large text, protocol errors, and over-limit results. Binary must not enter models, Web, ACP, or transcripts as base64; artifact directories/files must be 0700/0600 respectively, and verify Session hash isolation, content hash, quotas, and ACP host path hiding. Real GPT must complete rich result → large result → Read private artifact → Write; production DeepSeek Web GUI must display markers, size/SHA-256, artifact paths, and final replies, and prove traces, transcripts, PIDs, ports, and temporary directories have no raw `_meta`, base64, or credential residue. +10. `goal-paused-usage-trajectory.test.ts` +11. `workspace-agent-resources-trajectory.test.ts` -MCP Logging qualification must use real stdio servers covering `logging/setLevel`, `notifications/message`, runtime level adjustment, severity filtering, nested secret/URL/token/`_meta` redaction, 16 KiB projection, 8 KiB messages, 64 messages per second rate limiting, and Session rings. Log events must project to TUI/headless/Web/subagent/ACP, but provider messages and durable transcripts must maintain zero markers; ACP can only display opaque hashes. Real GPT must complete ToolSearch → logging MCP → Write; production DeepSeek Web GUI must display warning/error completion state diagnostic cards, MCP management panel logs and level buttons, final replies, and prove error logs do not increment failed tool counts, and PIDs, ports, traces, transcripts, and temporary directories have no raw credential residue. +The 40-cell `goal-paused-usage` extension is enabled only with +`REAL_API_RELEASE_MATRIX=1`. `workspace-agent-resources` always runs its +built-in DeepSeek skill trajectory and adds workspace isolation when GPT +credentials are configured. To run every release-only cell in the current +inventory, use: -MCP Server Instructions qualification must read real stdio initialize responses, covering NFKC, Unicode tag/Cf/Co/Cn cleaning, 1 MiB source, 8 KiB per server, 32 KiB per Session, JSON/XML boundary escaping, source hash, snapshot replacement, and connection generation revocation. Pseudo-`` content cannot override system/user/permission/trust; ACP can only project provenance hashes. Real transports must complete V1 → crash/remove → V2/re-add and reclaim two generations of PIDs. Real GPT and production DeepSeek GUI must, when users do not provide required code, obtain parameters solely from scoped instructions and complete MCP → Write; Web must also display instruction completion cards and MCP management panel security previews, and traces/transcripts must not contain hidden Unicode or credentials. - -MCP Completion qualification must use real stdio `completion/complete`, covering capability, prompt/resource template catalog ownership, rejecting unknown arguments/context before requests, 15-second timeout, turn cancellation, 4 concurrent per client, 1 MiB source, 100 values, 4 KiB per value, cumulative 64 KiB, NFKC, Cf/Co/Cn/tag/bidi/private-use cleaning, deduplication, and raw SHA-256. Same-named servers must maintain Session isolation and reclaim all PIDs. Real GPT and production DeepSeek GUI must complete ToolSearch → CompleteMcpArgument → scoped candidate → MCP tool → Write, and ignore pseudo system-reminders in candidates. The Web management panel must cover prompt/resource targets, partial values, safe candidates, hashes, truncation, and pending convergence. - -MCP Async Tasks qualification must use real stdio task-capable servers, covering capability and `taskSupport` catalog identity, default disabled, required automatic backgrounding, optional default foreground and explicit `StartMcpTask`, Session/workspace ownership, cancellation, and dispose cleanup. Raw server task IDs, result `_meta`, Bearers, and host paths must not enter models or UI; results must go through shared MCP Tool Result budgets. Fault injection must interrupt `tasks/get` and `tasks/result` respectively, new generations can only resume when task ID + `createdAt` match, and all old PIDs must exit. Real GPT and production DeepSeek GUI must complete ToolSearch → required task → opaque `mcp_task_*` → TaskOutput → Write; Web cards must update in-place from running to completed, and the management panel must display opt-in, Session limits, and poll intervals. - -Durable Session Archive qualification must take JSONL as single source of truth, covering direct archiving, fork/subagent inherited archiving, separately archiving descendants, retaining sub-archives after restoring roots, and independent cursor scope for active/archived. Archiving must acquire stable-order leases for the entire subtree Session; any queued/running descendant or external owner must prove zero additions to root transcript. SQLite recursive projection and JSONL fallback must match item by item; Runtime, metadata updates, Web write routes, and ACP `session/load` must all reject archiving Sessions before side effects. - -Real GPT must complete first turn, archive, Runtime/update dual rejection, restoration, and second turn with durable history. production DeepSeek Web GUI must archive from Session row Popover, prove active catalog is empty and archived messages return HTTP 409, then restore from Archive Popover and complete the same Session's second turn. Transcripts may only show two `archivedAt: timestamp -> null` migrations, rejected inputs must not be persisted; fresh tabs may only have normal Session state logs. After testing, server ports, temporary storage roots, and worktree processes must return to zero. - -Portable Session Markdown Export qualification must directly read stable JSONL snapshots and apply all durable rewind markers; current provider context, Web memory, or SQLite read models cannot substitute. Deterministic tests cover text/image/summary/reasoning, part updates, tool results without message parents, subagent/file activity, active/archived exact workspaces, and no leakage of system recovery content. Credential keys, Bearers, private keys, data URLs, signed URLs, hidden Unicode, and Unix/Windows host paths outside workspace must go through budget projection. Single-activity 64 KiB, total export 16 MiB, and ACP inline 1 MiB must all fail closed or explicit truncation; body SHA-256 must be independently recalculable from UTF-8 bytes after `---`. - -TUI must use `0600` exclusive create and prove same-named files are not overwritten; ACP must not write host paths; Web must reject responses missing hash/count provenance headers. Real GPT must actually call Read to read public markers, pseudo API keys, and host paths, exports must retain call/results and markers while hiding sensitive values, and obtain the same summary from TUI and ACP. production DeepSeek Web GUI must download exact Sessions respectively from active rows, archived Popovers, and fresh tabs; HTTP responses must be `no-store` safe filenames, body hashes match, and fresh tabs have no application console errors. After testing, ports, temporary roots, and download verification artifacts must return to zero. - -Durable Pending Interaction qualification must prove permissions, `AskUserQuestion`, MCP Elicitation, and Sampling requests are written to JSONL before being visible on surfaces, and user responses are written before unblocking tools. Deterministic tests cover size budgets, same-Session single pending, response idempotency, fork/rewind isolation, HTTP schema, TUI/ACP startup order, and Runtime mailbox reload. After process restarts, original tool side effects must not be automatically replayed; original tool calls must be closed, recovery results with provenance written, then pending-only turns launched via durable inbox. - -Real GPT must answer structured questions and actually call `Write` from pre-positioned pending Sessions respectively via Web response, ACP `session/load`, and TUI Runtime hook. Production DeepSeek GUI must display questions and pending badges on fresh load, automatically continue after answering, produce precise changed files, fresh tabs must not show questions again, and browser consoles must have no application errors. - -The completion window for Web response trajectories must be later than that model's Provider hard/idle watchdog; test timers must not preemptively classify still-running requests as recovery failures. Timeout diagnostics must be redacted and simultaneously include Bus terminal/stall events, Session task metadata, durable interaction/inbox/turn events, transcript tails, target files, and Runtime residency; regardless of success or failure, route controllers must shut down, proving active runs, Provider leases, and resident Runtimes are reclaimed. This window is only responsible for letting the runtime give authoritative terminal first, and must not add Provider or test retries. - -Session Permission Mode qualification must prove permission policies belong to durable Sessions, not process-globals or single UI Stores. Deterministic tests cover `default/autoEdit/yolo/plan`, latest update wins, legacy fallback, fork/task inheritance, illegal values fail closed, SessionStart Hook snapshots, pre-write persistence after Plan approval, zero execution on metadata failure, and explicit call overrides having higher priority than restored values. Web switching historical Sessions must restore corresponding modes, new tasks must reset to `autoEdit`, and cannot leak from previous `yolo` Sessions. - -Real GPT must set process default to `default`, persist `yolo` only in Session JSONL, then complete actual Write respectively via Web HTTP without mode, ACP `session/load`, headless `--resume`, and real TUI activation. All four trajectories must produce precise file bytes; Web/ACP must not show permission requests, and headless must not fail with "requires interactive confirmation". Production Web GUI must create full-access Sessions, fresh reload still shows full access; after clicking new tasks it must show auto-approval, then returning to original Sessions restores full access. Browser console must have no application errors. Real API trajectories must use unified 180-second Provider hard timeouts, zero retries, and have surface/test terminal windows later than Provider timeouts; Web failure paths must shut down route controllers, and cannot use partial 120-second budgets to preemptively misjudge long-tail responses or leave active runs. - -Session Reasoning Effort qualification must distinguish durable selection from Provider effective level. Deterministic tests cover `auto/off/minimal/low/medium/high/xhigh/max`, model capability projection, unsupported fail closed, active-turn rejection, Runtime service atomic replacement, metadata failure compensation rollback, fork/retry inheritance, and same-Session semantics across TUI/Web/ACP. - -Real GPT must complete `low + fast + low` requests through a local transparent proxy that does not record Authorization, destroy Runtime, update durable selection to `high + standard + high` then rebuild and complete the second request; the proxy must directly observe two `reasoning_effort` request values. production Web GUI must complete first round from Task Home, then subsequent rounds from Session Composer; fresh load must restore complete messages and Session settings. JSONL, API request body, and UI must be three-way consistent, and evidence files must not contain API keys. TUI Computer Use only counts as passing when the test bridge provides real raw TTY; Ink startup failures on non-raw stdin cannot impersonate UI qualification. - -Session Service Tier qualification must distinguish durable selection, effective tier, and Provider request value. Deterministic tests cover `auto/standard/fast/flex`, model capability projection, OpenAI `default/priority/flex`, Claude Fast Mode payload/beta headers, unsupported fail closed, active-turn rejection, model/effort/tier/verbosity/style setting group atomic replacement, metadata failure compensation rollback, fork/retry/subagent inheritance, and same-Session semantics across TUI/Web/ACP. - -Real GPT must complete `low + fast + low` requests through a local transparent proxy that does not record Authorization, destroy Runtime and restore `high + standard + high` from durable metadata then complete the second request; the proxy must directly observe both `priority` and `default` `service_tier` values, and no silent downgrading may occur. production Web GUI must complete first round from Task Home, then subsequent rounds from Session Composer; fresh load must restore complete messages and Session settings. Both upstream responses must be `200 text/event-stream`, JSONL, request body, and UI must be three-way consistent, and evidence files must not contain API keys. - -Session Response Verbosity qualification must distinguish durable selection, Provider effective value, and actual request projection. Deterministic tests cover `auto/low/medium/high`, GPT-5/Codex capability projection, Chat `verbosity`, Responses `text.verbosity`, Codex `textVerbosity`, payload hook merging, unsupported and fallback fail closed, active-turn rejection, model/effort/tier/verbosity/style setting group atomic replacement, metadata failure compensation rollback, fork/retry/subagent inheritance, and same-Session semantics across TUI/Web/ACP. - -Real GPT must complete `low + fast + low` requests through a local transparent proxy that does not record Authorization, destroy Runtime and restore `high + standard + high` from durable metadata then complete the second request; the proxy must directly observe both `low` and `high` `verbosity` values, and must not lose corresponding `reasoning_effort` or `service_tier`. production Web GUI must complete first round from Task Home, then subsequent rounds from Session Composer; fresh load must simultaneously restore complete messages and `high + standard + high`. Both upstream responses must be `200 text/event-stream`, JSONL, request body, and UI must be three-way consistent, and evidence files must not contain API keys. - -Session Communication Style qualification must prove it is orthogonal to Provider verbosity, and cannot escalate prompt permissions. Deterministic tests cover `auto/pragmatic/friendly/explanatory`, `auto` no injection, restricted section order and guards, style-only switching with zero Provider rebuild, active-turn rejection, model/effort/tier/verbosity/style setting group atomic persistence, metadata failure compensation rollback, JSONL/fork/retry, Task/Team/background/resume inheritance, and same-Session semantics across TUI/Web/ACP. Ordinary APIs must not accept arbitrary style prompts, file paths, or JSON. - -Trusted Custom Output Styles must additionally cover user/project/plugin namespacing, Folder Trust, active plugin policy, `.blade` same-namespace override of `.claude`, symlink/path escape, hidden Unicode, file/single-prompt/catalog bytes and count budgets, SHA-256 provenance, immutable Session snapshots, durable digest backfill/mismatch fail closed, and explicit built-in style restoration. Web/ACP catalogs may only expose `id/name/description/source/contentSha256`, not prompts or host paths. - -Real GPT must complete `pragmatic` requests through a local transparent proxy that does not record Authorization, destroy Runtime and restore `explanatory` from durable metadata then complete the second request; the proxy must directly observe corresponding style sections and permission guards in actual `system` or `developer` messages. production Web GUI must complete first round from Task Home, then subsequent rounds from Session Composer; fresh load must restore complete messages and `explanatory`. Both upstream responses must be `200 text/event-stream`, JSONL, request body, and UI must be three-way consistent, and evidence files must not contain API keys. - -Real GPT and production Web GUI qualification for custom styles must each complete at least one round of requests from project and plugin sources; transparent proxies directly observe corresponding markers located in restricted `communication_style` sections, and Session JSONL records namespaced IDs and digests. fresh load must restore custom selections; evidence directories must not contain credentials or absolute host paths beyond style prompt originals. - -MCP OAuth qualification must use real authorization servers and real Streamable HTTP MCP, covering RFC 9728/8414 discovery, dynamic client registration, state/PKCE, code exchange, short-lived access token `401` refresh, request replay, new client ledger recovery, logout, and callback/HTTP PID reclamation. Credential ledgers must verify 0600, atomic concurrency, symlink/mode/schema fail closed, endpoints/client/scopes must not cross wires; ordinary connects must have zero browser side effects, ACP/headless must not access host credentials or launch authorization. Real GPT must complete ToolSearch → OAuth MCP → Write. Production DeepSeek Web GUI must go through explicit Authorize/Continue authorization, Resume authorization after refresh, automatic reconnection, MCP and Write approvals, final markers, and fresh-tab restoration, with access/refresh tokens not appearing in traces. - -Workspace Agent resource isolation qualification uses two trusted projects, each containing native and plugin agents, skills, and commands. Deterministic tests must establish two workspace registries through real file loaders, copy Session snapshots then clear base tables, and prove Task/Skill/SlashCommand descriptions and execution still only contain owning-project resources. Real GPT must respectively call corresponding plugin commands in concurrent `SessionRuntime` and same-connection ACP dual-cwd Sessions, markers must not cross over; DeepSeek Flash/Pro must also complete code modifications and tests via production CLI `--agents -> Task`. production Web GUI must bind and trust A/B, execute corresponding SlashCommands respectively in independent worktrees, maintain respective markers after switching back, and fresh tab has no console errors. Task/Team foreground, background, and resume must all prove inheriting parent Session snapshots; `projectRoot` and execution `workspaceRoot` must not recouple. - -Trusted Contextual Project Rules qualification must cover hierarchy from Git root to target, `AGENTS.override.md` shadow, `CLAUDE.local.md`, `.claude/rules`, `.blade/rules`, `paths` glob, Folder Trust, symlink/path escape, hidden Unicode, file count and bytes budgets, Session snapshots, deduplication, compaction retention, and provenance mismatch fail closed. Conditional rules may only appear in the next provider request after first read-only reach; first write reach must block before side effects. JSONL may only save rule IDs, repository-relative paths, and SHA-256, not rule bodies or host absolute paths. - -Real API trajectories must delete on-disk rules after the first request, then load matching markers from Session snapshots via Read, prove non-matching globs never entered payloads, and complete rule-constrained code modifications and real tests. production Web GUI must display `Project Rules` activity cards, complete real responses, and fresh-tab restoration; CLI/headless and ACP must output the same security summary events. - -Session-owned User Shell Command qualification must cover 32 KiB input, UTF-8 fragmentation, ANSI cleaning, binary degradation, independent capture/stream budgets, async output ordering, exact workspace/env, durable resume, active-turn auxiliary steering, and whole process tree cancellation. TUI, Web, headless, and ACP must prove `!` does not create Agents; when ACP terminal is unavailable, it must fail closed and cannot fall back to Blade host shell. - -Real GPT must execute shell through a local transparent proxy that does not record Authorization, destroy Runtime, and restore the same Session; shell phase proxy request count must be 0, subsequent real provider payloads must directly contain `` and output markers. production DeepSeek Web GUI must create ordinary Sessions from Task Home, with network showing only `/shell` not task/message requests; then ordinary follow-up uses that marker via `/message`. fresh tab must restore one command card, two rounds of durable history, zero internal XML, and zero application console errors. TUI Computer Use only counts as passing when automation bridges can maintain real raw TTY focus and completely submit commands; otherwise they must rely on Ink rendering, real PTY, and process tree tests, and cannot count startup screenshots as complete TUI qualification. - -TUI Terminal Input qualification must cover ordinary multi-character stdin, fast characters within the same React batch, complete and split bracketed paste, CRLF, focus CSI, literal `[I`/`[O]`, TTY mode paired start/stop, and GracefulShutdown reset. Raw Ink tests must submit complete input to command handlers; production PTY must deliver bracketed payloads to newly built `dist/blade.js`. - -Real DeepSeek must directly observe complete pasted prompts in provider request bodies through a transparent proxy that does not record Authorization, and return expected markers composed of segmented tokens. Web GUI smoke must prove terminal-only changes did not affect Composer: multi-character `!` input still only goes through `/shell`, fresh tab restores one command card, internal XML and application console errors are both zero. Computer Use only counts as passing when tools can stably address independent terminal processes/windows; when bundle IDs point to old instances or focus may fall into user windows, UI operations must stop and switch to raw PTY evidence. - -Production Web bundle qualification must be calculated from fresh build artifacts, and when build callers set `NODE_ENV=test` they must still bundle production React runtime. CI must not read old `dist` to pass budgets; initial entry graphs, single entrypoints, and total JS gzip must all be re-verified after production builds. - -Plugin Marketplace qualification must use isolated HOME and local Marketplace snapshots. Deterministic tests cover `0600` strict ledgers, cross-process serial writes, Git `execFile` parameter boundaries, explicit source trust, symlink/path escape/credential URL/volume limits, digest tampering, failed update rollback, old root retention, and dependency deletion protection. Real GPT trajectories install v1 via ACP, Web refreshes and updates v2, active Sessions must continue calling v1, new Sessions must call v2, and after uninstallation subsequent Sessions no longer project commands. Production DeepSeek GUI must complete Marketplace addition, directory source selection, trusted installation, real SlashCommand, double-confirm update/uninstall, dependency blocking Marketplace deletion, and fresh-tab zero console errors. -Compatibility extensions must additionally prove same-Marketplace transitive dependencies commit atomically, ledgers have zero changes on cyclic or Blade/semver incompatibilities, runtime fixed-point degrades dependents, reverse dependencies cannot be uninstalled; source policies must cover host wildcard boundaries, local canonical roots, Marketplace identity, project tighten-only, `BLADE_PLUGIN_REQUIRE_SHA`, and checkout SHA mismatch. - -Workspace Model and Provider isolation qualification has two trusted projects configure the same channel ID and model config ID but use different endpoints. Deterministic tests modify project files and process-global catalogs after Session snapshot creation, and initial models and fallbacks must still resolve to their respective original endpoints. Real GPT qualification forwards the same real upstream through two local recording proxies, concurrent Sessions must each hit one proxy and successfully complete sampling; any request falling to the other project or later-modified fault endpoint fails. Task/Team foreground/background and resume, Prompt Hooks, Web dispatch/message, ACP new/load/fork must all inherit the same snapshot. Production Web GUI must switch between bound projects A/B, model buttons and expanded lists only show current project models, late old-workspace `/models` responses must not override new projects, switching back restores and console is empty. - -Real API project coverage includes production CLI trajectories: - -- Persisted file rollback: After model completes Read/Edit/Bash via production CLI and exits, host rebuilds snapshot manager with same session, verifying original paths and post-write hashes are still recoverable; after rollback file content, clean Git status, and test results must return to baseline; -- Single-file bugfix: Read, edit, run tests, and confirm diff scope; -- Multi-file API migration: Modify all production callers and run type-checks and tests; -- Temporary CLI settings: Load `--settings` file from startup directory, delete it before proxy forwards first-round requests, verify hidden system instructions have entered model context, and complete Read/Edit/Bash, independent tests, and diff verification; -- Layered project instructions: Inject `CLAUDE.md`, `AGENTS.md`, and `BLADE.md` by scope from Git root to CLI startup directory, prioritizing deeper rules within 32 KiB budget; transparent proxy verifies provenance, order, and override values in first-round model requests, and removes rule files before forwarding, proving final modifications do not depend on tool supplementary reading; -- Transient API recovery: Local proxy returns `503` on first model request, then forwards real API, CLI must retry within zero-output boundary and complete code modifications and tests; -- Context limit recovery: Local proxy returns `413 context_length_exceeded` on first model request, then transparently forwards real summaries and recovery requests; Headless must complete paired compaction lifecycle, persist replacement checkpoint, and have the second Runtime answer previous markers relying solely on that checkpoint; -- Tool crash recovery: Real DeepSeek first executes Write via production Headless, injects `tool_result` fsync failure after external file side effects have occurred; current run must fail closed before publishing results and second Provider requests. Second Runtime must write `sideEffectsUncertain` receipts for orphaned writes of terminated turns, real model resumes may only read and confirm existing files, and must not call Write/Edit again; -- Root turn auto-resume: After persisting original inbox messages, unclosed Writes, and flushed markers, release Session owner; new Runtime must first submit restart receipt, then restore original input from canonical JSONL model projection. Headless bare `--resume`, TUI `--resume`, and Web GUI SSE reconnect may each only execute one Read, Write/receipt/Read exactly once each, GUI results remain visible after reload and browser has no application/network errors. Final tokens must not appear in full in resume prompts or Read results; PTY final states must be jointly proven by exact inbox acknowledgement and corresponding `turn_completed`; -- Response commit recovery: After real DeepSeek produces final text, inject assistant message fsync failure; temporary content deltas may be observed, but current turn must be aborted, must not submit assistant or `turn_completed`. After cold start triggered by wake-up input, it must prioritize re-executing original durable inbox, second real response successfully commits without leaking underlying I/O errors; -- turn finalization recovery: After real DeepSeek's final assistant and final receipt are committed, inject process exit before inbox ack/terminal; cold start must first atomically supplement `inbox_acknowledged + turn_completed` and reload sidecars, then only process new input. Old inputs must not request Providers again, final history must have two completed, zero aborted turns; -- Goal finalization handoff: Exit when final assistant's host receipt is committed but Goal sidecar is still `verifying/pass`. New Runtime must idempotently supplement `complete` with exact goal ID, attempt, verifier Session, evidence digest, and revision; Headless playback, raw PTY, production Web GUI, and ACP `session/load` must not initiate Provider requests for old Goals. Then send new prompts from same entrypoint and complete real Flash/Pro responses through transparent proxy, proving work can continue after recovery; -- Goal verification attempt is a monotonically increasing recovery sequence number, not a fixed value. After FAIL/PARTIAL, format corrections, or evidence invalidation it may enter attempt 2 and subsequent attempts; release trajectories must require final Goal to be `complete`, with current attempt having fresh `PASS`, verifier Session ID, and SHA-256 evidence, but must not lock legitimate positive-integer attempts to `1`; -- Plan mode recovery: Restore sessions across two CLI processes and complete modifications; -- Mode boundary recovery: Deliberately call ExitPlanMode once in Yolo, runtime must return `validation_error`, model then continues Write/Bash, proving expired planning states cannot terminate already-approved work; -- Failure recovery: First reproduce test failures, then modify, finally verify pass; -- Timeout recovery: Reclaim complete process tree then continue tool loop, and confirm no descendant processes remain; -- Background shell hard-crash recovery: Independent Blade owner starts TERM-ignoring detached process group and hard-exits before dispose; new Runtime must reclaim old trees through durable leases and startup identities, PID reuse/identity mismatch and ownership changes within TERM grace periods must not kill mistakenly, damaged leases must fail closed, when lease commit fails gate wrappers must not execute user commands, sidecars must not contain commands, environments, outputs, or credentials; -- Foreground shell hard-crash recovery: Real DeepSeek must initiate foreground Bash with delayed writes respectively from parent and subagent; host `SIGKILL`s independent Blade owner before tool result, after new Runtime obtains parent/child Session leases it must first reclaim corresponding foreground trees, then close orphan Bash tool receipts. Delayed files must not appear, lease commit/gate release failures must have zero execution, PID identity mismatch must not kill mistakenly, damaged sidecars must block recovery, sidecars and CLI output must not contain commands, environments, outputs, or API keys. Subagent controls must not rely on fixed sleeps: TERM-ignoring descendants must wait for host gates, hosts only open gates after root PID exits and durable leases are deleted; if any process tree residue exists, forbidden side effects must deterministically appear and fail qualification; -- leaderless process group: Parent real API trajectories must first prove shell/gate root PID has exited, TERM-ignoring descendants still alive, then hard-kill Blade owner. Linux/macOS reapers must independently probe negative PGIDs and complete TERM/KILL before delayed side effects; root PID reuse during grace prohibits KILL and retains leases. Normal foreground/background/ACP local close must also verify all redirected descendants are reclaimed before terminal results; Windows continues to verify live-root `taskkill /T`, and does not claim POSIX PGID semantics; -- Session exit reclamation: After model starts background processes and ends CLI normally, verify runtime dispose waits for entire process tree to terminate; -- Interrupt recovery: Real signals interrupt active tool calls, persist one model-visible interrupt boundary, then safely recover via second CLI; -- Session exclusivity: Active runtime rejects second same-session CLI and does not persist its input, allows recovery and continues verification after owner exits; -- Transcript truncation recovery: Create uncommitted half-lines at session JSONL tail, complete Write/Bash tasks after recovery, and verify complete history line by line after repair; -- Context compaction continuation: Restricted context windows trigger one automatic compaction after Read; when transparent proxy pauses real summary requests, stdout must have already emitted `compacting: started` in real time, then maintain pure JSONL, flush automatic summaries, and execute Write after `compacting: completed`; -- Web surface: Submit tasks through production HTTP session routes and consume real SSE, verifying code modifications, host tests, canonical tool success, and `compaction.started` / `compaction.completed` visible in order before resumed Write; -- Structured user questions: Web must still emit `question.required` in `yolo`, SSE disconnect/reconnect only replays currently unresolved questions with unchanged IDs, continues Write/Bash after submitting structured answers; ACP must also collect single-choice answers through standard permission options in auto-approve mode. When ACP protocol cannot faithfully express multi-select it fails closed, and must not silently downgrade to single-select; -- Blocking interaction cancellation: TUI settles all confirmations before aborting turns; Web abort must immediately invalidate pending permissions/questions, wait for old turns to release runtime before returning idle, late answers return `404`; ACP cancel notifications must independently interrupt unresponsive reverse requests. Both Flash and Pro must cancel questions then continue completing Write/Bash in the same session, and cancelled final states must not be overwritten by completed/error; -- Permission scopes: `once`, `session`, `project` must be distinct contracts. TUI/Web explicitly display session-level and project-level choices; session approval only enters current runtime cache, cannot be written to disk, is reused in the same runtime's second independent turn, and is re-asked in new sessions; project approval is written to target workspace's `.blade/settings.local.json`, cannot land in server startup directories or leak to other projects, and new Web/ACP sessions must automatically load it. ACP `allow_always` / `reject_always` maps to real project-level persistent rules; both Flash and Pro verify through real Bash trajectories; -- Interactive background Shell: `WriteStdin` can only operate background Bash owned by current session, wait for write completion and can explicitly close stdin; cross-session, exited processes, and missing sessions must fail closed. Flash and Pro in TUI, Web, ACP must all complete `Bash(background) -> WriteStdin(close) -> TaskOutput(block)`, with host verifying actual files and three tool events; -- Bounded background output: After background Bash stdout/stderr each exceed 1 MiB, only recent output may be retained and earlier omitted bytes precisely reported; Flash and Pro in TUI, Web, ACP must all complete `Bash(background) -> TaskOutput(block) -> Write`, verifying tail markers, `output_truncated`, stream omitted bytes, shared display summaries, and host-proof files; -- Bounded foreground output: Host pre-written script outputs `1 MiB + 64 KiB` to each stdout/stderr, placing omitted-prefix sentinels within the first 4 KiB of each stream and independent nonce tails at ends. Flash/Pro must each call foreground Bash only once across four entrypoints: Headless, Chromium Web, raw PTY TUI, ACP SDK. Headless returns `false` on first stdout/stderr write and delays drain, during which no second raw write may appear; ACP delays each update and maximum in-flight must be 1; PTY host must resume final output after pausing reader; Web must reload mid-run, cursor reconnect, and retain same tool card on fresh terminal load. Web/PTY verify dual-stream total/retained/omitted; ACP verifies merged stdout, zero stderr, and `terminal_output_merged=true`. All entrypoints must retain dual tails, hide dual sentinels and API keys, and zero Chromium/page/SSE, PTY, ACP terminal, process identity, foreground lease, port, and temporary roots. When virtual lists remount completed cards, Web must atomically obtain durable `toolCallId`, re-expand when tool groups return to collapsed state and trigger real click handlers, then assert `aria-expanded=true` and bounded output, and must not treat remounts or layout actionability jitter between card/toggle two locator awaits as runtime failures. - Goal finalization fresh-load must simultaneously verify persisted Goal `complete` and DOM `complete` within the complete bounded qualification budget, reporting both sides' states and browser faults on failure, and cannot use shorter hydration sub-cutoffs to replace end-to-end budgets. - The foreground gate-release failure control must subscribe to stdout before release, inject errors after observing actual bytes exceeding retained budgets, and must not use fixed sleeps assuming output has arrived. Positive marker evidence for raw PTY must be monotonically latched; resize or subsequent redraws can only add evidence, not revoke observed facts from bounded tails. The source contract precisely enumerates 13 PTY runners: `backgroundSubagentCompletion`, `browserTool`, `foregroundBoundedOutput`, `foregroundCommandHandoff`, `foregroundProviderRecovery`, `goalFinalization`, `gracefulShutdown`, `rootTurnAutoResume`, `sessionRuntimeResidency`, `subagentResultAdoption`, `toolAdmission`, `tui`, `weightedProviderAdmission`; new runners must update inventory and explicitly complete marker-latching audits. Only facts explicitly required to remain visible after resize may be re-proven from new PTY data after resize; historical matches cannot be reused. Provider queues, child markers, and parent finals for background completion must consume the same bounded evidence deadline, and shorter first-stage cutoffs cannot be used to misjudge slow first responses. Computer Use serves only as supplementary visual evidence when hosts provide stable desktop bridges, and cannot replace automated raw PTY and protocol assertions; -- Cross-surface session branch: TUI `/branch` atomically switches to persisted child sessions; Web creates and selects child sessions via HTTP fork routes, active turns return `409`; ACP `/branch` returns child session IDs loadable by standard `session/load`. Both Flash and Pro must, after deleting original markers, continue Write/Bash relying solely on inherited Read results, and prove parent transcripts unchanged; -- durable turn rewind: Both Flash and Pro must first produce file checkpoints via real model Read/Edit/Read, then recover respectively from Runtime, TUI hooks, Web HTTP/SSE, and ACP `/rewind` entrypoints. Runtime/TUI/Web verify code returns to baseline, effective conversation is removed and JSONL retains `session_rewound`; ACP verifies Agent rebuild after conversation-only rewind, subsequent prompts use only projected history. Web must also verify button disabled states, checkpoint dialogs, code restore toggles, post-submit message lists, and disk effects through real browsers. Actual results are recorded in [durable rewind evidence ledger](/en/testing/durable-rewind-evidence.md); -- TUI runtime lifecycle: Clean up after completing real model turns via `useAgent`, re-acquire runtime leases with same session IDs, proving exit paths release Agent, background resources, and session ownership; -- ACP session/load: Create and destroy sessions through real ACP SDK NDJSON connections, load persisted history after deleting original marker files, replay user/assistant messages before responses, and continue Write/Bash relying solely on recovered context; MCP servers passed by clients use session-private registries, independently reclaiming on initialization failure or exit; -- ACP session model switch: After session initializes with Flash, switch to Pro via real `session/set_model`, transparent proxy must only observe subsequent sampling requests for Pro; atomically update providers and context windows during switches, reclaim old providers, and complete Read, source modifications, Bash, independent tests, and Git diff verification; -- Single-run Subagent: Inject model-exclusive rules existing only in subagent system prompts via `--agents`, main agent only exposes Task; both Flash and Pro must delegate to custom agents, completing Read/Edit/Bash, independent tests, precise file scopes, and pure JSONL verification; -- durable Subagent resume: Both Flash and Pro must first generate context existing only in child transcripts via real Tasks, then recover from four entrypoints: Runtime, TUI, Web, and ACP. follow-up prompts must not contain target values; child must give correct results relying solely on recovered history, proving source sidecars unchanged, child IDs newly created, lineage depth monotonically increasing, frozen models/permissions in effect, and no key leakage. Web must additionally verify depth 1 → 2 via real browsers, 2 → 3 after refresh, disabled states, and zero console errors. Actual results are recorded in [durable subagent resume evidence ledger](/en/testing/durable-subagent-resume-evidence.md); -- durable Subagent crash recovery: After real child Provider streams' first content delta is held by transparent proxy, host SIGKILLs Blade owner; second Runtime must first close child turn/tool receipts and rebuild sidecar history from JSONL, then resume with new immutable child IDs. follow-up must not contain original tokens, recovery failures must not allow Web/ACP to initiate false resumes; -- durable completed-Subagent result adoption: Both Flash and Pro must first generate markers existing only in child results via real foreground Tasks, then retain active parent turns, durable inboxes, and orphan parent Task calls, releasing Runtime before parent `tool_result` commit. Headless, raw PTY TUI, production Chromium Web GUI, and ACP `session/load` must adopt results from the same child sidecar, and must not start children again. Each cell verifies resumed Provider requests contain child-only markers, adopted result/parent abort/inbox ACK/parent final each occur once, child sidecar bytes unchanged, compound owner and lineage unique, `sideEffectsUncertain=false`, Web live/reload visibility, process/port/temporary root cleanup, and no credential leakage; -- durable background-Subagent completion wake-up: Both Flash and Pro must have real parents call `Task(run_in_background=true)`, after Task returns running parent continues independent Read, with zero `TaskOutput` throughout. child obtains markers not present in parent input through Read; terminal sidecars, hidden canonical receipts, and durable inboxes must automatically wake parents. Headless, raw PTY TUI, production Chromium Web GUI, and ACP `session/load` each verify terminal ref/inbox ACK/parent final, child and lineage unique, sidecar bytes stable, no pseudo-user messages, Web live/reload consistency, and resource/credential cleanup; -- Output protocol, tool calls, error events, and key leakage checks. - -### Durable Large-Prompt Offload Trajectory - -This capability proves that large user requests do not expand without bound in -the first Provider call while the model can still retrieve complete instructions: - -1. The fixture must exceed the 32 KiB inline threshold and place a unique hidden - authority in a middle region excluded from both preview ends. The final value - can only be assembled from separated tokens in that authority; the first - request cannot contain the hidden marker or complete final value. -2. A transparent proxy must directly prove that first-request user content is at - most 32 KiB, contains one valid opaque artifact ID, advertises - `ReadPromptArtifact`, and does not yet contain the hidden marker. -3. A later request may contain the hidden marker only in the corresponding tool - result after the assistant calls `ReadPromptArtifact` with the same ID. The - marker cannot occur in user, assistant, or system content. The model must read - through `[End of prompt artifact]` and return the exact final value. -4. JSONL must persist the same `userPromptArtifact` reference, complete - ReadPromptArtifact call/result trace, and exact final. Cold projection, PTY - resume, Web reload, and ACP `session/load` cannot issue another Provider - request for completed input. -5. Deterministic gates must additionally cover UTF-8 pagination, hash/mode/ - tamper failures, concurrent artifact-count quota, multimodal ordering, fork - reference copying, Session deletion cleanup, and the 1,000,000-character/ - 4 MiB input limits. - -The required matrix fixedly includes `deepseek-v4-flash` and -`deepseek-v4-pro`; each model must pass Headless, real raw PTY TUI, production -Chromium Web GUI, and ACP, totaling eight cells. Tests must reclaim -browser/page/SSE, PTY, ACP connection, server, port, Session lease, artifacts, -and temporary roots. API keys cannot enter transcripts, pages, terminals, ACP -updates, diagnostics, or structured proxy evidence. - -### Durable Subagent resume Trajectory - -This capability must cover the same immutable lineage contract: - -1. **Runtime**: foreground root first persists complete ChatContext, destroys and rebuilds manager/runtime then resumes as child, then resumes as grandchild. Each edge uses new IDs, source sidecar bytes and final states unchanged, `rootAgentId` stable, `resumeDepth` monotonically increasing. -2. **CLI/TUI**: Real `useAgent` owner continues ended agents via `/tasks resume`, updates subagent progress, but does not release parent Runtime. After exit it must normally release Runtime leases. -3. **Web**: Publishes `subagent.start/update/tool/complete` via exact `sessionId + projectPath` GET/POST routes and SSE. Message cards support follow-up, running polling, recoverable errors, and refresh reconstruction; must not mis-select ancestors as latest descendants. -4. **ACP**: Executes `/tasks resume` through real ACP SDK lifecycle, exposing new child IDs, states, and results using standard `tool_call` and `tool_call_update`. Private text events cannot replace tool protocols. - -sidecars must use atomic writes, `fsync`, `0600` file permissions, and `0700` directory permissions. Public Web schemas only return states, lineages, results, and statistics, not prompts, messages, config snapshots, workspaces, owner PIDs, or provider credentials. Cross-workspace, type conflicts, running sources, active parent turns, and durable pending inputs must all fail closed. - -The required matrix fixedly includes `deepseek-v4-flash` and `deepseek-v4-pro`. Each model must pass four real product entrypoints: Runtime, TUI, Web, ACP; worktree resume separately verifies new child IDs continue using source lease owners and retain failed or modified worktrees. - -### Durable completed-Subagent result adoption Trajectory - -This capability verifies cross-storage commit gaps between child terminals and parent Task results: - -1. Fixtures must first run foreground Tasks through real Providers, requiring models to generate child markers not present in parent input; then only persist parent `tool_call`, without committing results. -2. Runtime may only adopt `completed`/`failed` results from exact compound owner durable sidecars; any mismatch in child ID, description, explicit type, resume lineage, state, or bounded results must fall back to generic uncertain receipts. -3. Adoption batches must write one `tool_result`, one terminal `subtask_ref`, and one `turn_aborted(process_restart)` per original tool-call/message identity; second starts must not duplicate writes. -4. Headless, raw PTY TUI, production Chromium Web GUI, and ACP `session/load` must consume standard LoopEvents. Web must additionally locate original cards by durable child Session IDs, verifying live terminal summaries, parent finals, same state after reload, and zero browser faults. -5. Each surface's resumed Provider requests must contain child-only markers; child sidecar bytes, child counts, and lineages remain unchanged, proving no duplicate Task execution. - -The required matrix fixedly includes `deepseek-v4-flash` and `deepseek-v4-pro`, four surfaces totaling eight cells. This trajectory is release-blocking real API qualification, and must not be substituted by mocks, HTTP 200, JSONL-only checks, or incidental visibility after refresh. - -### Durable background-Subagent completion wake-up Trajectory - -This capability verifies background Tasks can drive long tasks without relying on model polling: - -1. Parent input may only contain marker filenames, not child markers. Real parents must first call `Task(run_in_background=true)` once, after receiving running results then complete one independent Read; `TaskOutput` call counts in parent transcripts must be zero. -2. child must use real Providers and Read tools to obtain markers and commit terminal sidecars. Runtime only accepts exact compound owners, `background=true`, canonical child IDs, type/description/resume lineages, and structurally valid bounded terminal results. -3. Parent must converge in order: `child sidecar fsync → hidden receipt + terminal subtask_ref → durable inbox → model consumption → inbox ACK`. deterministic inbox IDs, receipts, terminal refs, ACKs, and child lineages must all occur exactly once; cold starts must not duplicate notifications. -4. Headless must maintain Agent streams during child execution; raw PTY TUI, Web, and ACP must automatically continue parents without requiring manual input. ACP must not generate marker `user_message_chunk`; TUI/Web must not render pseudo-user messages. -5. production Chromium Web must verify live terminal cards, parent finals, same child ID/status/summary after reload, terminal sidecar bytes unchanged, and zero browser faults. Streaming Tasks must obtain canonical child IDs before persistence; when children complete early, late running results must not degrade live or fresh-load cards. - -The required matrix fixedly includes `deepseek-v4-flash` and `deepseek-v4-pro`, four surfaces totaling eight cells. After testing, browser/page/SSE, PTY, ACP connections, server/process trees, ports, temporary storage/workspace/trust roots must be reclaimed, proving API keys do not enter JSONL, sidecars, DOM, PTY, ACP updates, diagnostics, or recorded Provider bodies. - -### Bounded coordinated shutdown Trajectory - -This capability verifies normal process shutdown does not leave active turns for next cold start to repair: - -1. Each cell must first call one real foreground Bash through real Providers; hosts may only send corresponding production `SIGTERM` after child PID files exist and processes are still active. -2. Agent/Session owners must first close new work entrypoints, then abort active Provider/tool paths, wait for one `turn_aborted(cause="cancelled")` commit before releasing Runtime, Session leases, and transports. Active tools' `shouldExitLoop` must not preemptively bypass abort terminals; the same interrupted turn must not show `turn_completed` or second terminal records. -3. foreground children must ignore TERM and schedule delayed forbidden side effects; shutdown must reclaim complete process trees and durable leases, and after control windows forbidden files still do not exist. -4. Original durable inputs must remain recoverable. Same-Session production Headless resumes must not call Bash again; transparent proxies must observe exactly-one resume request directly containing complete `` system markers and original request markers, then produce non-empty finals and add one `turn_completed`. Whether models verbatim retell markers must not be treated as sole evidence of whether runtimes recovered those markers. -5. Headless, real ACP stdio + terminal, raw PTY TUI, and production Chromium Web GUI enter from independent processes respectively. Web must submit through real composers; closing viewers must not equate to server shutdown. Each cell reclaims browser/page, PTY, ACP connections, servers, ports, process trees, Session leases, temporary storage/workspace/trust roots, and full-scans credential absence. - -The required matrix fixedly includes `deepseek-v4-flash` and `deepseek-v4-pro`, four surfaces totaling eight cells. This trajectory is release-blocking real API qualification, and must not be substituted by mock signals, directly calling `SessionRuntime.dispose()`, only checking process exit codes, or cold `process_restart` repair. - -### Session discovery and durable fork Trajectory - -Session discovery and durable fork qualification must cover four mutually independent production entrypoints: one internal Runtime boundary, plus CLI/TUI, Web, ACP three user-visible integration surfaces. Each trajectory must enter from corresponding entrypoints, and must not replace product boundaries by directly calling models or only verifying proxy requests: - -1. **Runtime**: Creates children through real `SessionRuntime`, `Agent`, and public `SessionService.forkSession()`. Parent first executes Read, child executes Write/Bash relying solely on inherited Read results. Evidence includes precise file effects, parent JSONL bytes unchanged, parent/root lineages, child independent appends, runtime cleanup, and no secrets in evidence. -2. **CLI/TUI**: Executes initial prompts, `/fork `, and child prompts through real `useCommandHandler.executeCommand()`. Evidence includes slash-command routing, child activation, precise file effects, parent transcripts unchanged, lineages, active turn rejection semantics for `/fork`, and no secrets in evidence. -3. **Web**: Completes through production HTTP session routes and SSE; deterministic Web tests separately cover Sidebar actions and store activation. - - completed parent: After fork child SSE is ready and activates with `sessionId + projectPath`, child produces precise file effects relying solely on inherited history; parent JSONL remains unchanged and lineages correct. - - active parent: Forks committed stable JSONL prefixes while real provider requests remain active; parents are not cancelled and continue appending, children do not contain post-boundary parent content and run independently. - Both sub-items check compound workspace identities, structured HTTP/SSE evidence, resource cleanup, and secret absence. -4. **ACP**: Executes `session/list`, `session/fork` through real ACP SDK NDJSON codecs and dispatchers, then prompts returned children directly without calling `session/load` or replaying history. Evidence includes discovery metadata, precise file effects, parent immutability, lineages, child independent appends, session cleanup, and no secrets in evidence. - -The required matrix for this trajectory group fixedly includes DeepSeek Flash and Pro. If Claude, GPT, or domestic providers are explicitly configured, their configured models must also run four production entrypoints; missing required Flash/Pro fails closed. API keys may only be projected from restricted local storage to subprocess credential slots, and are not written to project configuration, command records, logs, snapshots, or raw request headers. Actual commands, model sets, exit codes, and rerun facts are recorded in [session discovery and durable fork evidence ledger](/en/testing/session-discovery-fork-evidence.md). - -Receiving only model text or HTTP 200 does not count as passing. Each trajectory must prove expected file or persisted side effects, structured events, and process exit status; trajectories involving code modifications must also record `git diff --name-only` and test or type-check exit codes. Session fork trajectories instead verify parent/child JSONL, lineages, precise fixture file content, and resource cleanup, without fabricating Git diff evidence. - -### Native Browser Tool Trajectory - -Native Browser Tool release qualification has two layers: +```bash +REAL_API_RELEASE_MATRIX=1 bun run test:real-api +``` -1. The keyless Chromium integration uses the real pinned Chromium and local - loopback fixtures. It covers navigation, ARIA snapshot/ref interaction, forms, - waits, console/network/find/screenshot inspection, history/reload, page - management, Context reset, cross-origin navigation rejection, cross-origin - iframe rejection, and Cookie isolation between two Sessions. -2. The real API matrix runs DeepSeek V4 Flash and Pro through Headless, raw PTY TUI, - production Web GUI, and ACP, for eight cells. +Deleting or renaming an inventory file makes the runner fail before Vitest +starts. New real API trajectories must be added explicitly to the relevant +inventory and source-contract test so directory changes cannot silently shrink +the gate. -Every real API cell must load all six deferred Browser tools through one exact -`ToolSearch`, then complete Navigate, explicit Snapshot, Interact, Wait, Inspect, -and Page management before returning a nonce that exists only after form submission. -Framework retry is fixed at `0`. A DOM-induced `browser_snapshot_stale` is accepted -only when the same trace subsequently completes a fresh Snapshot and successful -Interact. Any other Browser error blocks release. +Ordinary `test:all` and CI do not issue paid requests. Domestic model channels +are excluded from the release gate unless +`REAL_API_INCLUDE_OPTIONAL_PROVIDERS=1` is explicitly set. -The production Web cell runs an outer Chromium driving Blade UI and an inner -Chromium owned by Browser Tool in the server. Every cell must reclaim -BrowserContexts, pages, browser processes, servers, ports, PTYs, ACP connections, -Session leases, and temporary roots, and must prove the API key never reaches DOM, -PTY, ACP updates, tool results, metadata, transcripts, or diagnostics. +## Qualification Evidence -The trajectory is fixed in `realApiQualification.files`; mocks, HTTP-only success, -model text alone, or keyless-only tests cannot replace it. Chromium preflight uses -`blade browser status`. When missing, only an explicit `blade browser install` may -download it; qualification itself never installs a browser. +Each independent patch retains at least: -## Qualification Evidence +- the frozen candidate SHA, version, and date; +- command, result, and exit code for `bun run qualify:local`; +- command, per-file results, and exit code for `bun run qualify:production`; +- browser preflight, process/lease/terminal/port/temp-root cleanup, and + credential-absence assertions; +- bounded redacted diagnostics and cleanup results for the first failing cell; +- `git diff --check`, build, type-check, and lint results. -Each independent patch retains at minimum the following evidence: - -- Frozen Qualified candidate complete SHA, patch version, and date; -- Complete command and exit code for `bun run qualify:local`; -- Complete command for `bun run qualify:production`, per-item results and exit code for that patch's required Flash/Pro × production surface matrix; -- Host assertions for browser preflight, process/lease/terminal/port/temp-root cleanup, omitted sentinels, and credential absence; -- On failure, record first failed cell, redacted bounded tail, cleanup results, and rerun facts; only Provider transients with unchanged sources may trigger full reruns, and skipping tests cannot substitute for passing; -- Commands and exit codes for `git diff --check`, build, type-check, lint; -- Evidence may only be created after candidate code is frozen and passes real qualification. The only difference between candidate SHA and tag HEAD must be complete evidence files; files must not contain `TBD`, `TODO`, `NOT RUN`, or pre-filled `PASS`. - -The real API gate incurs costs, so it is not implicitly triggered by `test:all` or ordinary CI unit gates; release candidates, cross-provider changes, and Agent runtime core changes must run it explicitly. - -Current production qualification covers desktop TUI, CLI/headless, Web, and ACP. Mobile has no clear usage scenarios at this time and is not included in implementation or testing scope. - -## Active turn activity - -The release-blocking active-turn matrix runs DeepSeek Flash/Pro across Headless, real -ACP stdio, raw PTY TUI, and production Chromium Web for eight cells. Every cell -requires framework retry `0`, model `maxRetries=0`, one Bash call, ordered -thinking/tool/responding/clear activity, and an exact final response. Web must reload -while a host barrier keeps the tool active and restore activity from SSE -`connected.turnActivity`. Release the tool only after asserting that reconnect snapshot, -and collect evidence only after both independent SSE probes receive terminal clear; -the activity strip disappearing does not prove delivery to the probes. PTY must observe -the state in a real terminal capture rather than reading an internal store. - -The ordinary path requires two Provider requests. Exactly one empty-final correction is -allowed only when the second complete SSE response ends with `stop`, contains no text, -and emits no tool deltas. The third request must append only the exact correction to -the existing messages. That correction must already be persisted with -`clientVisible=false` and `emptyFinalCorrection=true`, parented to the sole successful -Bash result; terminal accounting must still show one tool execution. Nonempty, truncated, -or incomplete responses, duplicate requests, additional tools, and a fourth request do -not qualify. The proxy retains only bounded counts and completion categories, never -response text, reasoning, or tool arguments, and forwards the original streaming bytes. - -A separate eight-cell recovery matrix sets a fixed text prompt and `stop` only on the -second upstream request so the real Provider produces an empty response. It never -replaces responses or modifies subsequent requests. Each cell requires one durable -correction, an exact final answer, and one real Bash side effect. A negative control -without injection must fail for missing correction evidence, not pass through framework -retries. - -Commands, per-cell timings, privacy, and cleanup are in -[Active Turn Activity Qualification Evidence](./turn-activity-surface-evidence.md). +Only a Provider transient with unchanged source permits a complete rerun. +Skipped tests, model prose, and prefilled `PASS` text are not qualification +evidence. diff --git a/docs/reference/workspace-agent-resources.md b/docs/reference/workspace-agent-resources.md index 5f9c60ba2..bd0041a25 100644 --- a/docs/reference/workspace-agent-resources.md +++ b/docs/reference/workspace-agent-resources.md @@ -81,8 +81,9 @@ workspace registry,证明 Session 工具仍保持原快照且对方资源零 真实 API 资格包括: -- GPT 双 `SessionRuntime` 并发调用各自 plugin command; -- GPT 在同一 ACP connection 的双 cwd Session 中分别调用各自 plugin command; +- DeepSeek Flash 双 `SessionRuntime` 并发调用各自 plugin command; +- DeepSeek Flash 在同一 ACP connection 的双 cwd Session 中分别调用各自 plugin + command; - DeepSeek Flash/Pro 通过生产 CLI `--agents -> Task` 完成 Read/Edit/Bash; - production Web GUI 绑定并信任 A/B 两项目,在独立 worktree 中分别调用 `plugin-a:reveal` 与 `plugin-b:reveal`,回切后投影保持独立,fresh tab 无 console diff --git a/docs/testing/qualification.md b/docs/testing/qualification.md index 5ef1f8b95..c69874f59 100644 --- a/docs/testing/qualification.md +++ b/docs/testing/qualification.md @@ -1,35 +1,22 @@ # Blade Code 测试手段与生产准出 -Blade Code 将确定性回归与付费模型验证分成两道门禁。两道门禁都必须通过,才可以把一个功能 patch 标记为生产就绪。 +Blade Code 将确定性回归与付费模型验证分成两道门禁。两道门禁都必须通过, +功能 patch 才能标记为生产就绪。 ## 受控代码评测 -`bun run benchmark:repo -- --model <已配置模型ID>` 现在运行 `controlled-coding-v2`: -只读诊断、单文件修复、跨模块 API 迁移三个固定小型任务。它不是大型真实仓库基准, -通过六格(DeepSeek Flash/Pro × 三任务)不能证明已达到其他 coding agent 的整体能力。 -运行前先执行 `bun run build:cli`,评测使用 production dist、Node 与 npm,以及已配置的 -API-key 模型;不会安装依赖或创建 worktree。 +`bun run benchmark:repo -- --model <已配置模型ID>` 运行 +`controlled-coding-v2`:只读诊断、单文件修复和跨模块 API 迁移三个固定小型任务。 +它使用 production dist、Node、npm 和已配置的 API-key 模型,不安装依赖,也不创建 +worktree。 -每个任务拥有独立临时项目、HOME 与 Session 存储,禁用 MCP、LSP、hooks 和插件。 -分析题要求成功读取文件、精确结构化结论和文件清单/内容不变;修复题要求限定文件被修改、 -最后一次修改之后的 `npm test` 成功,并由宿主把源码字节复制到独立目录执行边界样例、 -比对返回值。修改测试或 package.json、增加文件/目录、符号链接、缺失工具证据、仅输出 -成功文本或仅 `exit(0)` 都不能通过。验证子进程有 5 秒预算,每个 Agent 有 240 秒与 -16 轮预算,输出限制为 4 MiB;模型额外传输重试为零。 +每个任务拥有独立临时项目、HOME 和 Session 存储,并禁用 MCP、LSP、hooks 与插件。 +宿主检查真实文件读取、限定路径修改、修改后的 `npm test`、独立目录中的边界样例及 +返回值。修改测试或 `package.json`、增加额外文件、缺失工具证据、仅输出成功文本或仅 +`exit(0)` 都不能通过。 -临时目录隔离不是 OS 安全沙箱:候选代码与工具仍以当前用户身份运行,不能用于敌对代码, -也不证明候选程序对所有输入正确。宿主只向子进程提供所需环境,评分不保存密钥、模型正文 -或原始工具输出。默认结果写入 `.blade/benchmarks/controlled-coding-v2-history.json`, -包含源码 SHA-256、改动路径、各项校验、累计请求 Token、成功读取数和 Agent 耗时。 -可用 `--history-path` 改位置;旧 v1 关键词评分文件不会被覆盖或合并。任一任务不通过时 -命令非零退出。历史文件不支持多个评测进程并发写入,请为并发运行使用不同路径。 - -跨界面迁移验收另外覆盖 DeepSeek Flash/Pro × production Chromium、Vite development -Chromium、raw PTY TUI、ACP 八格。每格必须实际修改两个源码文件、保持测试与 package.json -不变,并在修改结果提交后执行精确的 `npm test`。宿主按 durable tool-result 的提交顺序 -校验,而不是按调用开始顺序;最后将源码复制到独立验证目录检查行为。GUI 逐一核对工具 -ID 的可见结果,ACP 核对终态更新与 durable tool ID,TUI 等待精确回复持久化后退出。 -这些是受控任务的集成验收,不代表大型仓库成功率或原生桌面 Computer Use 测试。 +默认结果写入 `.blade/benchmarks/controlled-coding-v2-history.json`。该评测不是大型 +真实仓库基准,也不证明与其他 coding agent 的整体能力等价。 ## 本地门禁 @@ -39,1227 +26,116 @@ ID 的可见结果,ACP 核对终态更新与 durable tool ID,TUI 等待精 bun run qualify:local ``` -命令会按固定顺序执行 14 个检查: +命令按顺序执行 14 个检查: -1. `type-check` -2. `format:check` -3. `lint` +1. TypeScript 类型检查 +2. 格式检查 +3. lint 4. 单元测试 5. 集成测试 -6. CLI 集成测试 -7. headless/runtime 核心回归 +6. CLI 测试 +7. Headless/runtime 核心回归 8. E2E 9. snapshot 10. 安全测试 -11. 当前源码构建 +11. production build 12. Web 测试 13. Web 类型检查 14. 性能回归 -每一步都在独立子进程中执行。第一步非零退出会立即停止,后续步骤不会被计为通过。该门禁不访问付费模型,也不依赖 `~/.blade/config.json`。 - -`scripts/test.js` 的构建与测试命令在取消信号已触发时不启动子进程或 watchdog;运行中 -取消仍等待受管进程树退出,原超时预算不变。 - -Vitest setup 为每个 test-file lifecycle 创建唯一的临时 -`BLADE_STORAGE_ROOT`,并在 teardown 时同步删除。调用方显式传入的 -`BLADE_STORAGE_ROOT` 始终由调用方管理,测试 harness 不会删除。 - -real-api setup 仅在 `REAL_API_TEST=1` 时加载凭据配置、模型目录与应用 store; -未启用付费测试时仍建立并回收隔离存储,免凭据回归仍运行。本地免凭据 real-api -文件在现有四 worker 上限内并行加载,并保留独立进程及文件隔离;付费矩阵和 CI -仍为单 worker 串行。测试集合、重试规则及进程超时预算不变。 +任一步骤非零退出都会立即停止。该门禁不访问付费模型。 -GitHub `Quality Gate` 在 build 前重复执行全仓 format check 与 CLI lint,并由 workflow -source contract 固定 install → format → lint → build 顺序。root、CLI 与 Web 使用同一 -精确 Biome 版本,避免 workspace binary 解析差异让本地门禁和 CI 得到不同结果。 +`test:headless-core` 显式运行当前的 +`headless-boundaries.test.ts` 与 `headless-event-contract.test.ts`。后者直接验证 +`HEADLESS_EVENT_VERSION`、`createHeadlessJsonlEvent` 和 +`HeadlessJsonlEventSchema`。所有显式测试清单都会在启动 Vitest 前检查文件存在性, +缺失路径直接失败。 -V8 coverage 通过 `bun run --filter blade-code test:coverage` 单独执行。coverage -编排覆盖 unit、integration、CLI、E2E、snapshot、security 和不需凭证的 real-api -fixtures,但显式排除 wall-clock `performance` project;instrumentation 与并行项目负载 -会使启动耗时失去可比性。性能回归仍是 `qualify:local` 在 production build 之后的必需项。 +V8 coverage 由 `bun run test:coverage` 单独执行。它覆盖 unit、integration、CLI、 +E2E、snapshot、security 和无需凭据的 real-api fixture,并排除 wall-clock +performance project。 ## 真实 API 门禁 -真实 API 门禁必须使用当前源码刚构建的 `packages/cli/dist/blade.js`。Provider -credential 可以由 secret manager 注入测试子进程环境,也可以放在测试专用的 -`~/.blade/real-api-credentials.json`。后者必须是当前用户拥有的普通文件且权限为 -`0600`;符号链接、宽松权限、未知字段和超过 64 KiB 的文件都会 fail closed。不要把 -真实值写成 inline `KEY=value`、执行 `export` 留在 shell history,或复制到证据文档。 -命令记录只保留变量名、模型 ID 和是否存在,不记录变量值。环境准备完成后执行: - -```bash -bun run --filter blade-code browser:install # 首次或 Playwright 版本变化后 -bun run --filter blade-code browser:check -bun run qualify:production -``` - -`browser:check` 不联网、不隐式下载,只验证锁定 Playwright 版本的 Chromium executable -可执行并能 launch/close。production qualification 固定执行 16 个检查:14 个本地门禁、 -无密钥 Chromium preflight、最后才是付费真实 API;preflight 失败时不得启动 Provider -请求。浏览器页只访问 loopback Blade server,API key 不进入 page context。 +真实 API 测试必须使用当前源码刚构建的 `packages/cli/dist/blade.js`。Provider +credential 可以由 secret manager 注入子进程环境,也可以放在 +`~/.blade/real-api-credentials.json`。凭据文件必须是当前用户拥有的普通文件、权限 +`0600`、大小不超过 64 KiB;符号链接、宽松权限和未知字段都会 fail closed。 -每条 capability 轨迹的 Provider request timeout 与 silent-stream idle timeout 必须 -同时短于 Vitest test timeout,并为 abort、runtime dispose、临时目录删除和全局配置 -恢复保留确定性清理窗口。验证 permission -recovery 的轨迹关闭 Provider retry;retry/backoff 只在专用故障注入轨迹中测试,避免 -测试框架 timeout 后旧 `finally` 与 Vitest retry 并发运行。permission recovery 的 -Write 资格使用唯一绝对 `file_path` 与 Write-only 工具白名单,text-only 回答不能替代 -真实文件副作用。 +不要把真实值写成 inline `KEY=value`、保留在 shell history 或复制到证据文档。 +日志只保留变量名、模型 ID、计数、耗时和脱敏宿主证据。 -该命令运行固定的 release-blocking real matrix:真实 DeepSeek Flash/Pro Headless -bugfix、GPT Web structured output、Claude ACP structured output、DeepSeek headless -structured output、Web/ACP/TUI code review、durable interaction recovery、permission -recovery、ACP model switch、ACP durable fork Write+Bash capability routing、透明 503 -retry proxy、durable 413 compaction proxy、真实 mid-stream stall proxy、assistant -response fsync fail-stop/cold retry、turn-final receipt exactly-once recovery、 -foreground/background shell hard-crash recovery,以及 DeepSeek Flash 的 -Runtime/Web/ACP host-authoritative Goal completion verification。ACP fork 固定要求 -DeepSeek Flash/Pro,并对当前资格环境中已配置的 Claude、GPT 与国产模型执行同一 paired -SDK trajectory;未声明 terminal capability 的 Client 必须使用 Session-bound local -terminal,声明后 terminal 失败仍须 fail closed。 -Agent Team task-list coordination 固定运行 DeepSeek Flash/Pro;每个模型启动四个共享 -`taskListId` 的真实后台 teammate,要求每个 teammate 实际调用一次 `TaskCreate`,并 -验证最终任务 ID 连续且唯一、subject 全量保留、JSON 快照可解析、跨进程 lock 已释放、 -同进程 keyed coordination 已归零。纯文本声称创建任务、仅检查 HTTP 200、mock -ToolExecutor 或只验证单进程 manager 均不能替代该轨迹。 -Goal completion 的 fresh PASS authority 同时绑定当前 host run、mutation revision 与 -由 Goal ID、attempt、requested-at timestamp 组成的 completion candidate identity。 -模型重复提交相同的幂等 `UpdateGoal complete` 时,必须保留已经由该 host run 记录的 -verifier Session ID、verdict、evidence digest 与 finalization snapshot;候选变化、 -workspace mutation 或进程重启才能使 receipt 失效,不能因重复候选额外消耗 -verification retry budget。 -Goal premature-stop recovery 固定运行当前资格环境中的 DeepSeek、Claude、GPT 与国产 -模型。每个模型必须先产生一次无工具的 `self_deferral`,随后在没有用户输入的情况下 -接收 host recovery continuation,完成真实文件写入、读取验证和独立 Goal verifier。 -测试使用显式 token budget 防止失败夹具形成无界循环。 -Goal verifier feedback 轨迹必须先让真实 verifier 拒绝缺失产物,再证明执行 Agent 从 -持久化、脱敏后的具体缺口中恢复并完成;相同反馈指纹的二次升级与三次自动阻断由确定性 -状态机测试覆盖。 -Goal execution-host failure 轨迹固定运行 DeepSeek Flash/Pro × Headless、真实 ACP -stdio、raw PTY TUI 与 production Chromium Web 八格矩阵。每格由真实模型发起三次 Bash -tool call,production Bash adapter 产生三次 typed timeout,GoalStore 在第三个 logical turn -原子 blocked 且不发起第四次 continuation。普通非零退出不得进入该 streak;Web reload、 -ACP metadata、TUI 状态与 Headless JSONL 必须从同一 durable Goal snapshot 得到一致结果。 -详见[Goal 执行宿主故障保护资格验证证据](./goal-execution-host-failure-evidence.md)。 -Durable Goal turn lineage 轨迹固定运行 DeepSeek Flash/Pro × Headless、真实 ACP stdio、 -raw PTY TUI 与 production Chromium Web 八格矩阵。每格必须完成三个真实 upstream 模型 -工具决策和精确六次 downstream Provider 请求,形成 root/current/parent 连续链;framework -retry 与 model retry 都为 0。Goal sidecar、durable `turn_started`、ACP metadata、Headless -JSONL、TUI 状态与 Web reload 后的 DOM 属性必须一致,且 lineage 不得进入 Provider prompt。 -详见[Durable Goal 回合链资格验证证据](./goal-turn-lineage-evidence.md)。 -完整 `test:real-api` 另含 GPT Prompt Cache efficiency 轨迹:先等待真实 cache read, -再替换全部稳定 prompt block,并要求 runtime 输出 `system_prompt_changed` attribution。 -该轨迹同时验证自适应 token 阈值;不得用 mock usage、固定 cache counter 或仅比较 -累计命中率替代。由于 GPT 通道延迟与可用性不稳定,它不属于 release-blocking matrix。 -Browser Panel 资格使用真实 DeepSeek Flash Session 与 production Chromium:同一 Web -页面必须先完成真实 Provider 回合,再从右侧 Preview 打开 Browser。Preview 模式加载 -两个独立 loopback fixture,验证 iframe sandbox/no-referrer、后退、前进、刷新和系统 -浏览器打开;Test 模式必须通过独立 server Chromium 返回 PNG 与 ARIA ref,完成真实 -表单填入、点击和 console 诊断;移动端全屏模态下两种 surface 都不得越界。URL 边界 -拒绝非 HTTP(S)、嵌入凭据与 Preview 中的 Blade 自身 origin;console/page/request -fault、Provider credential、server/browser/port 与临时目录残留必须为零。测试固定 -加入 realApiQualification,不能退化为只跑 jsdom。 -前台有界输出固定运行 DeepSeek Flash/Pro × Headless/production Chromium Web/raw PTY -TUI/真实 ACP SDK terminal 八格;单格 Provider deadline 180 秒、测试 timeout 240 秒, -完整 realApiQualification watchdog 为 90 分钟,发布矩阵固定 framework `retry=0`。 -每格还验证 surface egress:Headless -等待 `write(false) -> drain`,ACP 最多一个 `sessionUpdate()` in-flight,raw PTY 暂停 -reader 后继续渲染,Web 在运行中 reload 后按 durable cursor 恢复同一 tool/final state。 -raw PTY 必须锁存已经观察到的 final marker、stdout/stderr retained tail 和 truncation -notice,不能让 resize redraw 轮换有界终端窗口后反向抹除已成立证据;同时必须从 resize -后的新 PTY 数据再次观察 truncation notice,不能用 resize 前的历史命中放行。 -模型的整个最终响应必须严格等于单格 marker;ACP 失败诊断只能保留有界、脱敏的最终文本 -预览,不能用放宽 marker 或 framework retry 掩盖模型偏离。 -Root-turn crash auto-resume 另固定运行 -DeepSeek Flash/Pro × Headless/raw PTY TUI/production Chromium Web/真实 ACP -`session/load` 八格,所有入口都不得依赖额外 wake-up prompt。最终响应 token 必须与 -恢复 prompt、marker 文件和 Read output 区分,且不能在 prompt 中完整出现,也不得用 -重复词段制造无关的 Provider 拼写歧义;Web 已观察到恢复前缀但完整 token 不匹配时必须 -立即输出有界、脱敏的 assistant 文本,不能退化为固定 180 秒 locator 超时。raw PTY -必须按精确 inbox message ID 观察 acknowledgement,以及同一 turn 随后的 -`turn_completed`,不能以终端历史命中或固定等待窗口代替 durable terminal。ACP -多 Provider 对照的终答窗口必须晚于 180 秒 runtime hard timeout,并为销毁连接与临时 -目录清理保留剩余测试窗口;ACP fork 的 parent Read 与 child Write/Bash 是两个顺序 -prompt stage,每段由宿主 180 秒 deadline 发送标准 `session/cancel` 并等待 prompt -收敛;外层预算固定为 420 秒,只覆盖两个 stage deadline 和 60 秒清理余量。不得通过 -Provider 或 framework retry 延长。Edit+rewind、 -Goal finalization crash handoff 使用同一 Flash/Pro × 四入口八格矩阵,恢复阶段必须 -零 Provider 请求,随后再从同一 surface 完成真实 API follow-up。PTY follow-up 用户输入 -不得包含完整预期响应 marker;Provider request body 必须先解析 JSON `messages` 再验证 -prompt,避免输入回显伪装 assistant completion。Completed-subagent -adoption 与 background-subagent completion wake-up 也分别固定运行 Flash/Pro × -Headless/raw PTY/production Chromium Web/真实 ACP 八格矩阵。 -Bounded coordinated shutdown 另固定运行同一 Flash/Pro × 四入口八格矩阵;每格在真实 -foreground Bash 进入 host-visible PID barrier 后发送 production `SIGTERM`,要求 -exactly-one cancelled abort、同 Session 恢复、延迟副作用对照和全量资源回收。 -开放式多文件迁移、compaction、进程树、 -并发 owner 与 crash-tail 等高成本 provider/capability soak 由以下命令单独运行: +首次运行或 Playwright 版本变化后安装 Chromium: ```bash -bun run test:real-api +bun run --filter blade-code browser:install ``` -国产模型通道属于可选 soak provider,不进入默认发布阻断矩阵。需要显式加入时设置 -`REAL_API_INCLUDE_OPTIONAL_PROVIDERS=1`;余额不足或共享通道限流不会降低 DeepSeek、 -Claude、GPT 的必需准出标准。下文按能力列出的扩展 required matrix 描述完整 soak -contract,不表示每个 patch 都要同步阻塞发布。 - -凭据文件格式如下,`baseURL` 和单模型 `model` 可省略: +生产准出执行: -```json -{ - "version": 1, - "providers": { - "deepseek": { - "apiKey": "...", - "baseURL": "https://api.deepseek.com", - "models": ["deepseek-v4-flash", "deepseek-v4-pro"] - }, - "claude": { - "apiKey": "...", - "baseURL": "https://gateway.example.com", - "model": "claude-opus-4-8" - }, - "gpt": { - "apiKey": "...", - "baseURL": "https://gateway.example.com", - "model": "gpt-5.5" - }, - "domestic": { - "apiKey": "...", - "baseURL": "https://gateway.example.com", - "model": "qwen3.8-max" - } - } -} +```bash +bun run qualify:production ``` -可通过 `BLADE_REAL_API_CREDENTIALS_FILE` 指向其他文件。显式 Provider 环境变量优先于 -文件中的同名字段;只要命令环境中存在任一显式 API key 且没有显式指定凭据文件,该 -环境变量集合就成为完整 allowlist,不会隐式合并默认凭据文件或个人模型。 - -`qualify:production` 在启动任何测试子进程前会 fail-closed 校验: - -- DeepSeek key 必须来自显式环境、受限凭据文件或 `~/.blade/auth.json`; -- `DEEPSEEK_MODELS` 必须同时包含 `deepseek-v4-flash` 和 `deepseek-v4-pro`; -- 未提供 `DEEPSEEK_BASE_URL` 时使用 `https://api.deepseek.com`; -- `DEEPSEEK_MODEL` 默认选择列表中的第一个模型,供单模型轨迹使用; -- Claude、GPT 与 domestic 资格配置会投影为独立 `modelProviders` 渠道;每个渠道使用 - 自己的 provider-level endpoint 和专属进程凭据槽,同协议渠道不会串 key; -- API key 不写入项目配置、源码、命令参数、日志、JSONL 或快照。 - -release-blocking matrix 固定包含 DeepSeek Flash 和 Pro,并通过 Claude、GPT 验证 -跨 provider 的 Web/ACP 结构化输出。完整 soak 中显式启用的 Claude、GPT 和 domestic -模型还必须通过基础 chat、streaming、usage、finish、tool calling,以及 -Runtime、TUI、Web、ACP 四条 production entrypoint。仅收到文本或 HTTP `200` 不算通过。 -跨端 fork 轨迹必须让 pi-ai 从自定义渠道解析凭据,不得通过模型级 `apiKey` 参数旁路。 -渠道健康资格还会分别从 Web route、TUI `/doctor` 和 ACP callback 对 GPT、domestic、 -Claude 发送最多 8 tokens 的真实 probe;结果必须使用 canonical failure 投影且不得 -包含模型原文、原始错误或 API key。 - -Provider Retry 资格必须通过透明本地代理在第一次真实模型请求返回带 -`Retry-After` 的 `429` 或 `503`,第二次请求原样转发到真实 Provider。单个用户 turn -必须在不依赖测试框架重跑的情况下完成真实工具修改与项目测试;Headless JSONL 必须 -依次出现 sanitized `provider_retry` 的 `scheduled`、`attempt`、`recovered`,只产生 -一次最终内容和一次工具副作用。确定性 transport 测试另行覆盖 jitter/cap、 -HTTP-date、`retry-after-ms`、backoff cancellation、quota/context fail-fast、部分输出 -后不重放、exhausted 与 fallback;Provider response body、headers 和 key 不得进入 -LoopEvent、SSE、ACP、JSONL 或 transcript。Production Web GUI 必须在同一 Session -StatusBar 显示 retry attempt 与有界等待,随后无刷新完成,fresh tab 恢复最终结果且 -browser console 无 application error。TUI 条件允许时通过 Computer Use 验证 loading -状态和 Esc 取消;ACP 必须通过 `session_info_update` metadata 投影且不污染 assistant -正文。 - -Bounded foreground Provider recovery 另固定运行 DeepSeek Flash/Pro × Headless、 -真实 ACP stdio + child-backed terminal、raw PTY TUI 与 production Chromium Web GUI -八格矩阵。透明代理必须让前四个模型请求返回 replay-safe `503`,只将第五个请求原样 -转发真实 Provider;因此旧默认 2 次 retry 无法通过。每格必须在同一 root turn 内: - -- 按序投影 attempt `1,2,3,4` 与 recovered `4/12`; -- 携带 `mode=bounded_foreground`、总预算和非负 elapsed/remaining; -- 只执行一次 Edit 与一次 Bash,并由宿主再次运行 fixture 测试; -- 保持 Provider payload、transcript 与最终 workspace 不出现重复副作用; -- TUI 显示有界恢复与 Esc,ACP 只发 metadata,Web StatusBar 实时显示恢复并在完成后 - 清除,reload 后保留最终结果; -- Headless、ACP、raw PTY TUI 与 Web 必须观察同一个 generation/revision fenced - `provider_recovery` 投影;Web reload/reconnect 必须在 readiness 前接收权威快照,终态 - clear 后未见 revision `0` 的迟到 live revision 不得复活状态; -- Web composer 上方的可访问 banner 和 TUI loading/status 两个表面都必须呈现 - circuit/retry 的有效主状态、绝对 deadline 倒计时和既有 Stop/Esc 取消入口; -- typed `model_fallback` 必须携带精确、净化后的 source/target identity 和封闭 trigger, - 且不能清空或覆盖 Runtime 权威快照; -- 回收 Provider proxy/socket、ACP terminal/process、PTY、browser/page、SSE、server、 - port、临时 HOME/storage/workspace; -- 不得把 Provider key、私有故障 body 或 raw error 写入 JSONL、SSE、ACP、DOM、终端 - capture、transcript 或测试证据。 - -确定性测试还必须覆盖 recovery deadline 在 backoff 与 in-flight stream 内获胜、 -caller abort、12 次追加尝试硬上限、fallback 共享时钟、timer cleanup,以及 text、 -reasoning、tool call、usage、finish 任一 chunk 后禁止重放。feature matrix 的发布证据 -必须使用 `retry=0`。 - -Shared Provider circuit 复用同一八格矩阵,并将 Open 时间配置为合法的 `2000ms` -(production 默认 `10000ms`)。透明代理在第 4 个 `503` 完成后记录 monotonic boundary, -后续请求不得在 Open 窗口内到达;到期后只允许一个 HalfOpen probe。各入口必须按同一 -typed schema 投影 `opened -> waiting -> probe -> closed`,TUI circuit 状态优先于普通 -retry 文案,Web/ACP 不得把 metadata 写入 assistant 正文或 durable transcript。 - -Web 与 ACP 的 Flash/Pro cell 还必须在同一进程中启动第二个 Session。Session B 在 -Session A 已 Open 后提交,Open 窗口内零 Provider 请求;两 Session 合计恰好一个首个 -probe,关闭后分别取得独立真实 Provider 结果。Headless/TUI 不承载同进程多 Session, -其 cell 只验证单 Session 状态机与 surface 投影。全部八格与双 Session 控制使用 -`retry=0`,并回收 secondary Session、SSE、terminal、browser/page、proxy 与临时目录。 - -Provider request admission 资格固定运行 DeepSeek Flash/Pro × Headless、真实 ACP、 -raw PTY TUI 和 production Chromium Web 八格,测试配置使用合法的 -`providerRequestConcurrency=1` 与 `providerRequestAdmissionMs=120000`。透明代理先把 -一个已 admitted 的真实 Provider request 停在 host barrier;第二个同 domain request -必须先投影 `queued`,且代理在 release 前只能观察到一个请求。release 后第二个 request -按序投影 `admitted` 并取得独立真实 Provider 结果,代理最大同域 in-flight 必须恰好为 -1。 - -Web/ACP cell 使用同进程双 Session 对照。Headless/TUI cell 使用 production background -Task child 持有 permit,parent 下一轮排队;child 与 parent 都必须访问真实 Provider, -允许代理只为确定性 Task trigger 生成 synthetic tool call。八格还必须验证 root owner -继承、零 framework retry、attempt 不包含 queue wait、metadata 不进入 assistant/transcript、 -Web reload/ACP load 清除瞬态状态、TUI Esc 可见以及全量 proxy/browser/PTY/process/profile -回收。 - -Weighted Provider admission 另固定运行 DeepSeek Flash/Pro × Headless、真实 ACP、 -raw PTY TUI 和 production Chromium Web 八格负向矩阵。测试配置在合法 -`providerRequestPendingBytes=65536` 下先让一个大上下文 parent 或 Session A request -立即 active,再让同进程 background child 或 Session B 的大上下文进入等待;后者必须投影 -`rejected/queue_full/pending_bytes`,且透明代理观察到零额外 Provider traffic。 -Headless/TUI 使用 parent 持有 active permit、background child 被拒绝,并分别通过 -schema-valid child JSONL 或“失败 sidecar + 可见 TUI failure”证明同一事实;Web/ACP -使用双 root Session。最终 queue-full turn 保留 failed abort 但确认本 turn 输入, -reload、SSE reconnect 或 ACP load 不得把 Session B marker 重放到 Provider。Web 必须从 -真实 composer 提交并保持零 console/page error。TUI failure 必须由宿主的有界 terminal -summary 直接投影,不能依赖模型复述;raw PTY 在结束前还必须观察 hidden completion 已 -ack,且其提交 seq 之后存在 `turn_completed`,避免 teardown 与 held proxy request 重叠。 -全部八格使用 `retry=0`, -同时继续运行前述正常 queued→admitted 矩阵,防止 byte policy 把正常长任务缩窄为只会 -fail closed。 - -Weighted top-level task admission 资格使用合法最小 -`maxQueuedTaskBytes=65536`。DeepSeek Flash/Pro × production Chromium Task Home 和 -真实 ACP stdio 四格负向 target 必须先让 task A 的真实 Provider request 停在 host -barrier,再提交包含唯一 marker 与大段非 ASCII 文本的 task B。B 必须以 -`pending_bytes` 在 Provider traffic 前拒绝:Web 返回 typed HTTP 429、显示内联错误且 -catalog/reload 无 ghost task;ACP 持久化并投影 retryable -`capacity/pending_bytes`,assistant 正文不含 admission metadata。随后普通小 task C -必须进入 queue position 1,并在 A 释放后取得独立真实 Provider 结果。代理请求 body -不得包含 B marker。Task Home 的 composer 可输入不代表 dispatch 已可用;Chromium -必须先观察显式 `data-blade-task-dispatch-ready=true`,该状态同时受 workspace 与 model -readiness 约束。readiness 超时必须输出结构化状态诊断,不得盲等 submit disabled 或通过 -重载/重试绕过。 - -非干扰对照固定运行 Flash/Pro Headless `--task-isolation local` coding task 与 Flash/Pro -raw PTY root turn;全部在同一最小 byte limit 下完成。既有 production Web task -dispatch Flash/Pro worktree coding/FIFO trajectory继续 release-blocking。target 与 -controls 全部使用 `retry=0`,并回收 rejected inbox/Runtime/worktree、accepted task、 -browser、ACP process/terminal、proxy/socket、port、HOME/storage/workspace。Chromium -只允许预期的 `/tasks` 429 resource error,其他 console/page/request fault 必须为零。 - -Bounded Session Runtime residency 资格固定使用 -`maxResidentSessionRuntimes=1` 与 `sessionRuntimeIdleMs=30000`。DeepSeek Flash/Pro -× production Chromium Web GUI 两格先持有 Session A 的真实 Provider 请求,再由 -Session B 验证 typed HTTP 429 `resident_runtimes`、零 rejected marker Provider -traffic、idle LRU slot reuse 与 Session A durable cold follow-up。浏览器只允许预期 -429 resource error,其他 console/page/request/SSE fault 必须为零。 - -真实 ACP stdio 两格必须广告并调用标准 `session/close`:active Session A 占满唯一 -slot 时,new Session B 在 task/worktree/Runtime/Provider 副作用前收到 bounded -JSON-RPC capacity error;close A 取消并结算 prompt、确认 cancelled inbox、释放 slot, -随后 B 完成真实 turn,close B 后 load A 从 durable transcript 完成真实 follow-up。 -Flash/Pro Headless 与 raw PTY 再各运行一格单 root coding control,证明 resident limit -不会干扰非 multiplexed Runtime。目标与对照共八格全部使用 `retry=0`,并检查 -Runtime reservation、ACP/TUI process、Provider proxy、browser/profile、 -HOME/storage/workspace 全量回收。 - -Deterministic keyed coordination reclamation 资格要求所有 keyed mutex registry -仅保留 active/queued operation。确定性门禁必须覆盖成功、同步异常、异步 rejection、 -same-key FIFO、cross-key parallelism、回收后新 generation、10,000 个历史 key 和 -4,096 个 queued operation/256 key;durable interaction、Goal、OAuth credential、 -Config、Worktree、Web message submission 与 task delivery 七个 owner 在高基数 churn -后都必须精确回到零。 - -真实 API target 固定运行 DeepSeek Flash/Pro × production Chromium Web GUI 与真实 -ACP 四格。Web 必须通过真实 composer 顺序完成多个 Session、同 Session follow-up 和 -reload,并在每次 settle 后断言 message/task coordination 为零;导航导致的旧 SSE -`ERR_ABORTED` 必须单独分类,其他 console/page/request fault 为零。ACP 必须在同一 -multiplexed connection 上完成多个 Session、标准 `session/close`、`session/load` -follow-up,并断言 interaction/Goal/OAuth coordination 与 Runtime residency 归零。 -统计只能通过进程内 test seam 读取,禁止进入 HTTP、ACP、CLI、transcript 或持久化 -schema。四格及既有 Headless/raw PTY 非干扰控制全部使用 framework `retry=0`。 - -Provider Stall 资格必须让透明 SSE 代理先转发真实模型内容,再在 hard idle timeout -之前暂停后续事件。Headless JSONL 必须按同一 stall count 输出 sanitized -`detected → recovered`,标记 `output_started=true`,随后完成真实回复;代理必须证明 -只收到一个 Provider 请求,且不能出现 retry、重叠 `iterator.next()`、重复内容或工具 -副作用。确定性 transport 测试另行覆盖首事件前 stall、mid-stream stall、warning 后 -hard timeout、caller abort、deadline reset 和 warning 后仍只有一个 pending read。 -Production Web GUI 必须在 StatusBar 显示 stall duration/hard deadline,恢复后无刷新 -完成且 console 无 application error;TUI 必须显示 stall 状态、hard deadline 与 Esc -取消入口;ACP 和 Headless 只投影 metadata,不污染 assistant 正文或 durable transcript。 - -Provider total-attempt deadline 资格必须先通过透明 SSE proxy 转发真实 DeepSeek Flash -content,再把 completion 延迟到 `timeout` 之后且 `streamIdleTimeout` 之前。Headless -必须在 45 秒 total deadline 产生 typed error、保留已交付 content、零 retry、单一 -Provider request,并同步 abort proxy 的上游 fetch。Production Chromium Web 必须显示 -同一 `[data-blade-session-error]`,证明 assistant partial content 可见、凭据不进入 DOM、 -console/page fault 为零;terminal durable resync 只允许旧 -`/sessions/:id/events` EventSource 产生恰好一次 `net::ERR_ABORTED`,其他 request fault -必须为零。browser、server、proxy、HOME/storage/workspace 必须全部回收。 - -Reactive Compaction 资格必须让透明代理在首个真实 turn 返回一次 -`413 context_length_exceeded`,后续压缩摘要和恢复请求原样转发到真实 Provider。 -runtime 必须在零输出 replay boundary 内发出 paired compaction lifecycle,先提交含 -exact replacement messages 的 JSONL checkpoint,再重试同一 turn;不得产生 -`provider_retry`、重复工具副作用或无限 compaction loop。第二个独立 Runtime 必须仅从 -checkpoint model projection 恢复此前 marker,同时完整 transcript 仍可供 UI 展示。 -Production Web GUI 必须显示“上下文超限,正在恢复…”,无刷新完成最终回复;fresh tab -恢复可见历史并继续回答 checkpoint marker,browser console 零 application error。 -真实 raw PTY TUI 必须显示“压缩中”与 Esc 入口并完成同一 marker;ACP、Server SSE 和 -Headless JSONL 只暴露 reason/strategy/outcome/token metadata,不外泄 Provider 错误体。 - -工具并发资格要求 GPT 在同一个 production stream 中同时调用两个已加载工具。两个 -工具在执行函数内互相等待,只有都进入 shared gate 才能释放;因此单纯缩短总耗时或 -顺序执行无法通过。确定性测试另行覆盖 exclusive FIFO、同路径文件锁、abort、fallback -epoch、Web 多卡刷新重建、TUI keyed progress 和 ACP 独立 tool-call ID。 - -Bounded fair tool admission 另固定运行 DeepSeek Flash/Pro × Headless、真实 ACP -stdio + PTY terminal、raw PTY TUI 与 production Chromium Web GUI 八格矩阵。模型必须 -在单个 response 中发出四个真实 foreground Bash;host 在四条 canonical call 全部 -提交后证明单 Session 初始只启动两项、第三/第四项等待、每释放一项只推进一个 -successor、durable call/result 保持 Provider 顺序。独立的双 Session Chromium 轨迹 -要求 Session A 占用两个 execute slot 并排队第三项时,Session B 使用剩余全局 slot -先完成;两个 Session reload 后仍保留终态。所有格同时验证 queue progress、typed -overload metadata、进程树/lease/port/browser/PTY/ACP/临时根回收与 Provider credential -absence。 - -Bounded foreground command handoff 固定运行 DeepSeek Flash/Pro × Headless、真实 -ACP stdio + child-backed terminal、raw PTY TUI 与 production Chromium Web GUI 八格 -矩阵。模型必须以 `run_in_background=false` 启动同一 host-barrier Bash;1 秒测试配置 -预算到达后,durable result 和 surface 必须先发布 `auto_backgrounded=true`、typed -reason/budget 与同一 `shell_id`。子进程仍活跃时模型完成独立 Read,host 才释放 -barrier;随后恰好一次 TaskOutput 获得交接前后两个 output marker。每格证明: - -- 命令只启动一次,local PID 或 ACP terminal child identity 不变; -- foreground lease 原子替换为 background lease,ACP 不提前 release 或 local fallback; -- Headless typed JSONL、TUI、Web SSE/DOM 与 ACP update 均看到 handoff; -- TaskOutput 终态后 process/terminal、foreground/background lease、port、browser、 - PTY、SSE、临时根和 Provider credential 全部清零。 - -Durable token-budget handoff 固定运行 DeepSeek Flash/Pro × Headless、raw PTY -TUI、production Chromium Web 与真实 ACP SDK 八格矩阵,顺序固定且不经过 release -surface 过滤。每格通过 loopback transparent proxy 转发真实 Provider 输出和真实 -compaction summary;proxy 改写前两个 task response 的 usage counters,让首个 -compaction request 返回受控 context overflow,要求第二个请求使用严格更小的 payload, -再返回一次 `503`。Pro 最后透明转发第三个真实摘要请求;Flash 则让第三个 compaction -继续返回 `503`,强制进入 token-targeted deterministic fallback。 -70%/80% 阈值来自 production model catalog 的 context window 与 output reserve,而不是 -测试硬编码。proxy 将对应请求的 `promptTokens` 固定为阈值减一;只有把真实 -`completionTokens` 与响应后的 tool/control 增量计入下一请求前的完整上下文投影,才会 -跨过边界,因此旧的 prompt-only 实现不能通过。每格证明: - -- handoff marker 在第二次 task Provider request 前持久化,当前 epoch 恰好一个 - `handoff-message-` durable identity,pre-compaction request 至多一次 marker, - compaction 与其后请求为零; -- threshold checkpoint 必须记录 `preTokenSource=provider_plus_estimate` 且 - `estimatedPendingTokens > 0`;Headless JSONL、Web SSE 和 ACP metadata 投影相同字段, - TUI/Web context meter 使用完整 Provider total,而不是 prompt-only usage; -- latest replacement checkpoint 位于 marker 后,replacement/effective suffix 不含 marker, - 七段 continuation ledger 精确保留 mutation、failed verification 和 pending action - sentinels,checkpoint 记录 `sampleAttempts: 3`、`inputReductions: 1`, - `messagesOmitted` 与 `filesOmitted` 至少一项非零;Pro 无 failure reason,Flash 记录 - `transient_exhausted` 且证明 `postTokens <= fallbackTargetTokens`, - `fallbackTargetTokens > 0`,原始 active-task 消息已去重且两个 fallback 消息计数均 - 存在。完整 final marker 只由成功的验证命令产生,最终 assistant 必须逐字回显; - 真实 Bash fail、Write、Bash pass 的 durable 顺序正确; -- Headless cold projection、PTY resume、ACP `session/load` 和 Web post-completion reload - 均不得新增 Provider request;internal event/tag/identity/reminder 不进入 terminal bytes、 - ACP updates/terminal、HTTP history、SSE、Zustand、DOM 或 HTML; -- Web 在真实执行中 reload,并在每次 navigation 前搬运 page-owned EventSource evidence; - PTY byte stream 与 raw Session JSONL 是 terminal authority;ACP 使用真实 paired NDJSON - codec 与 child-backed terminal; -- browser/page/SSE、PTY、ACP terminal/process、server/proxy/port、Session/foreground - lease、临时 HOME/storage/workspace 全部回收,evidence 与日志不含 Provider credentials。 - -该 release-blocking trajectory 在 `REAL_API_RELEASE_MATRIX=1` 时 framework retry 为 0。 -Desktop computer-use 只能作为非阻断视觉观察;它不能证明 JSONL、Provider request -顺序或 marker non-fan-out,因此不是本 runtime contract 的 authority。 - -Compaction rich-media elision 使用已配置的 DeepSeek Flash/Pro、Claude 与 GPT 真实 -Provider 直接生成摘要。每个请求包含文本、inline data URL 图片与远程图片 URL; -loopback proxy 必须在转发前证明两个原始图片 payload 均不存在、固定图片占位符和 -文本证据存在。测试还要求 `imagesOmitted` 精确计数、canonical source messages 不变、 -Provider 并发为 1,且 proxy evidence 不保留 Provider credential 或被省略的媒体标记。 -确定性测试覆盖相同 sanitizer、fallback 元数据以及 Headless JSONL、Server SSE 和 -ACP 投影。 - -Compaction effectiveness guard 使用同一 DeepSeek Flash/Pro、Claude 与 GPT 真实 -Provider 矩阵,并确保每格输入都超过 5,000-token 生效门槛。正常真实摘要必须使完整 -replacement 不超过原可压缩消息体的 80% 且不超过模型 context window 的 50%; -确定性负向对照注入非空但过大的摘要,要求宿主拒绝该 LLM candidate、保留已计费 -usage,以 `insufficient_reduction` 写入 durable fallback checkpoint,并通过 TUI、 -Headless JSONL、Server SSE 与 ACP 元数据投影稳定分类。测试不得只比较模型返回的 -summary 文本,必须比较实际 replacement context。 - -Token-targeted deterministic fallback 资格必须覆盖:按 token 而非消息数量选择最新 -完整单元;assistant tool call 与对应 tool results 不拆分;超大边界消息保留头尾并截断; -reasoning 与图片 payload 不进入 replacement;canonical source messages 不变; -`postTokens <= fallbackTargetTokens`。目标取 `max(5,000, 原消息体 80%)`、模型 -context window 50% 与 50,000 tokens 的最小值,仅当 exact continuation records 与 -active-task checkpoint 本身更大时提升到 mandatory payload 的实际大小。DeepSeek -Flash 真实 Provider 对照必须先产生一个真实 summary,再由完整 replacement 的 50% -headroom guard 确定性触发 fallback,并验证 -`fallbackTargetTokens`、`fallbackMessagesOmitted`、`fallbackMessagesTruncated` -持久化和跨端投影。 - -Fresh independent verification 资格要求主模型实际完成三个文件的非平凡实现,并在 -第一次尝试结束时由 runtime 强制启动新的内置 `verification` subagent。Verifier -必须处于独立 child Session,运行项目已配置的真实测试,返回恰好一个结构化 PASS; -FAIL/PARTIAL、缺失 verdict、PASS 后继续写入或重试耗尽均不得成功完成。确定性测试 -另行覆盖 backend/API/infrastructure 单文件触发、文档/fixture 排除、reserved agent、 -YOLO 只读 Bash、越界 cwd/env/background/重定向拒绝和 durable mutation revision -恢复。本地 verifier 的 sandbox 还必须证明 workspace 不可写、网络关闭、user -home/Blade storage 不可读且 provider key/Session env 不进入子进程。Production Web -GUI 必须显示唯一 verification 卡片和 PASS badge、三个 changed -files 与最终 marker;清空浏览器状态并重启 server 后仍须恢复同一张卡,且页面不得 -暴露内部 completion reminder 或 application console error。ACP 必须通过标准 Task -tool-call update 显示同一 verdict。 - -Native Read-Only Code Review 资格必须覆盖 `uncommitted`、`base` 与 `commit` -target、tracked/untracked SHA-256 digest、500 files/8 MiB 边界、精确 changed-line -校验、stale、abort、process-restart interrupted、fork/rewind 和 single-active-review。 -内置 `review` 与 `verification` 共用 audit authority:写工具、后台命令、env override、 -越界 cwd 和网络必须拒绝;sandbox 中 workspace 不可写、HOME/Blade storage/provider -key 不可读,同时 Git 通过隔离 config 正常读取目标。 - -真实 GPT 必须分别经 production Web route、ACP `/review` 与 TUI runtime hook 找出同一 -授权绕过,返回 target 内的结构化 P0/P1 finding,并由宿主证明 review 前后文件 bytes -与 Git status 完全一致。Production DeepSeek GUI 必须从 Task Home“评审”模板启动, -实时显示独立运行态、只读工具进度和 completed 报告,无需手动刷新;fresh tab 还须恢复 -priority、relative path、line 与 confidence。结构化 report chrome 必须支持中英文, -browser console 不得有 application error。 - -Hook trust 资格要求 GPT 通过 production stream 实际发出工具调用。相同的 -PreToolUse command 在 `untrusted` 状态必须零副作用,只有当前 SHA-256 摘要被显式 -信任后才能执行。确定性安全测试另行覆盖 `0600` 原子存储、symlink/owner/mode -fail-closed、Git worktree common root、配置变化自动进入 `modified`、跨 workspace -配置隔离、stale reviewed digest 返回 `409`,以及 managed Function hooks 不进入项目 -摘要。生产 Web GUI 必须验证 review、trust、modified、re-trust、revoke、Escape -焦点恢复和 fresh-tab console。 - -Workspace Trust 资格必须在隔离仓库中同时放置恶意模型 endpoint、stdio MCP marker、 -`permissions.allow: Bash(*)`、BASH_ENV、项目指令和带副作用的 package -`type-check` script。未信任轨迹必须过滤全部项目层, -通过用户渠道完成真实 GPT SessionRuntime 回合,并证明恶意 HTTP endpoint 请求数和 -MCP/package-script marker 都为 0。信任后的本地 YOLO Session 才允许 post-edit -验证,并且只使用声明的 `type-check` script;ACP、`default`/`autoEdit`、无 script -和未信任路径必须零执行。确定性测试另行覆盖 `0700/0600` 存储、symlink/owner/mode -fail-closed、父目录继承与子目录 deny、Git worktree identity、TUI 启动 review、 -Web/ACP 管理入口、验证进程取消/回收,以及 -plugins/commands/skills/agents/instructions 的 discovery gate。 -Production Web GUI 必须只展示 package script 名称而非命令正文,并分别在未信任 -Default、已信任 Auto Edit 模式完成真实模型 Write;两者都不得产生验证 marker, -fresh tab 不得出现应用 console error。 - -ApplyPatch 资格不能只验证最终文件内容。确定性测试必须注入发布中途 rename failure -并证明所有 source/destination 恢复、stage/backup 归零;还要覆盖完整 grammar、 -context mismatch 零副作用、symlink escape、同路径三方排队、多路径死锁、Add/Delete/ -Move Snapshot 整体 rewind、Hook 任一路径匹配、LSP didClose/didSave 和 ACP 远端 -ambiguous write failure 的 read-back 补偿回滚。还必须人工构造 `preparing` 与 -`committed` crash journal,证明 Session startup 分别执行回滚和仅清理,并验证两个 -独立调用受 0600 workspace lock 串行化。 - -真实 GPT 必须先 Read 两个 existing files,再只调用一次 ApplyPatch 同时更新两文件 -并新增第三文件;不得退回 Edit、Write 或 Bash,且 workspace 不得残留 transaction -文件。生产 DeepSeek Web GUI 必须在 Auto Edit 本地任务中复现同一轨迹,展示三个 -changed files 与每文件 diff,并在 fresh tab 保持零 application console error。 - -LSP 资格必须使用真实 stdio JSON-RPC 子进程,不接受纯 mock transport。确定性测试 -覆盖 initialize、didOpen/didChange/didSave、publishDiagnostics、全部语义查询、 -同名双 Session 进程与环境隔离、ACP 零本地进程、ContentModified 重试、request -abort、崩溃有界重启和 dispose PID 回收。项目 LSP command 必须进入 Workspace -Trust review,args/env 不得投影到 Web。 - -真实 GPT 必须先通过 ToolSearch 激活 deferred LSP schema,再实际调用 hover;下一 -回合 Write 必须从同一连接收到 `FAKE1001` 诊断,Session 销毁后 PID 归零。生产 -DeepSeek Web GUI 必须完成 Security trust、Auto Edit 本地任务、ToolSearch → LSP → -Write 诊断轨迹,并在 task 终态证明 LSP PID 已退出且 fresh tab 无 console error。 - -MCP Session 隔离资格使用两个均已信任的项目:进程从 project A 启动并让 Store 包含 -A 的 stdio server,production SessionRuntime 则为 project B 创建。真实 GPT 回合前 -必须与 B 的 MCP 完成握手,marker 只能出现在 B 的 cwd,A 的 marker 必须保持不存在。 -确定性测试还要证明 builtin tools 不读取全局 MCP registry、ACP/CLI 来源优先级、 -`--strict-mcp-config` 和 connecting/error 客户端回收。生产 Web GUI 必须从 Security -面板批准 B、派发真实 API task、看到目标回复,并证明 task 终态后 stdio 子进程归零。 - -MCP Elicitation 资格必须使用真实 stdio MCP transport,覆盖 Form、URL、 -`notifications/elicitation/complete`、无交互面、非法响应、tool abort 和重叠调用。 -Form 响应必须按原始 requested schema 校验,Elicitation/ElicitationResult Hook 不能 -绕过 schema;事件和 Session transcript 不得保存用户填写内容。ACP 必须对无法表达的 -必填自由文本或多选 fail closed。真实 GPT 必须先用 ToolSearch 激活 deferred MCP -工具,再消费只存在于 elicitation 结果中的 profile 并继续 Write。生产 DeepSeek Web -GUI 必须完成 MCP 工具审批、结构化表单、任务注意力、最终回复、fresh-tab 零应用 -console error,并证明任务终态后 stdio PID 已回收。 - -MCP Roots/Sampling 资格必须让真实 stdio server 主动调用 `roots/list` 和 -`sampling/createMessage`。确定性测试覆盖 canonical URI、worktree execution root、 -ACP 空 roots、能力协商、配置上限、text/image、unsupported content、请求次数、重叠 -调用和 parent abort。Sampling 默认不声明;显式 opt-in 后每次仍必须 one-shot 审批, -YOLO 不得绕过,TUI/Web/ACP 不得显示虚假的持久授权。真实 GPT 必须完成 ToolSearch → -MCP tool → nested sampling → Write;生产 DeepSeek Web GUI 必须显示请求预览和 token -上限、完成最终回复、fresh-tab 恢复,并证明 PID、端口和临时目录已回收。 - -MCP Call Lifecycle 资格必须用真实 stdio server 验证 progress token、顺序进度、 -parent abort、idle heartbeat、hard total timeout 和 disconnect PID 回收。非法、倒退 -或过量 progress 不得进入 Loop;progress 只作为瞬态 `tool_progress` 投影到 -TUI/Web/headless/ACP 和 subagent,不得进入模型 transcript。真实 GPT 必须完成 -ToolSearch → progressive MCP → Write;生产 DeepSeek Web GUI 必须在默认折叠工具组 -直接显示进度 message 和百分比,并完成最终回复与 fresh-tab console 资格。 - -MCP Tool Result 资格必须用真实 stdio server 返回 text、structured content、 -image/audio、resource text/blob、resource link、大文本、协议错误和超限结果。binary -不得以 base64 进入模型、Web、ACP 或 transcript;artifact 目录/文件必须分别为 -0700/0600,并验证 Session hash 隔离、内容 hash、配额和 ACP 宿主路径隐藏。真实 GPT -必须完成 rich result → large result → Read private artifact → Write;生产 DeepSeek -Web GUI 必须显示 marker、size/SHA-256、artifact path 和最终回复,并证明 trace、 -transcript、PID、端口和临时目录无原始 `_meta`、base64 或凭证残留。 - -MCP Logging 资格必须使用真实 stdio server 覆盖 `logging/setLevel`、 -`notifications/message`、运行时调级、严重度过滤、nested secret/URL/token/`_meta` -脱敏、16 KiB 投影、8 KiB message、每秒 64 条限流和 Session ring。日志事件必须投影 -到 TUI/headless/Web/subagent/ACP,但 provider messages 和 durable transcript 中必须 -保持零 marker;ACP 只能显示 opaque hash。真实 GPT 必须完成 ToolSearch → logging -MCP → Write;生产 DeepSeek Web GUI 必须显示 warning/error 完成态诊断卡、MCP 管理 -面板日志与级别按钮、最终回复,并证明 error 日志不增加 failed tool count,PID、端口、 -trace、transcript 和临时目录无原始凭证残留。 - -MCP Server Instructions 资格必须读取真实 stdio initialize response,覆盖 NFKC、 -Unicode tag/Cf/Co/Cn 清理、1 MiB source、每 server 8 KiB、每 Session 32 KiB、 -JSON/XML 边界转义、 -source hash、snapshot replacement 和 connection generation 撤销。伪 -`` 内容不能覆盖 system/user/permission/trust;ACP 只能投影 -provenance hash。真实 transport 必须完成 V1 → crash/remove → V2/re-add 并回收两代 -PID。真实 GPT 与生产 DeepSeek GUI 必须在用户不提供必填 code 时,仅从 scoped -instructions 得到参数并完成 MCP → Write;Web 还必须显示 instruction 完成态卡和 -MCP 管理面板安全预览,trace/transcript 中不得包含隐藏 Unicode 或凭证。 - -MCP Completion 资格必须使用真实 stdio `completion/complete`,覆盖 capability、 -prompt/resource template catalog ownership、未知 argument/context 请求前拒绝、15 秒 -超时、turn cancellation、每 client 4 并发、1 MiB source、100 values、单值 4 KiB、 -累计 64 KiB、NFKC、Cf/Co/Cn/tag/bidi/private-use 清理、去重和 raw SHA-256。 -同名 server 必须保持 Session 隔离且回收全部 PID。真实 GPT 与 production DeepSeek -GUI 必须完成 ToolSearch → CompleteMcpArgument → scoped candidate → MCP tool → Write, -并忽略候选中的伪 system-reminder。Web 管理面板必须覆盖 prompt/resource target、 -partial value、安全候选、hash、truncation 与 pending 收敛。 - -MCP Async Tasks 资格必须使用真实 stdio task-capable server,覆盖 capability 与 -`taskSupport` catalog identity、默认 disabled、required 自动后台化、optional 默认前台 -及显式 `StartMcpTask`、Session/workspace ownership、取消和 dispose cleanup。原始 -server task ID、result `_meta`、Bearer 与宿主路径不得进入模型或 UI;result 必须经过 -共享 MCP Tool Result 预算。故障注入必须分别中断 `tasks/get` 和 `tasks/result`,新 -generation 只能在 task ID + `createdAt` 一致时恢复,全部旧 PID 必须退出。真实 GPT -与 production DeepSeek GUI 必须完成 ToolSearch → required task → opaque -`mcp_task_*` → TaskOutput → Write;Web 卡片必须从 running 原地更新为 completed, -管理面板必须显示 opt-in、Session 上限和 poll interval。 +该命令先运行 14 个本地检查,再运行无密钥 Chromium preflight,最后才启动付费 +Provider 测试。preflight 失败时不会产生 Provider 流量。 -Durable Session Archive 资格必须以 JSONL 为唯一真相,覆盖直接归档、fork/subagent -继承归档、单独归档后代、恢复根后保留子归档,以及 active/archived 独立 cursor -scope。归档必须在稳定顺序获取整棵子树 Session lease;任一 queued/running 后代或 -外部 owner 都要证明根 transcript 零新增。SQLite 递归 projection 与 JSONL fallback -必须逐条一致;Runtime、metadata update、Web write route 和 ACP `session/load` 均需 -在副作用前拒绝归档 Session。 +### 发布阻断矩阵 -真实 GPT 必须完成第一回合、归档、Runtime/update 双拒绝、恢复和带 durable history -的第二回合。production DeepSeek Web GUI 必须从 Session 行 Popover 归档,证明 active -catalog 清空且 archived message 返回 HTTP 409,再从 Archive Popover 恢复并完成同一 -Session 第二回合。transcript 只能出现 `archivedAt: timestamp -> null` 两次迁移,被 -拒绝输入不得落盘;fresh tab 只能有正常 Session 状态日志。测试结束后 server port、 -临时 storage root 和 worktree 进程必须归零。 +`test:real-api:qualification` 由 `scripts/test-config.js` 中的固定白名单控制,共 9 个 +测试文件: -Portable Session Markdown Export 资格必须直接读取稳定 JSONL snapshot 并应用所有 -durable rewind marker;不能用当前 provider context、Web 内存或 SQLite read model -代替。确定性测试覆盖 text/image/summary/reasoning、part update、无 message parent 的 -tool result、subagent/file activity、active/archived exact workspace,以及 system -recovery 内容不外泄。credential key、Bearer、private key、data URL、签名 URL、隐藏 -Unicode 和 workspace 外 Unix/Windows host path 必须经过预算投影。单 activity 64 KiB、 -总导出 16 MiB 和 ACP inline 1 MiB 都需 fail closed 或显式 truncation;正文 SHA-256 -必须可从 `---` 后 UTF-8 字节独立复算。 +1. `agent-trajectory.test.ts`:生产 Agent 读取、修改和测试 +2. `structured-output-trajectory.test.ts`:结构化输出 +3. `durable-interaction-recovery-trajectory.test.ts`:durable interaction recovery +4. `release-coding-trajectory.test.ts`:跨 surface 代码迁移 +5. `task-list-team-trajectory.test.ts`:Agent Team 任务协调 +6. `cross-provider-fallback-trajectory.test.ts`:跨 Provider fallback +7. `goal-mode-trajectory.test.ts`:Goal 创建、执行和完成 +8. `browser-tool-trajectory.test.ts`:Native Browser Tool +9. `acp-remote-filesystem-trajectory.test.ts`:ACP remote filesystem -TUI 必须使用 `0600` exclusive create 并证明同名文件不覆盖;ACP 不得写宿主路径; -Web 必须拒绝缺少 hash/count provenance header 的响应。真实 GPT 必须实际调用 Read -读取公开 marker、伪 API key 和宿主路径,导出保留 call/result 与 marker、隐藏敏感值, -并从 TUI 和 ACP 得到同一摘要。production DeepSeek Web GUI 必须分别从 active row、 -archived Popover 和 fresh tab 下载 exact Session;HTTP 响应需为 `no-store` 安全文件名, -正文 hash 匹配,fresh tab 无 application console error。测试结束后端口、临时根和 -下载验证产物必须归零。 +发布矩阵设置 `REAL_API_TEST=1` 和 `REAL_API_RELEASE_MATRIX=1`,Vitest retry 固定为 +0,并强制使用 DeepSeek Flash/Pro。需要 Claude 或 GPT 的跨 Provider cell 在凭据 +缺失时 fail closed,而不是降级成 mock。 -Durable Pending Interaction 资格必须证明权限、`AskUserQuestion`、MCP Elicitation -与 Sampling 请求在 surface 可见前写入 JSONL,用户响应在解除工具阻塞前写入。确定性 -测试覆盖大小预算、同 Session 单 pending、响应幂等、fork/rewind 隔离、HTTP schema、 -TUI/ACP 启动顺序和 Runtime mailbox 重载。进程重启后不得自动重放原工具副作用;必须 -关闭原 tool call、写入带 provenance 的恢复结果,再通过 durable inbox 启动 -pending-only turn。 +这些轨迹必须通过真实 Provider 请求和宿主可观察副作用:文件内容、durable event、 +工具结果、浏览器状态、ACP update 或测试进程结果。模型自述、HTTP `200`、mock +ToolExecutor 和 jsdom-only 覆盖都不能替代。 -真实 GPT 必须从预置 pending Session 分别经 Web response、ACP `session/load` 和 TUI -Runtime hook 回答结构化问题并实际调用 `Write`。Production DeepSeek GUI 必须在 fresh -load 显示问题与 pending badge,回答后自动继续、产生精确 changed file,fresh tab 不得 -再次显示问题,browser console 不得有 application error。 +### 普通真实 API 集合 -Web response 轨迹的完成窗口必须晚于该模型的 Provider hard/idle watchdog,不能由测试 -timer 抢先把仍在运行的请求归类为 recovery failure。超时诊断必须脱敏并同时包含 Bus -terminal/stall 事件、Session task metadata、durable interaction/inbox/turn 事件、 -transcript tail、目标文件和 Runtime residency;无论成功或失败都必须 shutdown route -controller,证明 active run、Provider lease 与 resident Runtime 已回收。该窗口只负责 -让 runtime 先给出 authoritative terminal,不得增加 Provider 或测试重试。 +`bun run test:real-api` 也使用显式 inventory,不再隐式扫描目录。当前 inventory 包含 +上述 9 个文件,以及: -Session Permission Mode 资格必须证明权限策略属于 durable Session,而非进程全局或 -单一 UI Store。确定性测试覆盖 `default/autoEdit/yolo/plan`、latest update wins、 -legacy fallback、fork/task 继承、非法值 fail closed、SessionStart Hook snapshot、 -Plan 批准后的写前持久化、metadata 失败零执行,以及显式调用覆盖高于恢复值。Web -切换历史 Session 必须恢复对应模式,新任务必须重置为 `autoEdit`,不能从上一 -`yolo` Session 泄漏。 +10. `goal-paused-usage-trajectory.test.ts` +11. `workspace-agent-resources-trajectory.test.ts` -真实 GPT 必须将进程默认设为 `default`,仅在 Session JSONL 中持久化 `yolo`,随后 -分别通过不携带 mode 的 Web HTTP、ACP `session/load`、headless `--resume` 和真实 -TUI activation 完成实际 Write。四条轨迹都必须产生精确文件字节;Web/ACP 不得出现 -permission request,headless 不得以“需要交互确认”失败。Production Web GUI 必须 -创建完全访问 Session,fresh reload 后仍显示完全访问;点击新任务后必须显示自动审批, -再返回原 Session 时恢复完全访问。浏览器 console 必须无 application error。真实 API -轨迹必须使用统一 180 秒 Provider hard timeout、零 retry,并让 surface/test terminal -窗口晚于 Provider timeout;Web failure path 必须 shutdown route controller,不能用 -局部 120 秒预算抢先误判长尾响应或遗留 active run。 +`goal-paused-usage` 的 40-cell 扩展矩阵只在 +`REAL_API_RELEASE_MATRIX=1` 时启用;`workspace-agent-resources` 总会运行内置 +DeepSeek skill 轨迹,并在配置 GPT 凭据时追加 workspace 隔离轨迹。因此如需执行当前 +inventory 中所有 release-only cell,使用: -Session Reasoning Effort 资格必须区分 durable selection 与 Provider effective -level。确定性测试覆盖 `auto/off/minimal/low/medium/high/xhigh/max`、model -capability projection、unsupported fail closed、active-turn 拒绝、Runtime service -原子替换、metadata 失败补偿回滚、fork/retry 继承,以及 TUI/Web/ACP 的同一 -Session 语义。 - -真实 GPT 必须经不记录 Authorization 的本地透明代理完成 `low + fast + low` 请求, -销毁 Runtime,将 durable selection 更新为 `high + standard + high` 后重建并完成 -第二次请求;代理必须直接观察到两个 `reasoning_effort` 请求值。production Web GUI -必须从 Task Home 完成首轮,再从 Session Composer 完成后续轮;fresh load 必须恢复 -完整消息和 Session 设置。JSONL、API request body 与 UI 三方一致,证据文件不得包含 -API key。TUI Computer Use 只有在测试桥接提供真实 raw TTY 时计入通过,非 raw stdin -的 Ink 启动失败不能冒充 UI 资格。 - -Session Service Tier 资格必须区分 durable selection、effective tier 与 Provider -request value。确定性测试覆盖 `auto/standard/fast/flex`、模型 capability -projection、OpenAI `default/priority/flex`、Claude Fast Mode payload/beta header、 -unsupported fail closed、active-turn 拒绝、model/effort/tier/verbosity/style -设置组原子替换、metadata 失败补偿回滚、fork/retry/subagent 继承,以及 TUI/Web/ACP -的同一 Session 语义。 - -真实 GPT 必须经不记录 Authorization 的本地透明代理完成 `low + fast + low` 请求, -销毁 Runtime 并从 durable metadata 恢复 `high + standard + high` 后完成第二次请求; -代理必须直接观察到 `priority` 与 `default` 两个 `service_tier` 值,且不能发生静默 -降级。production Web GUI 必须从 Task Home 完成首轮,再从 Session Composer 完成 -后续轮;fresh load 必须恢复完整消息和 Session 设置。两次 upstream 响应都必须为 -`200 text/event-stream`,JSONL、request body 与 UI 三方一致,证据文件不得包含 -API key。 - -Session Response Verbosity 资格必须区分 durable selection、Provider effective 值 -与实际 request projection。确定性测试覆盖 `auto/low/medium/high`、GPT-5/Codex -capability projection、Chat `verbosity`、Responses `text.verbosity`、Codex -`textVerbosity`、payload hook 合并、unsupported 与 fallback fail closed、 -active-turn 拒绝、model/effort/tier/verbosity/style 设置组原子替换、metadata -失败补偿回滚、fork/retry/subagent 继承,以及 TUI/Web/ACP 的同一 Session 语义。 - -真实 GPT 必须经不记录 Authorization 的本地透明代理完成 `low + fast + low` 请求, -销毁 Runtime 并从 durable metadata 恢复 `high + standard + high` 后完成第二次请求; -代理必须直接观察到 `low` 与 `high` 两个 `verbosity` 值,且不能丢失对应的 -`reasoning_effort` 或 `service_tier`。production Web GUI 必须从 Task Home 完成首轮, -再从 Session Composer 完成后续轮;fresh load 必须同时恢复完整消息和 -`high + standard + high`。两次 upstream 响应都必须为 `200 text/event-stream`, -JSONL、request body 与 UI 三方一致,证据文件不得包含 API key。 - -Session Communication Style 资格必须证明它与 Provider verbosity 正交,且不能提升 -prompt 权限。确定性测试覆盖 `auto/pragmatic/friendly/explanatory`、`auto` 无注入、 -受限 section 顺序与 guard、仅 style 切换零 Provider 重建、active-turn 拒绝、 -model/effort/tier/verbosity/style 设置组原子持久化、metadata 失败补偿回滚、 -JSONL/fork/retry、Task/Team/background/resume 继承,以及 TUI/Web/ACP 的同一 -Session 语义。普通 API 不得接受任意 style prompt、文件路径或 JSON。 - -Trusted Custom Output Styles 还必须覆盖 user/project/plugin namespacing、Folder -Trust、active plugin policy、`.blade` 对 `.claude` 的同命名空间覆盖、symlink/path -escape、hidden Unicode、文件/单 prompt/catalog bytes 与 count 预算、SHA-256 -provenance、不可变 Session snapshot、durable digest backfill/mismatch fail closed -和显式内置 style 恢复。Web/ACP catalog 只能暴露 -`id/name/description/source/contentSha256`,不能暴露 prompt 或宿主路径。 - -真实 GPT 必须经不记录 Authorization 的本地透明代理完成 `pragmatic` 请求,销毁 -Runtime 并从 durable metadata 恢复 `explanatory` 后完成第二次请求;代理必须在实际 -`system` 或 `developer` message 中直接观察到对应 style section 和权限 guard。 -production Web GUI 必须从 Task Home 完成首轮,再从 Session Composer 完成后续轮; -fresh load 必须恢复完整消息和 `explanatory`。两次 upstream 响应都必须为 -`200 text/event-stream`,JSONL、request body 与 UI 三方一致,证据文件不得包含 -API key。 - -custom style 的真实 GPT 与 production Web GUI 资格必须至少各完成 project 与 plugin -来源的一轮请求;透明代理直接观察对应 marker 位于受限 `communication_style` -section,Session JSONL 记录 namespaced ID 与 digest。fresh load 必须恢复 custom -selection;证据目录不得包含 style prompt 原文之外的凭证或绝对宿主路径。 - -MCP OAuth 资格必须使用真实 authorization server 与真实 Streamable HTTP MCP, -覆盖 RFC 9728/8414 discovery、动态 client registration、state/PKCE、code exchange、 -短期 access token 的 `401` refresh、请求重放、新客户端账本恢复、logout 和 callback/ -HTTP PID 回收。凭证账本必须验证 0600、原子并发、symlink/mode/schema fail closed, -endpoint/client/scopes 不能串线;普通 connect 必须零浏览器副作用,ACP/headless -不得访问宿主凭证或启动授权。真实 GPT 必须完成 -ToolSearch → OAuth MCP → Write。生产 DeepSeek Web GUI 必须经过显式 -Authorize/Continue authorization、刷新后的 Resume authorization、自动重连、MCP 与 -Write 审批、最终 marker 和 fresh-tab 恢复,trace 中不得出现 access/refresh token。 - -Workspace Agent 资源隔离资格使用两个均已信任的项目,各自包含原生与 plugin -agents、skills 和 commands。确定性测试必须通过真实文件 loader 建立两个 workspace -registry,复制 Session 快照后清空基础表,并证明 Task/Skill/SlashCommand 的描述与 -执行仍只包含所属项目资源。真实 GPT 必须在并发 `SessionRuntime` 与同一连接的 ACP -双 cwd Session 中分别调用对应 plugin command,marker 不能交叉;DeepSeek Flash/Pro -还必须通过生产 CLI `--agents -> Task` 完成代码修改与测试。production Web GUI 必须 -绑定并信任 A/B,在独立 worktree 中分别执行对应 SlashCommand,回切后保持各自 marker, -fresh tab 无 console error。Task/Team 的 foreground、background 与 resume 均要证明 -继承父 Session 快照;`projectRoot` 与执行 `workspaceRoot` 不得重新耦合。 - -Trusted Contextual Project Rules 资格必须覆盖 Git root 到 target 的层级、 -`AGENTS.override.md` shadow、`CLAUDE.local.md`、`.claude/rules`、 -`.blade/rules`、`paths` glob、Folder Trust、symlink/path escape、hidden Unicode、 -文件数与 bytes 预算、Session snapshot、去重、compaction 保留和 provenance mismatch -fail closed。首次只读触达后下一次 provider request 才能出现 conditional rule;首次 -写入触达必须在副作用前阻断。JSONL 只能保存 rule ID、repository-relative path 与 -SHA-256,不能复制规则正文或宿主绝对路径。 - -真实 API 轨迹必须在首个请求后删除磁盘上的 rules,随后通过 Read 从 Session snapshot -加载匹配 marker,证明不匹配 glob 从未进入 payload,并完成受规则约束的代码修改与真实 -测试。production Web GUI 必须展示 `Project Rules` 活动卡、完成真实响应和 fresh-tab -恢复;CLI/headless 与 ACP 必须输出同一安全摘要事件。 - -Session-owned User Shell Command 资格必须覆盖 32 KiB 输入、UTF-8 分片、ANSI 清理、 -binary 降级、capture/stream 独立预算、async output 排序、exact workspace/env、 -durable resume、active-turn auxiliary steering 和整棵进程树取消。TUI、Web、headless -与 ACP 必须证明 `!` 不创建 Agent;ACP terminal 不可用时必须 fail closed,不能回退 -Blade host shell。 - -真实 GPT 必须经不记录 Authorization 的本地透明代理执行 shell、销毁 Runtime 并恢复 -同一 Session;shell 阶段代理请求数必须为 0,后续真实 provider payload 必须直接包含 -`` 与输出 marker。production DeepSeek Web GUI 必须从 Task Home -创建普通 Session,网络中只出现 `/shell` 而非 task/message 请求;随后普通 follow-up -通过 `/message` 使用该 marker。fresh tab 必须恢复一个 command card、两轮 durable -history、零内部 XML 和零应用 console error。TUI Computer Use 只有在自动化桥接能保持 -真实 raw TTY 焦点并完整提交命令时计入通过;否则必须依赖 Ink 渲染、真实 PTY 和进程树 -测试,不能把启动截图算作完整 TUI 资格。 - -TUI Terminal Input 资格必须覆盖普通 multi-character stdin、同一 React batch 内的 -快速字符、完整和 split bracketed paste、CRLF、focus CSI、literal `[I`/`[O]`、 -TTY mode 成对启停与 GracefulShutdown 复位。raw Ink 测试必须把完整输入提交给 -command handler;production PTY 必须将 bracketed payload 送入刚构建的 -`dist/blade.js`。 - -真实 DeepSeek 必须经不记录 Authorization 的透明代理直接观察完整 pasted prompt -位于 provider request body,并返回由分段 token 组成的预期 marker。Web GUI smoke -必须证明 terminal-only 改动没有影响 Composer:多字符 `!` 输入仍只走 `/shell`, -fresh tab 恢复一个 command card,内部 XML 和应用 console error 均为零。Computer -Use 只有在工具能稳定寻址独立 terminal process/window 时计入通过;bundle ID 指向旧 -实例或焦点可能落入用户窗口时必须停止 UI 操作,改用 raw PTY 证据。 - -Production Web bundle 资格必须从 fresh build 产物计算,且构建调用者设置 -`NODE_ENV=test` 时仍必须打入 production React runtime。CI 不得读取旧 `dist` 通过 -预算;initial entry graph、单入口和总 JS gzip 均需在 production build 后重新验证。 - -Plugin Marketplace 资格必须使用隔离 HOME 和本地 Marketplace snapshot。确定性测试 -覆盖 `0600` 严格账本、跨进程串行写、Git `execFile` 参数边界、显式 source trust、 -symlink/路径逃逸/凭据 URL/体积限制、摘要篡改、失败更新回滚、旧根保留和依赖删除保护。 -真实 GPT 轨迹由 ACP 安装 v1、Web 刷新并更新 v2,活动 Session 必须继续调用 v1,新 -Session 必须调用 v2,卸载后后续 Session 不再投影命令。生产 DeepSeek GUI 必须完成 -Marketplace 添加、目录选源、可信安装、真实 SlashCommand、双确认更新/卸载、依赖阻止 -Marketplace 删除和 fresh-tab 零 console error。 -兼容性扩展还必须证明同 Marketplace 传递依赖一次提交、循环或 Blade/semver 不兼容时 -账本零变化、运行时固定点降级 dependent、反向依赖不可卸载;来源策略需覆盖 host -wildcard 边界、本地 canonical root、Marketplace identity、项目 tighten-only、 -`BLADE_PLUGIN_REQUIRE_SHA` 和 checkout SHA mismatch。 - -Workspace 模型与 Provider 隔离资格让两个已信任项目配置相同 channel ID 和 model -config ID,但使用不同 endpoint。确定性测试在 Session 快照创建后修改项目文件和进程 -全局 catalog,初始模型与 fallback 仍必须解析到各自原 endpoint。真实 GPT 资格通过 -两个本地记录代理转发同一真实上游,并发 Session 必须各命中一个代理且成功完成采样; -任何请求落到另一项目或后改的故障 endpoint 都判失败。Task/Team 的前后台与 resume、 -Prompt Hook、Web dispatch/message、ACP new/load/fork 都必须继承同一快照。 -Production Web GUI 必须在绑定项目 A/B 之间切换,模型按钮和展开列表只显示当前项目 -模型,迟到的旧 workspace `/models` 响应不得覆盖新项目,回切恢复且 console 为空。 - -真实 API 项目覆盖生产 CLI 轨迹,包括: - -- 持久化文件回退:模型通过生产 CLI 完成 Read/Edit/Bash 后退出,宿主以同一 session 重建快照管理器,验证原路径和写后哈希仍可恢复;回退后文件内容、干净 Git 状态和测试结果必须回到基线; -- 单文件缺陷修复:读取、编辑、运行测试并确认 diff 范围; -- 多文件 API 迁移:修改所有生产调用方并运行类型检查和测试; -- 临时 CLI 设置:从启动目录加载 `--settings` 文件,在代理转发首轮请求前删除该文件,验证隐藏系统指令已经进入模型上下文,并完成 Read/Edit/Bash、独立测试和 diff 校验; -- 分层项目指令:从 Git 根到 CLI 启动目录按作用域注入 `CLAUDE.md`、`AGENTS.md` 和 `BLADE.md`,在 32 KiB 预算内优先保留深层规则;透明代理验证首轮模型请求中的来源、顺序和覆盖值,并在转发前移除规则文件,证明最终修改不依赖工具补读; -- 瞬时 API 恢复:本地代理让首个模型请求返回 `503`,随后转发真实 API,CLI 必须在零输出边界内重试并完成代码修改与测试; -- 上下文超限恢复:本地代理让首个模型请求返回 `413 context_length_exceeded`,随后 - 透明转发真实摘要与恢复请求;Headless 必须完成 paired compaction lifecycle、持久化 - replacement checkpoint,并由第二个 Runtime 仅依赖该 checkpoint 回答此前 marker; -- 工具崩溃恢复:真实 DeepSeek 先通过生产 Headless 执行 Write,在外部文件副作用已发生 - 后注入 `tool_result` fsync 失败;当前 run 必须在发布 result 和第二次 Provider 请求前 - fail closed。第二个 Runtime 必须为已终止 turn 的 orphan 写入 - `sideEffectsUncertain` receipt,真实模型 resume 只能读取并确认既有文件,不得再次调用 - Write/Edit; -- 根回合自动恢复:持久化原始 inbox message、未闭合 Write 和已落盘 marker 后释放 - Session owner;新 Runtime 必须先提交 restart receipt,再从 canonical JSONL model - projection 恢复原输入。Headless bare `--resume`、TUI `--resume` 和 Web GUI SSE - reconnect 均只能执行一次 Read,Write/receipt/Read 各恰好一次,GUI reload 后结果仍 - 可见且浏览器无 application/network error。最终 token 不得完整出现在恢复 prompt 或 - Read 结果中;PTY 终态必须由精确 inbox acknowledgement 与对应 `turn_completed` - 共同证明; -- 响应提交恢复:真实 DeepSeek 产生最终文本后注入 assistant message fsync 失败;临时 - content delta 可以被观察,但当前 turn 必须 aborted、不得提交 assistant 或 - `turn_completed`。冷启动由 wake-up 输入触发后必须优先重新执行原 durable inbox, - 第二次真实响应成功提交且不泄露底层 I/O 错误; -- turn finalization 恢复:真实 DeepSeek 的最终 assistant 与 final receipt 已提交后,在 - inbox ack/terminal 前注入进程退出;冷启动必须先原子补 - `inbox_acknowledged + turn_completed` 并重载 sidecar,随后只处理新输入。旧输入不得 - 再次请求 Provider,最终历史必须有两个 completed、零 aborted turn; -- Goal finalization handoff:最终 assistant 的 host receipt 已提交、Goal sidecar 仍为 - `verifying/pass` 时退出。新 Runtime 必须用 exact goal ID、attempt、verifier Session、 - evidence digest 与 revision 幂等补 `complete`;Headless 回放、raw PTY、production - Web GUI 和 ACP `session/load` 均不得为旧 Goal 发起 Provider 请求。随后同一入口发送 - 新 prompt 并通过透明代理完成真实 Flash/Pro 响应,证明恢复后仍可继续工作; -- Goal verification attempt 是单调递增的恢复序号,不是固定值。FAIL/PARTIAL、格式纠正 - 或 evidence 失效后可进入 attempt 2 及后续 attempt;release trajectory 必须要求最终 - Goal 为 `complete`,且当前 attempt 具有 fresh `PASS`、verifier Session ID 和 - SHA-256 evidence,但不得把合法的正整数 attempt 锁死为 `1`; -- 计划模式恢复:跨两个 CLI 进程恢复会话并完成修改; -- 模式边界恢复:在 Yolo 中故意调用一次 ExitPlanMode,运行时必须返回 `validation_error`,模型随后继续 Write/Bash,证明过期规划状态不能终止已经批准的工作; -- 失败恢复:先重现测试失败,再修改,最后验证通过; -- 超时恢复:回收完整进程树后继续工具循环,并确认没有后代进程遗留; -- 后台 shell 硬崩溃恢复:独立 Blade owner 启动 TERM-ignoring detached process group 并 - 在 dispose 前硬退出;新 Runtime 必须通过 durable lease 与启动身份回收旧树,PID - 复用/身份不匹配及 TERM grace period 内的 ownership 变化不得误杀,损坏 lease 必须 - fail closed,lease 提交失败时 gate wrapper 不得执行用户命令,sidecar 不得包含命令、 - 环境、输出或凭据; -- 前台 shell 硬崩溃恢复:真实 DeepSeek 必须分别从 parent 和 subagent 发起含延迟写入的 - foreground Bash;宿主在 tool result 前 `SIGKILL` 独立 Blade owner,新 Runtime 取得 - parent/child Session lease 后必须先回收对应 foreground tree,再闭合 orphan Bash tool - receipt。延迟文件不得出现,lease commit/gate release 失败必须零执行,PID identity - mismatch 不得误杀,损坏 sidecar 必须阻断恢复,sidecar 与 CLI 输出不得包含命令、 - 环境、输出或 API key。Subagent 对照不得依赖固定 sleep:TERM-ignoring descendant - 必须等待 host gate,host 只在 root PID 退出且 durable lease 删除后开放 gate;若进程树 - 有任何残留,forbidden side effect 必须确定性出现并使资格失败; -- leaderless process group:parent 真实 API 轨迹必须先证明 shell/gate root PID 已退出、 - TERM-ignoring 后代仍存活,再硬杀 Blade owner。Linux/macOS reaper 必须独立探测负 - PGID 并在延迟副作用前完成 TERM/KILL;root PID 在 grace 中复用时禁止 KILL 并保留 - lease。正常 foreground/background/ACP local close 也要验证全重定向后代在 terminal - result 前回收;Windows 继续验证 live-root `taskkill /T`,不宣称 POSIX PGID 语义; -- session 退出回收:模型启动后台进程后正常结束 CLI,验证 runtime dispose 等待整棵进程树终止; -- 中断恢复:真实信号中断活动工具调用,持久化一次模型可见的中断边界,再由第二个 CLI 安全恢复; -- session 独占:活动 runtime 拒绝第二个同 session CLI 且不持久化其输入,owner 退出后允许恢复并继续验证; -- transcript 截断恢复:在 session JSONL 尾部制造未提交半行,恢复后完成 Write/Bash 任务,并逐行验证修复后的完整历史; -- 上下文压缩续跑:受限上下文窗口在 Read 后触发一次自动压缩;透明代理暂停真实摘要请求时,stdout 必须已实时发出 `compacting: started`,随后保持纯 JSONL、落盘自动摘要,并在 `compacting: completed` 之后执行 Write; -- Web surface:通过生产 HTTP session 路由提交任务并消费真实 SSE,验证代码修改、宿主测试、canonical tool success,以及 `compaction.started` / `compaction.completed` 在 resumed Write 之前按序可见; -- 结构化用户问题:Web 在 `yolo` 中仍必须发出 `question.required`,SSE 断线重连只重放当前未解决且 ID 不变的问题,提交结构化答案后继续 Write/Bash;ACP 在自动批准模式下也必须通过标准 permission options 收集单选答案。ACP 协议无法保真表达多选时 fail closed,不得静默降级为单选; -- 阻塞交互取消:TUI 先结算所有 confirmation 再中止回合;Web abort 必须使 pending permission/question 立即失效,并等待旧回合释放 runtime 后才返回 idle,晚到回答返回 `404`;ACP cancel notification 必须独立打断未响应的 reverse request。Flash 和 Pro 都要在同一 session 中取消问题后继续完成 Write/Bash,且 cancelled 终态不得被 completed/error 覆盖; -- 权限作用域:`once`、`session`、`project` 必须是不同契约。TUI/Web 显式展示会话级与项目级选择;session approval 只进入当前 runtime cache,不能写盘,并在同一 runtime 的第二个独立 turn 复用、在新 session 重新询问;project approval 写入目标 workspace 的 `.blade/settings.local.json`,不能落到 server 启动目录或泄漏到其他项目,新 Web/ACP session 必须自动加载。ACP `allow_always` / `reject_always` 映射为真实项目级持久规则;Flash 和 Pro 都通过真实 Bash 轨迹验证; -- 交互式后台 Shell:`WriteStdin` 只能操作当前 session 拥有的后台 Bash,等待写入完成并可显式关闭 stdin;跨 session、已退出进程和缺失 session 必须 fail closed。TUI、Web、ACP 中的 Flash 和 Pro 都要完成 `Bash(background) -> WriteStdin(close) -> TaskOutput(block)`,并由宿主验证实际文件和三个工具事件; -- 有界后台输出:后台 Bash 的 stdout/stderr 各自超过 1 MiB 后只能保留最近输出并精确报告更早省略字节数;TUI、Web、ACP 中的 Flash 和 Pro 都要完成 `Bash(background) -> TaskOutput(block) -> Write`,验证尾部标记、`output_truncated`、stream 省略字节数、共享展示摘要和宿主证明文件; -- 有界前台输出:宿主预写脚本向 stdout/stderr 各输出 `1 MiB + 64 KiB`,每流最初 - 4 KiB 内放 omitted-prefix sentinel,末尾放独立 nonce tail。Flash/Pro 必须在 - Headless、Chromium Web、raw PTY TUI、ACP SDK 四入口各只调用一次 foreground Bash。 - Headless 首次 stdout/stderr write 返回 `false` 并延迟 drain,期间不得出现第二次 - raw write;ACP 每次 update 延迟且最大 in-flight 必须为 1;PTY host 暂停 reader 后 - 必须恢复最终输出;Web 必须运行中 reload、cursor reconnect 并在 fresh terminal load - 保留同一 tool card。Web/PTY 验证双流 total/retained/omitted;ACP 验证 merged - stdout、zero stderr 和 `terminal_output_merged=true`。所有入口都必须保留双 tail、 - 隐藏双 sentinel 与 API key,并清零 Chromium/page/SSE、PTY、ACP terminal、 - process identity、foreground lease、port 和临时根。虚拟列表重挂载 completed card 时, - Web 必须原子取得 durable `toolCallId`,在 tool group 回到折叠态时重新展开并触发真实 - click handler,再断言 `aria-expanded=true` 和有界输出,不得把 card/toggle 两次 - locator await 之间的重挂载或布局 actionability 抖动当成 runtime failure。 - Goal finalization fresh-load 必须在完整有界 qualification budget 内同时验证 persisted - Goal `complete` 与 DOM `complete`,失败时报告两侧状态及 browser faults,不能用更短的 - hydration 子截止点替代端到端预算。 - foreground gate-release failure 对照必须在释放前订阅 stdout,并在观察到超过 retained - budget 的实际字节后注入错误,不得用固定 sleep 假设输出已经到达。raw PTY 的正向 - marker evidence 必须单调锁存;resize 或后续 redraw 只能增加证据,不能从 bounded tail - 撤销已观察事实。source contract 精确枚举 `backgroundSubagentCompletion`、 - `browserTool`、`foregroundBoundedOutput`、`foregroundCommandHandoff`、 - `foregroundProviderRecovery`、`goalFinalization`、`gracefulShutdown`、 - `rootTurnAutoResume`、`sessionRuntimeResidency`、`subagentResultAdoption`、 - `toolAdmission`、`tui`、`weightedProviderAdmission` 13 个 PTY runner;新增 runner - 必须更新 inventory 并显式完成 marker-latching 审计。只有明确要求 resize 后仍可见的 - 事实才能由 resize 后的新 PTY 数据重新证明,不能复用历史匹配。background completion - 的 Provider queue、child marker 与 parent final 必须消费同一个有界 evidence - deadline,不能用更短的首阶段截止点误判慢首响应。Computer Use 仅在宿主 - 提供稳定桌面桥接时作为补充视觉证据,不能替代自动 raw PTY 与协议断言; -- 跨表面 session branch:TUI `/branch` 原子切换到持久化子会话;Web 通过 HTTP fork 路由创建并选中子会话,活动回合返回 `409`;ACP `/branch` 返回可由标准 `session/load` 加载的子会话 ID。Flash 和 Pro 都必须在删除原 marker 后,仅依赖继承的 Read 结果继续 Write/Bash,并证明父 transcript 未改变; -- durable turn rewind:Flash 和 Pro 都必须先通过真实模型 Read/Edit/Read 产生文件 - checkpoint,再分别从 Runtime、TUI hook、Web HTTP/SSE 和 ACP `/rewind` 入口恢复。 - Runtime/TUI/Web 验证代码回到 baseline、有效 conversation 被移除且 JSONL 保留 - `session_rewound`;ACP 验证 conversation-only rewind 后重建 Agent,后续 prompt - 只使用投影历史。Web 还必须通过真实浏览器验证按钮禁用态、checkpoint 对话框、 - code restore 开关、提交后的消息列表和磁盘效果。实际结果记录在 - [durable rewind 证据账本](./durable-rewind-evidence.md); -- TUI runtime 生命周期:通过 `useAgent` 完成真实模型回合后清理,并用同一 session ID 重新获取 runtime lease,证明退出路径释放 Agent、后台资源和会话所有权; -- ACP session/load:通过真实 ACP SDK NDJSON 连接新建并销毁会话,删除原始 marker 文件后加载持久化历史,在响应前回放用户/助手消息,并仅依赖恢复上下文继续 Write/Bash;客户端传入的 MCP server 使用会话私有注册表,初始化失败或退出时独立回收; -- ACP 会话模型切换:会话以 Flash 初始化后通过真实 `session/set_model` 切换到 Pro,透明代理必须只观察到 Pro 的后续采样请求;切换期间原子更新 provider 与上下文窗口、回收旧 provider,并完成 Read、源码修改、Bash、独立测试与 Git diff 校验; -- 单次运行 Subagent:通过 `--agents` 注入只存在于子代理系统提示中的模型专属规则,主代理仅开放 Task;Flash 和 Pro 都必须委派到自定义代理,完成 Read/Edit/Bash、独立测试、精确文件范围和纯 JSONL 校验; -- durable Subagent resume:Flash 和 Pro 都必须先由真实 Task 产生只存在于 child - transcript 的上下文,再从 Runtime、TUI、Web 和 ACP 四个入口恢复。follow-up prompt - 不得包含目标值;child 必须只依赖恢复历史给出正确结果,并证明 source sidecar 不变、 - child ID 新建、lineage 深度递增、冻结模型/权限生效及无密钥泄漏。Web 还要通过真实 - 浏览器验证 depth 1 → 2、刷新后 2 → 3、禁用态和零 console error。实际结果记录在 - [durable subagent resume 证据账本](./durable-subagent-resume-evidence.md); -- durable Subagent crash recovery:真实 child Provider stream 在首个 content delta 后 - 由透明代理保持,宿主 SIGKILL Blade owner;第二 Runtime 必须先闭合 child turn/tool - receipt 并从 JSONL 重建 sidecar history,再以新 immutable child ID resume。follow-up - 不得包含原 token,恢复失败不得允许 Web/ACP 发起虚假 resume; -- durable completed-Subagent result adoption:Flash 和 Pro 都必须先通过真实 foreground - Task 生成只存在于 child 结果中的 marker,再保留 active parent turn、durable inbox 与 - orphan parent Task call,并在 parent `tool_result` 提交前释放 Runtime。Headless、raw - PTY TUI、production Chromium Web GUI 和 ACP `session/load` 必须从同一 child sidecar - 采用结果,不得再次启动 child。每格验证 resumed Provider request 含 child-only marker、 - adopted result/parent abort/inbox ACK/parent final 各一次、child sidecar 字节不变、 - compound owner 与 lineage 唯一、`sideEffectsUncertain=false`、Web live/reload 可见、 - 进程/端口/临时根清理及凭据不泄漏; -- durable background-Subagent completion wake-up:Flash 和 Pro 都必须由真实 parent - 调用 `Task(run_in_background=true)`,Task 返回 running 后 parent 继续独立 Read,且 - 全程零 `TaskOutput`。child 通过 Read 取得 parent input 中不存在的 marker;terminal - sidecar、hidden canonical receipt 与 durable inbox 必须自动唤醒 parent。Headless、 - raw PTY TUI、production Chromium Web GUI 与 ACP `session/load` 每格验证 terminal - ref/inbox ACK/parent final、child 与 lineage 唯一、sidecar 字节稳定、无伪用户消息、 - Web live/reload 一致及资源/凭据清理; -- 输出协议、工具调用、错误事件和 key 泄漏检查。 - -### Durable 大型 Prompt 分流轨迹 - -该能力验证大型用户请求不会在第一次 Provider 调用中无界展开,同时模型仍能可靠取得 -完整指令: - -1. fixture 必须超过 32 KiB inline 阈值,并把唯一 hidden authority 放在头尾摘要都无法 - 覆盖的中段;最终值只能由该 authority 中分离的 token 组合得到,首轮请求不能包含 - hidden marker 或完整最终值。 -2. 透明代理必须直接验证首轮 user content 不超过 32 KiB、包含一个合法 opaque - artifact ID、声明 `ReadPromptArtifact`,且尚未包含 hidden marker。 -3. 后续请求只有在 assistant 使用同一 artifact ID 调用 `ReadPromptArtifact` 后,才允许 - hidden marker 出现在对应 tool result;marker 不能出现在 user、assistant 或 system - 内容中。模型必须读取到 `[End of prompt artifact]` 并返回精确最终值。 -4. JSONL 必须持久化同一 `userPromptArtifact` 引用、完整 ReadPromptArtifact - tool-call/result 轨迹和精确 final;cold projection、PTY resume、Web reload 与 ACP - `session/load` 都不能为已完成输入新增 Provider 请求。 -5. 确定性门禁还必须覆盖 UTF-8 分页、哈希/权限/篡改失败、并发 artifact 数量配额、 - 多模态顺序、fork 引用复制、Session 删除清理,以及 1,000,000 字符/4 MiB 输入上限。 - -required matrix 固定包含 `deepseek-v4-flash` 和 `deepseek-v4-pro`,每个模型都必须通过 -Headless、真实 raw PTY TUI、production Chromium Web GUI 和 ACP 四个入口,共八格。 -测试结束必须回收 browser/page/SSE、PTY、ACP connection、server、port、Session -lease、artifact 与临时根;API key 不得进入 transcript、页面、终端、ACP update、 -诊断或结构化代理证据。 - -### Durable Subagent resume 轨迹 - -该能力必须覆盖相同的 immutable lineage 契约: - -1. **Runtime**:foreground root 先持久化完整 ChatContext,销毁并重建 manager/runtime - 后恢复为 child,再恢复为 grandchild。每条边使用新 ID,源 sidecar 字节和终态不变, - `rootAgentId` 稳定,`resumeDepth` 单调递增。 -2. **CLI/TUI**:真实 `useAgent` owner 通过 `/tasks resume` 继续已结束 agent,更新 - subagent progress,但不释放 parent Runtime。退出后必须正常释放 Runtime lease。 -3. **Web**:通过 exact `sessionId + projectPath` 的 GET/POST routes 和 SSE 发布 - `subagent.start/update/tool/complete`。消息卡片支持 follow-up、running polling、 - recoverable error 和刷新重建;不得把 ancestor 误选为最新 descendant。 -4. **ACP**:通过真实 ACP SDK lifecycle 执行 `/tasks resume`,使用标准 `tool_call` - 和 `tool_call_update` 暴露新 child ID、状态和结果。不得用私有文本事件代替工具协议。 - -sidecar 必须使用原子写、`fsync`、`0600` 文件权限和 `0700` 目录权限。公共 Web -schema 只返回状态、lineage、结果与统计,不得返回 prompt、messages、配置快照、 -workspace、owner PID 或 provider credential。跨 workspace、类型冲突、running source、 -active parent turn 和 durable pending input 都必须 fail closed。 - -required matrix 固定包含 `deepseek-v4-flash` 和 `deepseek-v4-pro`。每个模型都要通过 -Runtime、TUI、Web、ACP 四条真实产品入口;worktree resume 另行验证新的 child ID -继续使用源 lease owner 并保留失败或有改动的 worktree。 - -### Durable completed-Subagent result adoption 轨迹 - -该能力验证 child terminal 与 parent Task result 之间的跨存储 commit gap: - -1. fixture 必须先通过真实 Provider 运行 foreground Task,并要求模型生成不在 parent - input 中出现的 child marker;随后只持久化 parent `tool_call`,不提交 result。 -2. Runtime 只能从 exact compound owner 的 durable sidecar 采用 - `completed`/`failed` 结果;child ID、description、显式 type、resume lineage、状态或 - 有界结果任一不匹配时必须回退通用 uncertain receipt。 -3. 采用批次必须按原 tool-call/message identity 写入一个 `tool_result`、一个 terminal - `subtask_ref` 和一个 `turn_aborted(process_restart)`;第二次启动不能重复写入。 -4. Headless、raw PTY TUI、production Chromium Web GUI 与 ACP `session/load` 必须消费 - 标准 LoopEvent。Web 还要以 durable child Session ID 定位原 card,验证 live terminal - summary、parent final、reload 后同一状态和零 browser fault。 -5. 每个 surface 的 resumed Provider request 都必须包含 child-only marker;child sidecar - 字节、child 数量与 lineage 保持不变,证明没有重复执行 Task。 - -required matrix 固定包含 `deepseek-v4-flash` 和 `deepseek-v4-pro`,四个 surface 共八格。 -该轨迹属于 release-blocking real API qualification,不得由 mock、HTTP 200、仅 JSONL -检查或刷新后偶然可见替代。 - -### Durable background-Subagent completion wake-up 轨迹 - -该能力验证 background Task 不依赖模型轮询即可推动长任务: - -1. parent input 只能包含 marker 文件名,不能包含 child marker。真实 parent 必须先调用 - 一次 `Task(run_in_background=true)`,收到 running result 后再完成一次独立 Read; - parent transcript 中 `TaskOutput` 调用数必须为零。 -2. child 必须使用真实 Provider 和 Read 工具取得 marker 并提交 terminal sidecar。 - Runtime 只接受 exact compound owner、`background=true`、canonical child ID、 - type/description/resume lineage 与结构有效的 bounded terminal result。 -3. parent 必须按 `child sidecar fsync → hidden receipt + terminal subtask_ref → - durable inbox → model consumption → inbox ACK` 顺序收敛。deterministic inbox ID、 - receipt、terminal ref、ACK 和 child lineage 都必须恰好一次;冷启动不得重复通知。 -4. Headless 必须在 child 运行期间保持 Agent stream;raw PTY TUI、Web 与 ACP 必须自动 - 继续 parent,不能要求人工输入。ACP 不得生成 marker `user_message_chunk`;TUI/Web - 不能渲染伪用户消息。 -5. production Chromium Web 必须验证 live terminal card、parent final、reload 后相同 - child ID/status/summary、terminal sidecar 字节不变和零 browser fault。streaming Task - 必须在持久化前获得 canonical child ID;child 早完成时,迟到 running result 不能 - 降级 live 或 fresh-load card。 - -required matrix 固定包含 `deepseek-v4-flash` 和 `deepseek-v4-pro`,四个 surface 共八格。 -测试结束必须回收 browser/page/SSE、PTY、ACP connection、server/process tree、port、 -临时 storage/workspace/trust root,并证明 API key 不进入 JSONL、sidecar、DOM、PTY、 -ACP update、diagnostics 或录制的 Provider body。 - -### Bounded coordinated shutdown 轨迹 - -该能力验证正常进程关闭不会把 active turn 留给下次冷启动修复: - -1. 每格必须先通过真实 Provider 调用一次真实 foreground Bash;host 只能在 child PID - 文件存在且进程仍活跃后发送对应 production `SIGTERM`。 -2. Agent/Session owner 必须先关闭新工作入口,再中止 active Provider/tool path,等待 - 一个 `turn_aborted(cause="cancelled")` 提交后释放 Runtime、Session lease 与 transport。 - active tool 的 `shouldExitLoop` 不得抢先绕过 abort terminal;同一 interrupted turn - 不得出现 `turn_completed` 或第二个 terminal record。 -3. foreground child 必须忽略 TERM 并安排延迟 forbidden side effect;shutdown 必须 - 回收完整进程树和 durable lease,等待对照窗口后 forbidden 文件仍不存在。 -4. 原 durable input 必须保持可恢复。同 Session 的 production Headless resume 不得再次 - 调用 Bash;透明代理必须观察 exactly-one resume request,且该请求直接包含完整 - `` system marker 与原请求 marker,随后产生非空 final 并新增一个 - `turn_completed`。不得把模型是否逐字复述 marker 当成 runtime 是否恢复该 marker - 的唯一证据。 -5. Headless、真实 ACP stdio + terminal、raw PTY TUI 与 production Chromium Web GUI - 分别从独立进程进入。Web 必须通过真实 composer 提交;关闭 viewer 不能等同于 server - shutdown。每格回收 browser/page、PTY、ACP connection、server、port、process tree、 - Session lease、临时 storage/workspace/trust root,并全量扫描 credential absence。 - -required matrix 固定包含 `deepseek-v4-flash` 和 `deepseek-v4-pro`,四个 surface 共八格。 -该轨迹属于 release-blocking real API qualification,不得以 mock signal、直接调用 -`SessionRuntime.dispose()`、仅检查进程退出码或 cold `process_restart` 修复替代。 - -### Session discovery 与 durable fork 轨迹 - -Session discovery 与 durable fork 的准出必须覆盖四个相互独立的 production entrypoint: -一个内部 Runtime boundary,以及 CLI/TUI、Web、ACP 三个用户可见 integration surface。 -每条轨迹都必须从对应 entrypoint 进入,不得以直接调用模型或只验证代理请求代替产品 -边界: - -1. **Runtime**:通过真实 `SessionRuntime`、`Agent` 和 public - `SessionService.forkSession()` 创建 child。parent 先执行 Read,child 仅依赖继承的 - Read 结果执行 Write/Bash。证据包括精确文件效果、parent JSONL 字节不变、parent/root - lineage、child 独立追加、runtime 清理和 evidence 中无 secret。 -2. **CLI/TUI**:通过真实 `useCommandHandler.executeCommand()` 执行初始 prompt、 - `/fork ` 和 child prompt。证据包括 slash-command routing、child activation、 - 精确文件效果、parent transcript 不变、lineage、active turn 对 `/fork` 的拒绝语义以及 - evidence 中无 secret。 -3. **Web**:通过 production HTTP session routes 和 SSE 完成;deterministic Web tests - 另行覆盖 Sidebar action 与 store activation。 - - completed parent:fork 后 child SSE ready 并以 `sessionId + projectPath` 激活,child - 仅依赖继承历史产生精确文件效果;parent JSONL 保持不变且 lineage 正确。 - - active parent:在真实 provider request 保持 active 时 fork 已提交的稳定 JSONL - prefix;parent 不被取消并继续追加,child 不包含边界后的 parent 内容且独立运行。 - 两个子项都检查 compound workspace identity、结构化 HTTP/SSE evidence、资源清理和 - secret absence。 -4. **ACP**:通过真实 ACP SDK NDJSON codec 和 dispatcher 执行 `session/list`、 - `session/fork`,随后直接向返回的 child prompt,不调用 `session/load`,也不 replay - history。证据包括 discovery metadata、精确文件效果、parent immutability、lineage、 - child 独立追加、session 清理和 evidence 中无 secret。 - -该组轨迹的 required matrix 固定包含 DeepSeek Flash 和 Pro。若显式配置 Claude、GPT -或 domestic provider,其配置模型也必须运行四个 production entrypoint;缺少 required -Flash/Pro 时 fail closed。API key 只能从受限本机存储投影到子进程凭据槽,不写入项目 -配置、命令记录、日志、快照或原始请求头。实际命令、模型集合、退出码和复跑事实记录在 -[session discovery 与 durable fork 证据账本](./session-discovery-fork-evidence.md)。 - -仅收到模型文本或 HTTP 200 不算通过。每条轨迹都必须证明预期的文件或持久化 -副作用、结构化事件和进程退出状态;涉及代码修改的轨迹还必须记录 -`git diff --name-only` 以及测试或类型检查退出码。session fork 轨迹改为验证 parent/child -JSONL、lineage、精确 fixture 文件内容和资源清理,不虚构 Git diff 证据。 - -### Native Browser Tool 轨迹 - -原生 Browser Tool 的发布门禁分为两层: - -1. keyless Chromium 集成测试使用真实固定版本 Chromium 与本机 loopback fixture, - 覆盖导航、ARIA snapshot/ref、表单交互、等待、console/network/find/screenshot、 - history/reload、页面管理、Context reset、跨 origin 跳转拒绝、跨 origin iframe - 拒绝及两个 Session 的 Cookie 隔离; -2. 真实 API 固定矩阵使用 DeepSeek V4 Flash 和 Pro,分别从 Headless、raw PTY TUI、 - production Web GUI 与 ACP 进入,共八格。 - -每个真实 API cell 必须通过一次精确 `ToolSearch` 加载六个 deferred Browser tools, -再完成 Navigate、显式 Snapshot、Interact、Wait、Inspect 与 Page 管理,并返回只在 -表单提交后出现的隐藏 nonce。框架 retry 固定为 `0`。DOM 变化导致的 -`browser_snapshot_stale` 只有在同一轨迹随后成功执行新 Snapshot 和 Interact 时才允许; -其他 Browser 错误均阻断发布。 +```bash +REAL_API_RELEASE_MATRIX=1 bun run test:real-api +``` -production Web cell 同时运行驱动 Blade UI 的外层 Chromium 和 server 内由 Browser Tool -管理的内层 Chromium。所有 cell 结束后必须证明 BrowserContext、页面、Browser 进程、 -server、port、PTY、ACP connection、Session lease 与临时根均已回收,并全量检查 API -key 未进入 DOM、PTY、ACP update、tool result、metadata、transcript 或诊断。 +删除或重命名 inventory 文件后,runner 会在启动 Vitest 前失败。新增真实 API 轨迹也 +必须显式加入对应 inventory 和 source-contract 单测,避免目录变化造成静默缩减。 -该轨迹固定列入 `realApiQualification.files`,不能用 mock、仅 HTTP 成功、仅模型文本或 -keyless 测试替代。Chromium preflight 使用 `blade browser status`;缺失时只能显式执行 -`blade browser install`,资格测试本身不得下载浏览器。 +普通 `test:all` 和 CI 不会产生付费请求。国产模型通道默认不进入发布阻断;仅在显式设置 +`REAL_API_INCLUDE_OPTIONAL_PROVIDERS=1` 时作为可选渠道加入。 ## 准出证据 -每个独立 patch 至少保留以下证据: - -- 已冻结的 Qualified candidate 完整 SHA、patch 版本和日期; -- `bun run qualify:local` 的完整命令和退出码; -- `bun run qualify:production` 的完整命令、该 patch 所需 Flash/Pro × production - surface 矩阵逐项结果和退出码; -- browser preflight、process/lease/terminal/port/temp-root cleanup、omitted sentinel 与 - credential absence 的宿主断言; -- 失败时记录首个失败 cell、redacted bounded tail、清理结果和复跑事实;只有 source - 未变化的 Provider transient 才能整套重跑,不得用跳过测试替代通过; -- `git diff --check`、build、type-check、lint 的命令与退出码; -- evidence 只能在候选代码冻结并通过真实资格后创建。候选 SHA 到 tag HEAD 的唯一差异 - 必须是完整 evidence 文件;文件不得包含 `TBD`、`TODO`、`NOT RUN` 或预填 `PASS`。 - -真实 API 门禁会产生费用,因此不会被 `test:all` 或普通 CI 单元门禁隐式触发;发布候选、跨 provider 改动和 Agent runtime 核心改动必须显式运行。 - -当前生产准出覆盖桌面 TUI、CLI/headless、Web 和 ACP。移动端没有明确使用场景,暂不纳入实现与测试范围。 - -## 当前回合活动状态 - -当前回合活动的 release-blocking 矩阵固定运行 DeepSeek Flash/Pro × Headless、真实 ACP -stdio、raw PTY TUI 与 production Chromium Web,共八格。每格要求 framework retry `0`、 -model `maxRetries=0`、一次 Bash 调用、连续 thinking/tool/responding/clear 状态和精确终答。 -Web 必须在工具仍被 host barrier 阻塞时 reload,并从 SSE `connected.turnActivity` 恢复 -活动状态;重连快照断言完成后才释放工具,两个独立 SSE 探针都收到终态清除后才收集 -证据,不能用页面活动条消失代替探针到达。PTY 必须从真实 terminal capture 观察状态, -不得读取内部 store 代替。 - -常规路径需要两次 Provider 请求;只在第二次完整 SSE 响应为 `stop`、正文为空且没有工具 -增量时,才允许一次额外的空回复纠正。第三次请求必须只在原消息后追加精确的纠正文本, -该文本必须先以 `clientVisible=false`、`emptyFinalCorrection=true` 持久化并连接到唯一 -成功 Bash 结果,终态记录仍只能包含一次工具执行。非空、截断、不完整响应、重复请求、 -额外工具或第四次请求均不能作为该纠正通过。代理只记录有界响应计数与完成类别,不保留 -响应正文、推理文本或工具参数,且保持原始流式字节透传。 +每个独立 patch 至少保留: -独立八格恢复矩阵只对第二次上游请求设置固定文本提示与 `stop`,让真实 Provider 产生 -空正文;不替换响应,后续请求不注入。必须观察一次持久化纠正、精确终答及一次真实 Bash -副作用。关闭注入的负向对照必须因缺失纠正证据而失败,不能靠 framework retry 取绿灯。 +- 冻结候选的完整 SHA、版本和日期; +- `bun run qualify:local` 的命令、结果和退出码; +- `bun run qualify:production` 的命令、逐文件结果和退出码; +- browser preflight、process/lease/terminal/port/temp-root cleanup 与凭据缺失断言; +- 首个失败 cell 的有界脱敏诊断及清理结果; +- `git diff --check`、build、type-check 和 lint 结果。 -详细命令、逐格耗时、隐私与清理结果见 -[当前回合活动状态资格验证证据](./turn-activity-surface-evidence.md)。 +只有源码未变化的 Provider transient 才允许整套重跑。跳过测试、模型文本或预填 +`PASS` 不能作为资格证据。 diff --git a/package.json b/package.json index 44290f897..0e40268be 100644 --- a/package.json +++ b/package.json @@ -71,11 +71,7 @@ "devDependencies": { "@biomejs/biome": "2.5.7", "@types/node": "^25.5.0", - "@types/react": "19.2.10", - "@types/react-dom": "19.2.3", "knip": "^5.80.0", - "react": "19.2.4", - "react-dom": "19.2.4", "typescript": "^5.9.2" }, "trustedDependencies": [ diff --git a/packages/cli/scripts/build.ts b/packages/cli/scripts/build.ts index a5060289f..1d075f12e 100644 --- a/packages/cli/scripts/build.ts +++ b/packages/cli/scripts/build.ts @@ -1,6 +1,6 @@ import { spawn } from "node:child_process"; import { existsSync } from "node:fs"; -import { dirname, join } from "node:path"; +import { dirname, join, resolve } from "node:path"; import { fileURLToPath } from "node:url"; import { createWebBuildEnvironment } from "./buildEnvironment.js"; @@ -17,6 +17,20 @@ const externals = [ console.log("Building backend..."); +const rawTextPlugin: Bun.BunPlugin = { + name: "raw-text", + setup(build) { + build.onResolve({ filter: /\.(?:md|sql)\?raw$/ }, args => ({ + path: resolve(dirname(args.importer), args.path.slice(0, -4)), + namespace: "raw-text", + })); + build.onLoad({ filter: /.*/, namespace: "raw-text" }, async args => ({ + contents: await Bun.file(args.path).text(), + loader: "text", + })); + }, +}; + const result = await Bun.build({ entrypoints: ["src/blade.tsx"], outdir: "dist", @@ -24,7 +38,8 @@ const result = await Bun.build({ format: "esm", splitting: true, minify: true, - external: externals + external: externals, + plugins: [rawTextPlugin], }); if (!result.success) { diff --git a/packages/cli/scripts/test-config.d.ts b/packages/cli/scripts/test-config.d.ts index 7fc40f794..4fce7c811 100644 --- a/packages/cli/scripts/test-config.d.ts +++ b/packages/cli/scripts/test-config.d.ts @@ -15,6 +15,11 @@ export const testTypes: Record & { realApiQualification: TestTypeConfig; }; +export function assertConfiguredTestFilesExist( + config: Pick, + rootDirectory: string +): void; + export function resolveTestTimeout( config: Pick, options: { coverage?: boolean } diff --git a/packages/cli/scripts/test-config.js b/packages/cli/scripts/test-config.js index ee82ac258..879cc3981 100644 --- a/packages/cli/scripts/test-config.js +++ b/packages/cli/scripts/test-config.js @@ -1,3 +1,6 @@ +import { statSync } from 'node:fs'; +import path from 'node:path'; + export const testTypes = { unit: { name: '单元测试', @@ -15,6 +18,19 @@ export const testTypes = { project: 'real-api', timeout: 60 * 60 * 1_000, requiresProductionBuild: true, + files: [ + 'tests/integration/real-api/acp-remote-filesystem-trajectory.test.ts', + 'tests/integration/real-api/agent-trajectory.test.ts', + 'tests/integration/real-api/browser-tool-trajectory.test.ts', + 'tests/integration/real-api/cross-provider-fallback-trajectory.test.ts', + 'tests/integration/real-api/durable-interaction-recovery-trajectory.test.ts', + 'tests/integration/real-api/goal-mode-trajectory.test.ts', + 'tests/integration/real-api/goal-paused-usage-trajectory.test.ts', + 'tests/integration/real-api/release-coding-trajectory.test.ts', + 'tests/integration/real-api/structured-output-trajectory.test.ts', + 'tests/integration/real-api/task-list-team-trajectory.test.ts', + 'tests/integration/real-api/workspace-agent-resources-trajectory.test.ts', + ], env: { REAL_API_TEST: '1', }, @@ -27,53 +43,13 @@ export const testTypes = { files: [ 'tests/integration/real-api/agent-trajectory.test.ts', 'tests/integration/real-api/structured-output-trajectory.test.ts', - 'tests/integration/real-api/code-review-trajectory.test.tsx', 'tests/integration/real-api/durable-interaction-recovery-trajectory.test.ts', - 'tests/integration/real-api/acp-session-fork-trajectory.test.ts', 'tests/integration/real-api/release-coding-trajectory.test.ts', 'tests/integration/real-api/task-list-team-trajectory.test.ts', - 'tests/integration/real-api/provider-retry-trajectory.test.ts', 'tests/integration/real-api/cross-provider-fallback-trajectory.test.ts', - 'tests/integration/real-api/provider-attempt-deadline-web-trajectory.test.ts', - 'tests/integration/real-api/prompt-cache-surface-trajectory.test.ts', - 'tests/integration/real-api/action-stationarity-trajectory.test.ts', 'tests/integration/real-api/goal-mode-trajectory.test.ts', - 'tests/integration/real-api/root-turn-auto-resume-trajectory.test.ts', - 'tests/integration/real-api/goal-finalization-handoff-trajectory.test.ts', - 'tests/integration/real-api/subagent-result-adoption-trajectory.test.ts', - 'tests/integration/real-api/background-subagent-completion-trajectory.test.ts', - 'tests/integration/real-api/durable-task-unread-trajectory.test.ts', - 'tests/integration/real-api/tui-task-attention-trajectory.test.ts', - 'tests/integration/real-api/foreground-bounded-output-trajectory.test.ts', - 'tests/integration/real-api/foreground-command-handoff-trajectory.test.ts', - 'tests/integration/real-api/token-budget-handoff-trajectory.test.ts', - 'tests/integration/real-api/browser-preview-trajectory.test.ts', 'tests/integration/real-api/browser-tool-trajectory.test.ts', - 'tests/integration/real-api/large-prompt-offload-trajectory.test.ts', - 'tests/integration/real-api/compaction-rich-media-trajectory.test.ts', - 'tests/integration/real-api/foreground-provider-recovery-trajectory.test.ts', - 'tests/integration/real-api/provider-rate-limit-cooldown-trajectory.test.ts', - 'tests/integration/real-api/turn-activity-surface-trajectory.test.ts', - 'tests/integration/real-api/provider-request-admission-acp-trajectory.test.ts', - 'tests/integration/real-api/provider-request-admission-web-trajectory.test.ts', 'tests/integration/real-api/acp-remote-filesystem-trajectory.test.ts', - 'tests/integration/real-api/weighted-provider-admission-acp-trajectory.test.ts', - 'tests/integration/real-api/weighted-provider-admission-web-trajectory.test.ts', - 'tests/integration/real-api/weighted-task-admission-acp-trajectory.test.ts', - 'tests/integration/real-api/weighted-task-admission-web-trajectory.test.ts', - 'tests/integration/real-api/keyed-coordination-reclamation-trajectory.test.ts', - 'tests/integration/real-api/session-runtime-residency-acp-trajectory.test.ts', - 'tests/integration/real-api/session-runtime-residency-controls-trajectory.test.ts', - 'tests/integration/real-api/session-runtime-residency-web-trajectory.test.ts', - 'tests/integration/real-api/graceful-shutdown-trajectory.test.ts', - 'tests/integration/real-api/tool-admission-trajectory.test.ts', - 'tests/integration/real-api/side-conversation-trajectory.test.ts', - 'tests/integration/real-api/follow-up-queue-trajectory.test.ts', - 'tests/integration/real-api/compaction-memory-consolidation-trajectory.test.ts', - 'tests/integration/real-api/goal-execution-host-failure-trajectory.test.ts', - 'tests/integration/real-api/goal-turn-lineage-trajectory.test.ts', - 'tests/integration/real-api/goal-paused-usage-trajectory.test.ts', - 'tests/integration/real-api/textual-tool-call-trajectory.test.ts', ], env: { REAL_API_TEST: '1', @@ -92,15 +68,17 @@ export const testTypes = { timeout: 120_000, requiresProductionBuild: true, files: [ - 'tests/unit/cli/headless.test.ts', - 'tests/unit/cli/headless-events.test.ts', + 'tests/unit/cli/headless-boundaries.test.ts', + 'tests/unit/cli/headless-event-contract.test.ts', 'tests/integration/cli/blade-help.test.ts', 'tests/unit/agent-runtime/context/jsonl-recovery.test.ts', + 'tests/unit/agent-runtime/agent/active-turn-mailbox.test.ts', 'tests/unit/agent-runtime/agent/session-lease.test.ts', - 'tests/unit/agent-runtime/agent/session-runtime.test.ts', + 'tests/unit/agent-runtime/agent/completion-policy.test.ts', 'tests/unit/agent-runtime/agent/subagent-registry.test.ts', - 'tests/unit/agent-runtime/server/session-routes.test.ts', - 'tests/unit/agent-runtime/acp/session.test.ts', + 'tests/unit/agent-runtime/server/task-routes.test.ts', + 'tests/unit/agent-runtime/acp/bladeAgent.test.ts', + 'tests/unit/services/session-interaction-recovery.test.ts', ], }, e2e: { @@ -136,6 +114,20 @@ export const testTypes = { }, }; +export function assertConfiguredTestFilesExist(config, rootDirectory) { + for (const file of config.files ?? []) { + let exists = false; + try { + exists = statSync(path.resolve(rootDirectory, file)).isFile(); + } catch { + // Report one stable configuration error below. + } + if (!exists) { + throw new Error(`${config.name} contains missing test file: ${file}`); + } + } +} + export function resolveTestTimeout(config, options) { return options.coverage ? (config.coverageTimeout ?? config.timeout) diff --git a/packages/cli/scripts/test.js b/packages/cli/scripts/test.js index 683477f0a..438097518 100755 --- a/packages/cli/scripts/test.js +++ b/packages/cli/scripts/test.js @@ -5,6 +5,7 @@ import os from 'node:os'; import path from 'path'; import { fileURLToPath } from 'url'; import { + assertConfiguredTestFilesExist, createTestExecutionStages, resolveTestTimeout, testTypes, @@ -103,6 +104,8 @@ async function runTest(testType, options = {}, requestedFiles = []) { process.exit(1); } + assertConfiguredTestFilesExist(config, path.join(__dirname, '..')); + const baseArgs = []; if (options.watch) { baseArgs.push('--watch'); diff --git a/packages/cli/src/acp/AcpFileRequestCoordinator.contracts.ts b/packages/cli/src/acp/AcpFileRequestCoordinator.contracts.ts index dea975f25..868468137 100644 --- a/packages/cli/src/acp/AcpFileRequestCoordinator.contracts.ts +++ b/packages/cli/src/acp/AcpFileRequestCoordinator.contracts.ts @@ -116,9 +116,3 @@ export function createAcpRemoteConnectionPathIdentity( .update(remotePath.collisionIdentity) .digest('hex')}`; } - -export function isAcpRemoteMutationRecoveryLease( - lease: AcpRemoteMutationLease | AcpRemoteMutationRecoveryLease | undefined -): lease is AcpRemoteMutationRecoveryLease { - return lease !== undefined && 'finish' in lease; -} diff --git a/packages/cli/src/acp/AcpFileSystemService.ts b/packages/cli/src/acp/AcpFileSystemService.ts index 554e19a0e..e767e9c10 100644 --- a/packages/cli/src/acp/AcpFileSystemService.ts +++ b/packages/cli/src/acp/AcpFileSystemService.ts @@ -1,9 +1,4 @@ -/** - * ACP 文件系统服务适配器 - * - * 将文件操作转发给 IDE(ACP Client)执行。 - * 当 IDE 声明支持 fs 能力时,可以使用此服务替代本地文件操作。 - */ +/** ACP 文件系统服务适配器 将文件操作转发给 IDE(ACP Client)执行。 当 IDE 声明支持 fs 能力时,可以使用此服务替代本地文件操作。 */ import { createHash } from 'node:crypto'; import type { @@ -115,11 +110,7 @@ export class AcpFileSystemService implements FileSystemService { this.pathProfile = cloneAcpRemotePathProfile(pathProfile); } - /** - * 读取文本文件 - * - * 如果 IDE 不支持 readTextFile,则 fail-closed。 - */ + /** 读取文本文件 如果 IDE 不支持 readTextFile,则 fail-closed。 */ async readTextFile( filePath: string, options?: { @@ -197,11 +188,7 @@ export class AcpFileSystemService implements FileSystemService { } } - /** - * 写入文本文件 - * - * 如果 IDE 不支持 writeTextFile,则 fail-closed。 - */ + /** 写入文本文件 如果 IDE 不支持 writeTextFile,则 fail-closed。 */ async writeTextFile( filePath: string, content: string, @@ -284,11 +271,7 @@ export class AcpFileSystemService implements FileSystemService { } } - /** - * 检查文件是否存在 - * - * 只通过 ACP 远端 read 判断文件存在性;缺少 read 能力时 fail-closed。 - */ + /** 检查文件是否存在 只通过 ACP 远端 read 判断文件存在性;缺少 read 能力时 fail-closed。 */ async exists(filePath: string): Promise { if (!this.capabilities.readTextFile) { throw new AcpFileSystemCapabilityError('readTextFile'); @@ -308,31 +291,19 @@ export class AcpFileSystemService implements FileSystemService { } } - /** - * 读取二进制文件 - * - * ACP 协议目前只支持文本文件读取,二进制文件 fail-closed。 - */ + /** 读取二进制文件 ACP 协议目前只支持文本文件读取,二进制文件 fail-closed。 */ async readBinaryFile(filePath: string): Promise { void filePath; throw new AcpFileSystemCapabilityError('readBinaryFile'); } - /** - * 获取文件统计信息 - * - * ACP 协议暂不支持 stat 操作,fail-closed。 - */ + /** 获取文件统计信息 ACP 协议暂不支持 stat 操作,fail-closed。 */ async stat(filePath: string): Promise { void filePath; throw new AcpFileSystemCapabilityError('stat'); } - /** - * 创建目录 - * - * ACP 协议暂不支持 mkdir 操作,fail-closed。 - */ + /** 创建目录 ACP 协议暂不支持 mkdir 操作,fail-closed。 */ async mkdir( dirPath: string, _options?: { recursive?: boolean; mode?: number } @@ -341,9 +312,7 @@ export class AcpFileSystemService implements FileSystemService { throw new AcpFileSystemCapabilityError('mkdir'); } - /** - * 获取 IDE 支持的文件系统能力 - */ + /** 获取 IDE 支持的文件系统能力 */ getCapabilities(): FileSystemCapabilities { return { ...this.capabilities }; } @@ -356,16 +325,12 @@ export class AcpFileSystemService implements FileSystemService { return parseAcpRemotePath(filePath, this.pathProfile.style); } - /** - * 检查是否支持读取文件 - */ + /** 检查是否支持读取文件 */ canReadTextFile(): boolean { return this.capabilities.readTextFile ?? false; } - /** - * 检查是否支持写入文件 - */ + /** 检查是否支持写入文件 */ canWriteTextFile(): boolean { return this.capabilities.writeTextFile ?? false; } @@ -397,20 +362,6 @@ export class AcpFileSystemService implements FileSystemService { .digest('hex')}`; } - async readTextFileIfExists( - filePath: string, - options?: { - signal?: AbortSignal; - deadlineAt?: number; - purpose?: AcpRemoteFileRequestPurpose; - userReadPermit?: AcpRemoteUserReadPermit; - lease?: AcpRemoteMutationLease | AcpRemoteMutationRecoveryLease; - } - ): Promise<{ exists: false } | { exists: true; content: string }> { - const remotePath = this.parsePath(filePath); - return this.readTextFileIfExistsForParsedPath(remotePath, options); - } - async readTextFileIfExistsForParsedPath( remotePath: AcpRemotePath, options?: { diff --git a/packages/cli/src/acp/AcpServiceContext.ts b/packages/cli/src/acp/AcpServiceContext.ts index cfd49f918..6f40adc31 100644 --- a/packages/cli/src/acp/AcpServiceContext.ts +++ b/packages/cli/src/acp/AcpServiceContext.ts @@ -1,17 +1,9 @@ -/** - * ACP 服务上下文管理器 - * - * 管理 ACP 模式下的各种服务(文件系统、终端等), - * 使工具可以透明地使用 IDE 提供的能力或回退到本地实现。 - */ +/** ACP 服务上下文管理器 管理 ACP 模式下的各种服务(文件系统、终端等), 使工具可以透明地使用 IDE 提供的能力或回退到本地实现。 */ import type { AgentSideConnection, ClientCapabilities, SessionNotification, - ToolCallContent, - ToolCallStatus, - ToolKind, } from '@agentclientprotocol/sdk'; import { type ForegroundProcessOwnership, @@ -44,9 +36,7 @@ import { parseAcpRemoteWorkspaceDescriptor } from './AcpRemoteWorkspace.js'; const logger = createLogger(LogCategory.AGENT); const ACP_TERMINAL_OUTPUT_READ_TIMEOUT_MS = 5_000; -/** - * 终端服务接口 - */ +/** 终端服务接口 */ export interface TerminalService { /** * 执行命令 @@ -59,9 +49,7 @@ export interface TerminalService { options?: TerminalExecuteOptions ): Promise; - /** - * 检查是否支持终端操作 - */ + /** 检查是否支持终端操作 */ isAvailable(): boolean; } @@ -103,9 +91,7 @@ export interface TerminalExecuteResult { }; } -/** - * 本地终端服务(使用 child_process) - */ +/** 本地终端服务(使用 child_process) */ class LocalTerminalService implements TerminalService { constructor(private readonly defaultCwd?: string) {} @@ -328,10 +314,7 @@ class UnavailableFileSystemService implements FileSystemService { } } -/** - * ACP 终端服务 - * 通过 ACP 协议在 IDE 中执行命令 - */ +/** ACP 终端服务 通过 ACP 协议在 IDE 中执行命令 */ class AcpTerminalService implements TerminalService { constructor( private readonly connection: AgentSideConnection, @@ -719,9 +702,7 @@ class AcpTerminalService implements TerminalService { } } -/** - * 单个会话的服务上下文 - */ +/** 单个会话的服务上下文 */ interface SessionServices { fileSystemService: FileSystemService; terminalService: TerminalService; @@ -764,12 +745,7 @@ function remoteSurfaceOwnerKey( return JSON.stringify([sessionId, exactIdentity]); } -/** - * ACP 服务上下文管理器 - * - * 按 sessionId 管理服务,支持多会话并发。 - * 每个会话有独立的服务实例,互不影响。 - */ +/** ACP 服务上下文管理器 按 sessionId 管理服务,支持多会话并发。 每个会话有独立的服务实例,互不影响。 */ export class AcpServiceContext { private static sessions: Map = new Map(); private static remoteSurfaceOwners: Map = @@ -778,18 +754,6 @@ export class AcpServiceContext { private static sessionRegistrationGeneration = 0n; private static currentSessionId: string | null = null; - private constructor() { - // 私有构造函数,使用静态方法 - } - - /** - * 获取单例实例(兼容旧 API) - * @deprecated 使用 getForSession(sessionId) 代替 - */ - static getInstance(): AcpServiceContext { - return new AcpServiceContext(); - } - /** * 初始化会话的 ACP 服务 * @@ -892,11 +856,7 @@ export class AcpServiceContext { return registration; } - /** - * 销毁会话服务 - * - * 只清理指定会话,不影响其他会话。 - */ + /** 销毁会话服务 只清理指定会话,不影响其他会话。 */ static destroySession(sessionId: string): void { const ownerKey = AcpServiceContext.remoteSurfaceOwnerKeysBySessionId.get(sessionId); if (ownerKey) { @@ -927,25 +887,19 @@ export class AcpServiceContext { AcpServiceContext.destroySession(registration.sessionId); } - /** - * 获取指定会话的服务 - */ + /** 获取指定会话的服务 */ static getSessionServices(sessionId: string): SessionServices | null { return AcpServiceContext.sessions.get(sessionId) || null; } - /** - * 设置当前活跃会话 - */ + /** 设置当前活跃会话 */ static setCurrentSession(sessionId: string): void { if (AcpServiceContext.sessions.has(sessionId)) { AcpServiceContext.currentSessionId = sessionId; } } - /** - * 获取当前活跃会话 ID - */ + /** 获取当前活跃会话 ID */ static getCurrentSessionId(): string | null { return AcpServiceContext.currentSessionId; } @@ -954,6 +908,35 @@ export class AcpServiceContext { return AcpServiceContext.sessions.get(sessionId)?.remoteFileSystem ?? false; } + static isAcpMode(): boolean { + return AcpServiceContext.currentSessionId !== null; + } + + static getFileSystemService(sessionId?: string): FileSystemService { + if (sessionId !== undefined) { + return ( + AcpServiceContext.sessions.get(sessionId)?.fileSystemService ?? + new UnavailableFileSystemService() + ); + } + const currentSessionId = AcpServiceContext.currentSessionId; + if (currentSessionId) { + const services = AcpServiceContext.sessions.get(currentSessionId); + if (services) return services.fileSystemService; + } + return new LocalFileSystemService(); + } + + static getTerminalService(sessionId?: string): TerminalService { + const targetSessionId = sessionId ?? AcpServiceContext.currentSessionId; + if (targetSessionId) { + const services = AcpServiceContext.sessions.get(targetSessionId); + if (services) return services.terminalService; + if (sessionId) return new UnavailableTerminalService(); + } + return new LocalTerminalService(); + } + static getRemoteSurfaceOwnerSnapshot( sessionId: string, descriptor: AcpRemoteWorkspaceDescriptorV1 @@ -983,159 +966,22 @@ export class AcpServiceContext { terminal: binding.terminal, }; } - - // ==================== 兼容旧 API(实例方法)==================== - - /** - * 初始化 ACP 服务(兼容旧 API) - * @deprecated 使用 AcpServiceContext.initializeSession() 代替 - */ - initialize( - connection: AgentSideConnection, - sessionId: string, - clientCapabilities: ClientCapabilities | undefined, - cwd?: string - ): void { - AcpServiceContext.initializeSession( - connection, - sessionId, - clientCapabilities, - cwd || getCwd() - ); - } - - /** - * 重置服务(兼容旧 API) - * @deprecated 使用 AcpServiceContext.destroySession(sessionId) 代替 - */ - reset(): void { - // 只重置当前会话,而不是所有会话 - if (AcpServiceContext.currentSessionId) { - AcpServiceContext.destroySession(AcpServiceContext.currentSessionId); - } - } - - /** - * 检查是否在 ACP 模式下运行 - */ - isAcpMode(): boolean { - return AcpServiceContext.currentSessionId !== null; - } - - /** - * 获取文件系统服务(当前会话) - */ - getFileSystemService(sessionId?: string): FileSystemService { - if (sessionId !== undefined) { - return ( - AcpServiceContext.sessions.get(sessionId)?.fileSystemService ?? - new UnavailableFileSystemService() - ); - } - const targetSessionId = sessionId ?? AcpServiceContext.currentSessionId; - if (targetSessionId) { - const services = AcpServiceContext.sessions.get(targetSessionId); - if (services) return services.fileSystemService; - } - return new LocalFileSystemService(); - } - - /** - * 获取终端服务(当前会话) - */ - getTerminalService(sessionId?: string): TerminalService { - const targetSessionId = sessionId ?? AcpServiceContext.currentSessionId; - if (targetSessionId) { - const services = AcpServiceContext.sessions.get(targetSessionId); - if (services) return services.terminalService; - if (sessionId) return new UnavailableTerminalService(); - } - return new LocalTerminalService(); - } - - /** - * 获取 ACP 连接(当前会话) - */ - getConnection(): AgentSideConnection | null { - if (AcpServiceContext.currentSessionId) { - const services = AcpServiceContext.sessions.get( - AcpServiceContext.currentSessionId - ); - if (services) return services.connection; - } - return null; - } - - /** - * 获取当前会话 ID - */ - getSessionId(): string | null { - return AcpServiceContext.currentSessionId; - } - - /** - * 获取客户端能力(当前会话) - */ - getClientCapabilities(): ClientCapabilities | null { - if (AcpServiceContext.currentSessionId) { - const services = AcpServiceContext.sessions.get( - AcpServiceContext.currentSessionId - ); - if (services) return services.clientCapabilities; - } - return null; - } - - /** - * 发送工具调用状态更新 - */ - async sendToolUpdate( - toolCallId: string, - status: ToolCallStatus, - title: string, - content?: ToolCallContent[], - kind?: ToolKind - ): Promise { - const sessionId = AcpServiceContext.currentSessionId; - if (!sessionId) return; - - const services = AcpServiceContext.sessions.get(sessionId); - if (!services) return; - - if (!services.sendUpdate) return; - try { - await services.sendUpdate({ - sessionUpdate: 'tool_call', - toolCallId, - status, - title, - content: content || [], - kind: kind || 'other', - }); - } catch (error) { - logger.warn('[AcpServiceContext] Failed to send tool update:', error); - } - } } -/** - * 便捷函数:获取终端服务 - */ +/** 便捷函数:获取终端服务 */ export function getAcpFileSystemService(sessionId?: string): FileSystemService { - return AcpServiceContext.getInstance().getFileSystemService(sessionId); + return AcpServiceContext.getFileSystemService(sessionId); } export function getTerminalService(sessionId?: string): TerminalService { - return AcpServiceContext.getInstance().getTerminalService(sessionId); + return AcpServiceContext.getTerminalService(sessionId); } -/** - * 便捷函数:检查是否在 ACP 模式 - */ +/** 便捷函数:检查是否在 ACP 模式 */ export function isAcpMode(sessionId?: string): boolean { return sessionId ? AcpServiceContext.getSessionServices(sessionId) !== null - : AcpServiceContext.getInstance().isAcpMode(); + : AcpServiceContext.isAcpMode(); } export function isAcpRemoteFileSystem(sessionId?: string): boolean { diff --git a/packages/cli/src/acp/BladeAgent.ts b/packages/cli/src/acp/BladeAgent.ts index b3e9e143a..fa448bc70 100644 --- a/packages/cli/src/acp/BladeAgent.ts +++ b/packages/cli/src/acp/BladeAgent.ts @@ -1,9 +1,4 @@ -/** - * Blade ACP Agent 实现 - * - * 实现 ACP 协议的 Agent 接口,使 Blade 可以被 Zed、JetBrains 等编辑器调用。 - * - */ +/** Blade ACP Agent 实现 实现 ACP 协议的 Agent 接口,使 Blade 可以被 Zed、JetBrains 等编辑器调用。 */ import path from 'node:path'; import type * as acp from '@agentclientprotocol/sdk'; @@ -168,11 +163,7 @@ function remoteSessionCwd(metadata: SessionMetadata): string { return metadata.remoteWorkspace.wirePath; } -/** - * Blade ACP Agent - * - * 实现 ACP 协议的 Agent 接口,处理来自 IDE 的请求。 - */ +/** Blade ACP Agent 实现 ACP 协议的 Agent 接口,处理来自 IDE 的请求。 */ export class BladeAgent implements AcpAgentInterface { private sessions: Map = new Map(); private sessionLoadQueues: Map> = new Map(); @@ -221,9 +212,7 @@ export class BladeAgent implements AcpAgentInterface { } } - /** - * 初始化连接,协商协议版本和能力 - */ + /** 初始化连接,协商协议版本和能力 */ async initialize(params: acp.InitializeRequest): Promise { logger.info('[BladeAgent] Initializing ACP connection'); logger.debug( @@ -258,9 +247,7 @@ export class BladeAgent implements AcpAgentInterface { }; } - /** - * 认证(Blade 目前不需要认证) - */ + /** 认证(Blade 目前不需要认证) */ async authenticate( _params: acp.AuthenticateRequest ): Promise { @@ -268,9 +255,7 @@ export class BladeAgent implements AcpAgentInterface { return; } - /** - * 创建新会话 - */ + /** 创建新会话 */ async newSession(params: acp.NewSessionRequest): Promise { this.assertNotDestroyed(); const sessionId = createSessionId('acp'); @@ -513,9 +498,7 @@ export class BladeAgent implements AcpAgentInterface { } } - /** - * 恢复持久化会话并在响应前按协议回放历史。 - */ + /** 恢复持久化会话并在响应前按协议回放历史。 */ async loadSession(params: acp.LoadSessionRequest): Promise { this.assertNotDestroyed(); logger.info(`[BladeAgent] Loading session: ${params.sessionId}`); @@ -904,9 +887,7 @@ export class BladeAgent implements AcpAgentInterface { } } - /** - * 处理提示请求 - */ + /** 处理提示请求 */ async prompt(params: acp.PromptRequest): Promise { const lease = this.runtimeResidency.acquire(params.sessionId); if (!lease) { @@ -919,9 +900,7 @@ export class BladeAgent implements AcpAgentInterface { } } - /** - * 取消当前操作 - */ + /** 取消当前操作 */ async cancel(params: acp.CancelNotification): Promise { logger.info( `[BladeAgent] Cancel notification received for session: ${params.sessionId}` @@ -939,9 +918,7 @@ export class BladeAgent implements AcpAgentInterface { } } - /** - * 设置会话模式(权限模式) - */ + /** 设置会话模式(权限模式) */ async setSessionMode( params: acp.SetSessionModeRequest ): Promise { @@ -957,9 +934,7 @@ export class BladeAgent implements AcpAgentInterface { return {}; } - /** - * 设置会话配置选项(如模型切换) - */ + /** 设置会话配置选项(如模型切换) */ async setSessionConfigOption?( params: acp.SetSessionConfigOptionRequest ): Promise { @@ -1017,9 +992,7 @@ export class BladeAgent implements AcpAgentInterface { } } - /** - * 清理资源 - */ + /** 清理资源 */ destroy(): Promise { if (this.destroyPromise) return this.destroyPromise; this.destroyPromise = this.destroyOwnedResources(); diff --git a/packages/cli/src/acp/Session.ts b/packages/cli/src/acp/Session.ts index c937119a2..cc5ff0b0e 100644 --- a/packages/cli/src/acp/Session.ts +++ b/packages/cli/src/acp/Session.ts @@ -1,9 +1,4 @@ -/** - * ACP 会话管理 - * - * 封装 Blade Agent,处理 ACP 协议的 prompt 请求, - * 将 Agent 的流式输出转发给 IDE。 - */ +/** ACP 会话管理 封装 Blade Agent,处理 ACP 协议的 prompt 请求, 将 Agent 的流式输出转发给 IDE。 */ import { type AgentSideConnection, @@ -77,6 +72,7 @@ import type { McpElicitationField, } from '../mcp/McpElicitation.js'; import { Bus } from '../server/bus.js'; +import { projectSessionLoopEvent } from '../server/routes/sessionLoopEventProjection.js'; import type { ContentPart, Message } from '../services/ChatServiceInterface.js'; import { CodeReviewService, renderCodeReview } from '../services/CodeReviewService.js'; import { isClientVisibleMessage } from '../services/clientMessageVisibility.js'; @@ -132,15 +128,8 @@ const logger = createLogger(LogCategory.AGENT); type AcpPendingResumeKind = 'pending_input' | 'goal'; -/** - * ACP 会话类 - * - * 每个会话对应一个 Blade Agent 实例, - * 处理来自 IDE 的 prompt 请求并返回流式响应。 - */ -/** - * ACP 模式 ID(与 BladeAgent 返回的 availableModes 对应) - */ +/** ACP 会话类 每个会话对应一个 Blade Agent 实例, 处理来自 IDE 的 prompt 请求并返回流式响应。 */ +/** ACP 模式 ID(与 BladeAgent 返回的 availableModes 对应) */ export type AcpModeId = 'default' | 'auto-edit' | 'yolo' | 'plan'; export type AcpSessionRoots = @@ -488,10 +477,7 @@ export class AcpSession { }); } - /** - * 初始化会话 - * 创建 Blade Agent 实例并初始化 ACP 服务 - */ + /** 初始化会话 创建 Blade Agent 实例并初始化 ACP 服务 */ async initialize(): Promise { logger.debug(`[AcpSession ${this.id}] Initializing...`); await this.persistPermissionMode( @@ -534,7 +520,7 @@ export class AcpSession { const mcpServers = this.options.mcpServers ? toMcpServers(this.options.mcpServers) : undefined; - const terminalService = AcpServiceContext.getInstance().getTerminalService(this.id); + const terminalService = AcpServiceContext.getTerminalService(this.id); const workspace: SessionWorkspace | undefined = this.roots.kind === 'acp-remote' ? { @@ -819,8 +805,7 @@ export class AcpSession { clearTimeout(this.availableCommandsTimer); } - // 延迟发送,确保在 session/new 响应之后 - // 使用较长的延迟确保 Zed 已准备好接收 + // 延迟发送,确保在 session/new 响应之后 使用较长的延迟确保 Zed 已准备好接收 logger.debug( `[AcpSession ${this.id}] Scheduling available commands update (500ms delay)` ); @@ -845,9 +830,7 @@ export class AcpSession { await this.sendAvailableCommands(); } - /** - * 处理 slash command - */ + /** 处理 slash command */ private async handleSlashCommand( message: string, signal: AbortSignal @@ -1173,8 +1156,7 @@ export class AcpSession { if (persistedMessages.length > 0) this.messages = persistedMessages; } - // 发送结果给 IDE - // 优先使用 content(完整内容),否则使用 message(简短状态) + // 发送结果给 IDE 优先使用 content(完整内容),否则使用 message(简短状态) const displayContent = result.content || result.message; if (displayContent) { this.sendUpdate({ @@ -1196,8 +1178,7 @@ export class AcpSession { if (signal.aborted || (error instanceof Error && error.name === 'AbortError')) { return { stopReason: 'cancelled' }; } - // 注意:abortHandler 在 try 块内定义,catch 无法直接访问 - // 但由于 signal 是 WeakRef 的,GC 会自动清理 + // 注意:abortHandler 在 try 块内定义,catch 无法直接访问 但由于 signal 是 WeakRef 的,GC 会自动清理 logger.error(`[AcpSession ${this.id}] Slash command error:`, error); this.sendUpdate({ sessionUpdate: 'agent_message_chunk', @@ -1471,8 +1452,7 @@ export class AcpSession { }, }; - // 4. 调用 Agent chatStream(Phase 4: 事件驱动消费) - // stream_end 不外发给 ACP 客户端(保持内部语义) + // 4. 调用 Agent chatStream(Phase 4: 事件驱动消费) stream_end 不外发给 ACP 客户端(保持内部语义) const loopResult = await drainLoop( this.agent.chatStream(message, context, { pendingInputOnly: internalOptions.pendingInputOnly, @@ -1736,102 +1716,18 @@ export class AcpSession { break; case 'provider_admission': providerAdmissionVisible = event.phase !== 'admitted'; - this.sendUpdate({ - sessionUpdate: 'session_info_update', - updatedAt: new Date().toISOString(), - _meta: { - 'blade/providerAdmission': - event.phase !== 'admitted' - ? { - phase: event.phase, - requestClass: event.requestClass, - resource: event.resource, - scope: event.scope, - ...(event.reason !== undefined - ? { reason: event.reason } - : {}), - queuePosition: event.queuePosition, - queueDepth: event.queueDepth, - inFlight: event.inFlight, - limit: event.limit, - waitMs: event.waitMs, - maxWaitMs: event.maxWaitMs, - ...(event.recoveryRemainingMs !== undefined - ? { - recoveryRemainingMs: event.recoveryRemainingMs, - } - : {}), - } - : null, - }, - }); + this.sendSessionMetadata( + 'blade/providerAdmission', + event.phase === 'admitted' + ? null + : projectSessionLoopEvent(event, { omitUndefined: true })?.properties + ); break; case 'provider_circuit': - this.sendUpdate({ - sessionUpdate: 'session_info_update', - updatedAt: new Date().toISOString(), - _meta: { - 'blade/providerCircuit': { - phase: event.phase, - reason: event.reason, - ...(event.statusCode !== undefined - ? { statusCode: event.statusCode } - : {}), - ...(event.retryAfterMs !== undefined - ? { retryAfterMs: event.retryAfterMs } - : {}), - ...(event.nextProbeAt !== undefined - ? { nextProbeAt: event.nextProbeAt } - : {}), - openDurationMs: event.openDurationMs, - ...(event.sampleCount !== undefined - ? { sampleCount: event.sampleCount } - : {}), - ...(event.failureCount !== undefined - ? { failureCount: event.failureCount } - : {}), - ...(event.recoveryRemainingMs !== undefined - ? { - recoveryRemainingMs: event.recoveryRemainingMs, - } - : {}), - }, - }, - }); + this.sendProjectedLoopMetadata('blade/providerCircuit', event); break; case 'provider_retry': - this.sendUpdate({ - sessionUpdate: 'session_info_update', - updatedAt: new Date().toISOString(), - _meta: { - 'blade/providerRetry': { - phase: event.phase, - attempt: event.attempt, - maxRetries: event.maxRetries, - reason: event.reason, - ...(event.statusCode !== undefined - ? { statusCode: event.statusCode } - : {}), - ...(event.delayMs !== undefined ? { delayMs: event.delayMs } : {}), - ...(event.nextRetryAt !== undefined - ? { nextRetryAt: event.nextRetryAt } - : {}), - ...(event.mode !== undefined ? { mode: event.mode } : {}), - ...(event.recoveryBudgetMs !== undefined - ? { recoveryBudgetMs: event.recoveryBudgetMs } - : {}), - ...(event.recoveryElapsedMs !== undefined - ? { recoveryElapsedMs: event.recoveryElapsedMs } - : {}), - ...(event.recoveryRemainingMs !== undefined - ? { recoveryRemainingMs: event.recoveryRemainingMs } - : {}), - ...(event.exhaustedBy !== undefined - ? { exhaustedBy: event.exhaustedBy } - : {}), - }, - }, - }); + this.sendProjectedLoopMetadata('blade/providerRetry', event); break; case 'turn_recovery': this.sendUpdate({ @@ -1843,20 +1739,7 @@ export class AcpSession { }); break; case 'provider_stall': - this.sendUpdate({ - sessionUpdate: 'session_info_update', - updatedAt: new Date().toISOString(), - _meta: { - 'blade/providerStall': { - phase: event.phase, - stallCount: event.stallCount, - durationMs: event.durationMs, - warningAfterMs: event.warningAfterMs, - timeoutMs: event.timeoutMs, - outputStarted: event.outputStarted, - }, - }, - }); + this.sendProjectedLoopMetadata('blade/providerStall', event); break; case 'provider_recovery': // SessionRuntime publishes this event on the Session Bus. Sending it @@ -1882,20 +1765,7 @@ export class AcpSession { }); break; case 'action_stationarity': - this.sendUpdate({ - sessionUpdate: 'session_info_update', - updatedAt: new Date().toISOString(), - _meta: { - 'blade/actionStationarity': { - phase: event.phase, - toolName: event.toolName, - runLength: event.runLength, - nudgeThreshold: event.nudgeThreshold, - haltThreshold: event.haltThreshold, - progressAware: event.progressAware, - }, - }, - }); + this.sendProjectedLoopMetadata('blade/actionStationarity', event); break; // --- 业务事件 --- @@ -2656,9 +2526,7 @@ export class AcpSession { } } - /** - * 取消当前操作 - */ + /** 取消当前操作 */ cancel(): void { logger.info(`[AcpSession ${this.id}] Cancel requested`); let cancelled = this.clearPendingResumeRequest(); @@ -2683,13 +2551,11 @@ export class AcpSession { } /** - * 设置会话模式(权限模式) - * - * 可用模式: - * - default: 所有操作都需要确认 - * - auto-edit: 文件编辑自动批准,命令需要确认 - * - yolo: 所有操作自动批准 - * - plan: 只读模式,不允许写操作 + + * 设置会话模式(权限模式)

可用模式: - default: 所有操作都需要确认 - auto-edit: 文件编辑自动批准,命令需要确认 - yolo: + + * 所有操作自动批准 - plan: 只读模式,不允许写操作 + */ async setMode(mode: string): Promise { const validModes: AcpModeId[] = ['default', 'auto-edit', 'yolo', 'plan']; @@ -2708,9 +2574,7 @@ export class AcpSession { }); } - /** - * 将 ACP 模式映射到 Blade 权限模式 - */ + /** 将 ACP 模式映射到 Blade 权限模式 */ private mapModeToPermissionMode(): PermissionMode | undefined { return this.mapModeIdToPermissionMode(this.mode); } @@ -2730,12 +2594,11 @@ export class AcpSession { } /** - * 检查操作是否需要确认 - * - * ToolKind 枚举值: - * - 'readonly': 只读操作(Read, Glob, Grep 等) - * - 'write': 写操作(Edit, Write 等) - * - 'execute': 执行操作(Bash 等) + + * 检查操作是否需要确认

ToolKind 枚举值: - 'readonly': 只读操作(Read, Glob, Grep 等) - 'write': + + * 写操作(Edit, Write 等) - 'execute': 执行操作(Bash 等) + */ private shouldAutoApprove(toolKind: string): boolean { switch (this.mode) { @@ -2755,9 +2618,7 @@ export class AcpSession { } } - /** - * 设置会话模型 - */ + /** 设置会话模型 */ async setModel(modelId: string): Promise { logger.info(`[AcpSession ${this.id}] Model set to: ${modelId}`); @@ -2973,9 +2834,7 @@ export class AcpSession { ); } - /** - * 销毁会话 - */ + /** 销毁会话 */ destroy(options: { discardPendingInput?: boolean } = {}): Promise { if (this.destroyPromise) { return this.destroyFinished ? Promise.resolve() : this.destroyPromise; @@ -3112,9 +2971,7 @@ export class AcpSession { }; } - /** - * 发送会话更新通知 - */ + /** 发送会话更新通知 */ private canSendUpdates(): boolean { return ( !this.destroyed && @@ -3168,25 +3025,13 @@ export class AcpSession { private sendFollowUpQueueSnapshot(snapshot: unknown): void { const parsed = FollowUpQueueSnapshotSchema.safeParse(snapshot); if (!parsed.success) return; - this.sendUpdate({ - sessionUpdate: 'session_info_update', - updatedAt: new Date().toISOString(), - _meta: { - 'blade/followUpQueue': followUpQueueMetadata(parsed.data), - }, - }); + this.sendSessionMetadata('blade/followUpQueue', followUpQueueMetadata(parsed.data)); } private sendProviderRecoveryProjection(snapshot: unknown): void { const parsed = ProviderRecoveryProjectionSchema.safeParse(snapshot); if (!parsed.success) return; - this.sendUpdate({ - sessionUpdate: 'session_info_update', - updatedAt: new Date().toISOString(), - _meta: { - 'blade/providerRecovery': parsed.data, - }, - }); + this.sendSessionMetadata('blade/providerRecovery', parsed.data); } private sendTurnActivityProjection(snapshot: unknown): void { @@ -3200,12 +3045,19 @@ export class AcpSession { } this.lastTurnActivityGeneration = parsed.data.generation; this.lastTurnActivityRevision = parsed.data.revision; + this.sendSessionMetadata('blade/turnActivity', parsed.data); + } + + private sendProjectedLoopMetadata(key: string, event: LoopEvent): void { + const projection = projectSessionLoopEvent(event, { omitUndefined: true }); + if (projection) this.sendSessionMetadata(key, projection.properties); + } + + private sendSessionMetadata(key: string, value: unknown): void { this.sendUpdate({ sessionUpdate: 'session_info_update', updatedAt: new Date().toISOString(), - _meta: { - 'blade/turnActivity': parsed.data, - }, + _meta: { [key]: value }, }); } diff --git a/packages/cli/src/acp/index.ts b/packages/cli/src/acp/index.ts index 034d79b9b..89f589ed3 100644 --- a/packages/cli/src/acp/index.ts +++ b/packages/cli/src/acp/index.ts @@ -1,8 +1,6 @@ /** - * ACP (Agent Client Protocol) 集成模块 - * - * 提供 Blade 作为 ACP Agent 的能力,使其可以被 Zed、JetBrains、Neovim 等编辑器调用。 - * + * ACP (Agent Client Protocol) 集成模块

提供 Blade 作为 ACP Agent 的能力,使其可以被 + * Zed、JetBrains、Neovim 等编辑器调用。 */ import { Readable, Writable } from 'node:stream'; diff --git a/packages/cli/src/agent/Agent.ts b/packages/cli/src/agent/Agent.ts index 210dc5144..78592a88c 100644 --- a/packages/cli/src/agent/Agent.ts +++ b/packages/cli/src/agent/Agent.ts @@ -1,13 +1,7 @@ /** - * Agent核心类 - 无状态设计 - * - * 设计原则: - * 1. Agent 本身不保存任何会话状态(sessionId, messages 等) - * 2. 所有状态通过 context 参数传入 - * 3. Agent 实例可以每次命令创建,用完即弃 - * 4. 历史连续性由外部 SessionContext 保证 - * - * 负责:LLM 交互、工具执行、循环检测 + * Agent核心类 - 无状态设计

设计原则: 1. Agent 本身不保存任何会话状态(sessionId, messages 等) 2. 所有状态通过 + * context 参数传入 3. Agent 实例可以每次命令创建,用完即弃 4. 历史连续性由外部 SessionContext 保证

负责:LLM + * 交互、工具执行、循环检测 */ import { randomUUID } from 'node:crypto'; @@ -105,10 +99,7 @@ function isTerminalProviderAdmissionRejection(result: LoopResult): boolean { ); } -/** - * Skill 执行上下文 - * 用于跟踪当前活动的 Skill 及其工具限制 - */ +/** Skill 执行上下文 用于跟踪当前活动的 Skill 及其工具限制 */ interface SkillExecutionContext { skillName: string; allowedTools?: string[]; @@ -121,8 +112,7 @@ export class Agent { private isInitialized = false; private activeTask?: AgentTask; private toolExecutor: ToolExecutor; - // systemPrompt 已移除 - 改为从 context 参数传入(无状态设计) - // sessionId 已移除 - 改为从 context 参数传入(无状态设计) + // systemPrompt 已移除 - 改为从 context 参数传入(无状态设计) sessionId 已移除 - 改为从 context 参数传入(无状态设计) // 核心组件 private chatService!: IChatService; @@ -203,9 +193,7 @@ export class Agent { // sessionId 不再存储在 Agent 内部,改为从 context 传入 } - /** - * 创建默认的工具执行器 - */ + /** 创建默认的工具执行器 */ private createDefaultToolExecutor(): ToolExecutor { const registry = new ToolRegistry(); // 合并基础权限配置和运行时覆盖 @@ -293,10 +281,7 @@ export class Agent { await this.switchModelIfNeeded(modelId); } - /** - * 快速创建并初始化 Agent 实例(静态工厂方法) - * 使用 Store 获取配置 - */ + /** 快速创建并初始化 Agent 实例(静态工厂方法) 使用 Store 获取配置 */ static async create(options: AgentOptions = {}): Promise { if (options.sessionId) { throw new Error( @@ -337,8 +322,7 @@ export class Agent { mergedOptions.toolBlacklist = config.disallowedTools; } - // 4. 创建并初始化 Agent - // 将 options 作为运行时参数传递 + // 4. 创建并初始化 Agent 将 options 作为运行时参数传递 const agent = new Agent(config, mergedOptions); await agent.initialize(); @@ -378,9 +362,7 @@ export class Agent { return agent; } - /** - * 初始化Agent - */ + /** 初始化Agent */ public async initialize(): Promise { if (this.isInitialized) { return; @@ -431,9 +413,7 @@ export class Agent { } } - /** - * 执行任务 - */ + /** 执行任务 */ public async executeTask(task: AgentTask): Promise { if (!this.isInitialized) { throw new Error('Agent未初始化'); @@ -1306,13 +1286,8 @@ export class Agent { return drainLoop(this.chatStream(message, context, options)); } - /** - * 运行 Plan 模式循环 - 专门处理 Plan 模式的逻辑 - * Plan 模式特点:只读调研、系统化研究方法论、最终输出实现计划 - */ - /** - * Plan 模式入口 - 准备 Plan 专用配置后调用通用循环 - */ + /** 运行 Plan 模式循环 - 专门处理 Plan 模式的逻辑 Plan 模式特点:只读调研、系统化研究方法论、最终输出实现计划 */ + /** Plan 模式入口 - 准备 Plan 专用配置后调用通用循环 */ private async *runPlanLoop( message: UserMessageContent, context: ChatContext, @@ -1342,8 +1317,7 @@ export class Agent { : { projectInstructionSourcePath: this.sessionRuntime?.projectRoot }), }); - // Plan 模式差异 2: 在用户消息中注入 system-reminder - // 处理多模态消息:提取文本部分添加 reminder + // Plan 模式差异 2: 在用户消息中注入 system-reminder 处理多模态消息:提取文本部分添加 reminder let messageWithReminder: UserMessageContent; if (typeof message === 'string') { messageWithReminder = createPlanModeReminder(message); @@ -1374,9 +1348,7 @@ export class Agent { return yield* this.executeLoop(messageWithReminder, context, options, systemPrompt); } - /** - * 普通模式入口 - 准备普通模式配置后调用通用循环 - */ + /** 普通模式入口 - 准备普通模式配置后调用通用循环 */ private async *runLoop( message: UserMessageContent, context: ChatContext, @@ -1391,9 +1363,7 @@ export class Agent { return yield* this.executeLoop(message, context, options, systemPrompt); } - /** - * 按需构建系统提示词(用于未传入 context.systemPrompt 的场景) - */ + /** 按需构建系统提示词(用于未传入 context.systemPrompt 的场景) */ private async buildSystemPromptOnDemand(context: ChatContext): Promise { const replacePrompt = this.runtimeOptions.systemPrompt; const appendPrompt = this.runtimeOptions.appendSystemPrompt; @@ -1423,9 +1393,7 @@ export class Agent { return result.prompt; } - /** - * 核心执行循环 — 返回 AsyncGenerator 事件流 - */ + /** 核心执行循环 — 返回 AsyncGenerator 事件流 */ private executeLoop( message: UserMessageContent, context: ChatContext, @@ -1505,9 +1473,7 @@ export class Agent { return events; } - /** - * 构建 LoopDependencies(从 Agent 实例注入到 generator) - */ + /** 构建 LoopDependencies(从 Agent 实例注入到 generator) */ private buildLoopDependencies(): import('./loop/types.js').LoopDependencies { return { chatService: this.chatService, @@ -1544,9 +1510,7 @@ export class Agent { }; } - /** - * 带系统提示的聊天接口 - */ + /** 带系统提示的聊天接口 */ public async chatWithSystem(systemPrompt: string, message: string): Promise { if (!this.isInitialized) { throw new Error('Agent未初始化'); @@ -1561,30 +1525,22 @@ export class Agent { return response.content; } - /** - * 获取当前活动任务 - */ + /** 获取当前活动任务 */ public getActiveTask(): AgentTask | undefined { return this.activeTask; } - /** - * 获取Chat服务 - */ + /** 获取Chat服务 */ public getChatService(): IChatService { return this.chatService; } - /** - * 获取上下文管理器 - 返回执行引擎的上下文管理功能 - */ + /** 获取上下文管理器 - 返回执行引擎的上下文管理功能 */ public getContextManager(): ContextManager | undefined { return this.executionEngine?.getContextManager(); } - /** - * 获取Agent状态统计 - */ + /** 获取Agent状态统计 */ public getStats(): Record { return { initialized: this.isInitialized, @@ -1596,23 +1552,17 @@ export class Agent { }; } - /** - * 获取可用工具列表 - */ + /** 获取可用工具列表 */ public getAvailableTools(): Tool[] { return this.toolExecutor ? this.toolExecutor.getRegistry().getAll() : []; } - /** - * 获取工具注册表(用于子 Agent 工具隔离) - */ + /** 获取工具注册表(用于子 Agent 工具隔离) */ public getToolRegistry(): ToolRegistry { return this.toolExecutor.getRegistry(); } - /** - * 应用工具白名单(仅保留指定工具) - */ + /** 应用工具白名单(仅保留指定工具) */ public applyToolWhitelist(whitelist: string[]): void { const registry = this.toolExecutor.getRegistry(); const allTools = registry.getAll(); @@ -1641,9 +1591,7 @@ export class Agent { logger.debug(`Applied tool blacklist: ${blacklist.join(', ')}`); } - /** - * 获取工具统计信息 - */ + /** 获取工具统计信息 */ public getToolStats() { const tools = this.getAvailableTools(); const toolsByKind = new Map(); @@ -1660,9 +1608,7 @@ export class Agent { }; } - /** - * 销毁Agent - */ + /** 销毁Agent */ public async destroy(): Promise { if (this.destroyPromise) return this.destroyPromise; @@ -1698,23 +1644,17 @@ export class Agent { this.sessionRuntime.getCurrentModelMaxContextTokens(); } - /** - * 生成任务ID - */ + /** 生成任务ID */ private generateTaskId(): string { return `task_${Date.now()}_${Math.random().toString(36).substring(2, 9)}`; } - /** - * 日志记录 - */ + /** 日志记录 */ private log(message: string, data?: unknown): void { logger.debug(`[MainAgent] ${message}`, data || ''); } - /** - * 错误记录 - */ + /** 错误记录 */ private error(message: string, error?: unknown): void { logger.error(`[MainAgent] ${message}`, error || ''); } @@ -1770,9 +1710,7 @@ export class Agent { } } - /** - * 注册内置工具 - */ + /** 注册内置工具 */ private async registerBuiltinTools(): Promise { try { // 使用默认 sessionId(因为注册时还没有会话上下文) @@ -1805,9 +1743,7 @@ export class Agent { } } - /** - * 注册 MCP 工具 - */ + /** 注册 MCP 工具 */ private async registerMcpTools(): Promise { try { const mcpServers = await resolveWorkspaceMcpConfig({ @@ -1895,9 +1831,7 @@ export class Agent { } } - /** - * 加载 subagent 配置 - */ + /** 加载 subagent 配置 */ private async loadSubagents(): Promise { const resources = await resolveWorkspaceAgentResources(getCwd()); this.agentResources = snapshotWorkspaceAgentResources(resources); @@ -1938,8 +1872,7 @@ export class Agent { if (tool.name === 'ReadPromptArtifact') { return true; } - // 检查工具名称是否在 allowed-tools 列表中 - // 支持精确匹配和通配符模式(如 Bash(git:*)) + // 检查工具名称是否在 allowed-tools 列表中 支持精确匹配和通配符模式(如 Bash(git:*)) return allowedTools.some((allowed) => { // 精确匹配 if (allowed === tool.name) { @@ -1963,10 +1896,7 @@ export class Agent { return filteredTools; } - /** - * 清除 Skill 执行上下文 - * 当 Skill 执行完成或需要重置时调用 - */ + /** 清除 Skill 执行上下文 当 Skill 执行完成或需要重置时调用 */ public clearSkillContext(): void { if (this.activeSkillContext) { logger.debug(`Skill "${this.activeSkillContext.skillName}" deactivated`); @@ -2041,9 +1971,7 @@ export class Agent { } } - /** - * 构建附件文本块(供 processAtMentionsForContent 使用) - */ + /** 构建附件文本块(供 processAtMentionsForContent 使用) */ private buildAttachmentText(attachments: Attachment[]): string { const contextBlocks: string[] = []; const errors: string[] = []; diff --git a/packages/cli/src/agent/ExecutionEngine.ts b/packages/cli/src/agent/ExecutionEngine.ts index 4974fff7b..eb2e63b34 100644 --- a/packages/cli/src/agent/ExecutionEngine.ts +++ b/packages/cli/src/agent/ExecutionEngine.ts @@ -1,12 +1,6 @@ /** - * ExecutionEngine - 执行引擎 - * - * 职责: - * - 管理上下文(ContextManager) - * - 提供临时消息管理(MemoryAdapter) - * - 执行简单任务 - * - * 注:并行执行由 LLM 自主决定(在一个回复中发起多个 Task 工具调用) + * ExecutionEngine - 执行引擎

职责: - 管理上下文(ContextManager) - 执行简单任务

注:并行执行由 LLM + * 自主决定(在一个回复中发起多个 Task 工具调用) */ import { ContextManager } from '../context/ContextManager.js'; @@ -15,22 +9,9 @@ import type { IChatService, Message } from '../services/ChatServiceInterface.js' import { getCwd } from '../utils/cwd.js'; import type { AgentResponse, AgentTask } from './types.js'; -/** - * 内存消息适配器 - 为向后兼容提供简单的消息管理 - * 注意:这只用于 ExecutionEngine 内部的临时消息管理 - * 实际的持久化由真实的 ContextManager 处理 - */ -export interface MemoryMessageAdapter { - getMessages(): Message[]; - addMessage(message: Message): void; - clearContext(): void; - getContextSize(): number; -} - export class ExecutionEngine { private chatService: IChatService; private contextManager: ContextManager; - private memoryAdapter: MemoryMessageAdapter; constructor( chatService: IChatService, @@ -45,44 +26,14 @@ export class ExecutionEngine { projectPath: projectPath || getCwd(), ...(stateStorage ? { stateStorage } : {}), }); - this.memoryAdapter = this.createMemoryAdapter(); } - /** - * 创建内存消息适配器(临时消息管理) - */ - private createMemoryAdapter(): MemoryMessageAdapter { - const messages: Message[] = []; - - return { - getMessages: () => [...messages], - addMessage: (message: Message) => { - messages.push(message); - }, - clearContext: () => { - messages.length = 0; - }, - getContextSize: () => messages.length, - }; - } - - /** - * 获取上下文管理器(返回真实的 ContextManager) - */ + /** 获取上下文管理器(返回真实的 ContextManager) */ public getContextManager(): ContextManager { return this.contextManager; } - /** - * 获取内存适配器(用于临时消息) - */ - public getMemoryAdapter(): MemoryMessageAdapter { - return this.memoryAdapter; - } - - /** - * 执行任务 - */ + /** 执行任务 */ async executeTask(task: AgentTask): Promise { const messages: Message[] = [{ role: 'user', content: task.prompt }]; const response = await this.chatService.chat(messages); diff --git a/packages/cli/src/agent/ExecutionSummary.ts b/packages/cli/src/agent/ExecutionSummary.ts index 658928fb6..7f2f17db5 100644 --- a/packages/cli/src/agent/ExecutionSummary.ts +++ b/packages/cli/src/agent/ExecutionSummary.ts @@ -1,8 +1,4 @@ -/** - * ExecutionSummary — 生成 Agent 执行摘要 - * - * 在 agent loop 完成后,汇总执行统计信息用于 CLI 输出、日志和可观测性。 - */ +/** ExecutionSummary — 生成 Agent 执行摘要 在 agent loop 完成后,汇总执行统计信息用于 CLI 输出、日志和可观测性。 */ import { estimateCostUsd } from '../services/pricing.js'; diff --git a/packages/cli/src/agent/loop/ConversationState.ts b/packages/cli/src/agent/loop/ConversationState.ts index 4a1e12c29..420232b8a 100644 --- a/packages/cli/src/agent/loop/ConversationState.ts +++ b/packages/cli/src/agent/loop/ConversationState.ts @@ -111,11 +111,6 @@ export class ConversationState { return this._history.length; } - /** 获取 history 引用(用于压缩服务读取) */ - get history(): Message[] { - return this._history; - } - /** 获取 pending 引用(用于调试) */ get pending(): ReadonlyArray { return this._pending; @@ -125,32 +120,16 @@ export class ConversationState { return this._contextRevision; } - /** - * 组装完整的 LLM 消息数组 - * - * = systemMessages + history + pending - */ + /** 组装完整的 LLM 消息数组 = systemMessages + history + pending */ toLLMMessages(): Message[] { return [...this.systemMessages, ...this._history, ...this._pending]; } - /** toLLMMessages 的别名,与计划 API 保持一致 */ - getMessagesForLLM(): Message[] { - return this.toLLMMessages(); - } - /** 返回 history 的副本 */ getHistory(): Message[] { return this._history; } - /** - * 追加消息到 pending(当前轮次的 assistant/tool/user 消息) - */ - appendPending(msg: Message): void { - this._pending.push(msg); - } - /** 追加用户消息到 pending */ appendUser(msg: Message): void { this._pending.push(msg); @@ -184,10 +163,7 @@ export class ConversationState { ); } - /** - * 追加控制消息到 pending。 - * role === 'system' 时抛出异常(根系统提示只能通过构造函数设置)。 - */ + /** 追加控制消息到 pending。 role === 'system' 时抛出异常(根系统提示只能通过构造函数设置)。 */ appendControl(role: string, msg: Message): void { if (role === 'system') { throw new Error( @@ -228,10 +204,7 @@ export class ConversationState { this._history.push(msg); } - /** - * 将 pending 提交到 history 并清空 pending。 - * 在每轮工具执行结束后、进入下一轮 LLM 调用前调用。 - */ + /** 将 pending 提交到 history 并清空 pending。 在每轮工具执行结束后、进入下一轮 LLM 调用前调用。 */ commitPending(): void { if (this._pending.length > 0) { this._history.push(...this._pending); @@ -300,11 +273,4 @@ export class ConversationState { this.commitPending(); this.context.messages = [...this._contextualSystemMessages, ...this._history]; } - - /** - * 是否有根系统提示(用于压缩重建逻辑) - */ - get hasSystemPrompt(): boolean { - return this.systemMessages.length > 0; - } } diff --git a/packages/cli/src/agent/loop/StreamingToolExecutor.ts b/packages/cli/src/agent/loop/StreamingToolExecutor.ts index 0b1709df0..14c34baa3 100644 --- a/packages/cli/src/agent/loop/StreamingToolExecutor.ts +++ b/packages/cli/src/agent/loop/StreamingToolExecutor.ts @@ -1,16 +1,9 @@ /** - * StreamingToolExecutor — 流式工具执行器 - * - * 在 LLM 流式输出过程中即开始执行工具,节省 RTT。 - * - * 设计: - * - STREAMING_PRELAUNCH_ALLOWLIST 中的工具 -> 立即启动(流式预启动) - * - 不在 allowlist 中的工具 -> 排队到流提交后交给 ToolExecutor 公平调度 - * - parallelism=shared 的工具共享执行,exclusive 工具形成 FIFO 屏障 - * - discard() 用于流式降级到非流式时清理,递增 epoch 阻止旧世代结果 - * - * 流式预启动要求 allowlist 和 isConcurrencySafe 同时成立。allowlist 防止在 - * provider 流提交前启动不可回放的副作用;批内语义由 parallelism 负责。 + * StreamingToolExecutor — 流式工具执行器

在 LLM 流式输出过程中即开始执行工具,节省 RTT。

设计: - + * STREAMING_PRELAUNCH_ALLOWLIST 中的工具 -> 立即启动(流式预启动) - 不在 allowlist 中的工具 -> 排队到流提交后交给 + * ToolExecutor 公平调度 - parallelism=shared 的工具共享执行,exclusive 工具形成 FIFO 屏障 - discard() + * 用于流式降级到非流式时清理,递增 epoch 阻止旧世代结果

流式预启动要求 allowlist 和 isConcurrencySafe + * 同时成立。allowlist 防止在 provider 流提交前启动不可回放的副作用;批内语义由 parallelism 负责。 */ import type { ContextManager } from '../../context/ContextManager.js'; @@ -50,11 +43,7 @@ export type ToolDispatchStatus = 'prelaunched' | 'queued' | 'rejected'; const logger = createLogger(LogCategory.AGENT); -/** - * 允许在流式阶段提前执行的工具白名单。 - * 仅纯读、无副作用的工具才应出现在此列表中。 - * 此列表与 isConcurrencySafe(文件锁语义)完全独立。 - */ +/** 允许在流式阶段提前执行的工具白名单。 仅纯读、无副作用的工具才应出现在此列表中。 此列表与 isConcurrencySafe(文件锁语义)完全独立。 */ export const STREAMING_PRELAUNCH_ALLOWLIST: ReadonlySet = new Set([ 'Read', 'Glob', @@ -127,9 +116,7 @@ export class StreamingToolExecutor { this.rollbackAdmission = rollbackAdmission; } - /** - * 流式中调用:在 allowlist 中的工具立即执行,否则排队 - */ + /** 流式中调用:在 allowlist 中的工具立即执行,否则排队 */ addTool( toolCall: FunctionToolCall, params: Record @@ -196,9 +183,7 @@ export class StreamingToolExecutor { return this.queued.map((queued) => queued.toolCall); } - /** - * 流结束后调用:按添加顺序 yield 所有结果 - */ + /** 流结束后调用:按添加顺序 yield 所有结果 */ async *getRemainingResults(): AsyncGenerator { this.dispatchQueuedTools(); @@ -223,9 +208,7 @@ export class StreamingToolExecutor { } } - /** - * 非阻塞获取已完成的结果 - */ + /** 非阻塞获取已完成的结果 */ getCompletedResults(): ToolExecResult[] { const results = Array.from(this.completed.values()); this.completed.clear(); @@ -267,16 +250,12 @@ export class StreamingToolExecutor { ); } - /** - * 是否有工具被添加 - */ + /** 是否有工具被添加 */ hasTools(): boolean { return this.order.length > 0; } - /** - * 获取当前 epoch(仅供测试使用) - */ + /** 获取当前 epoch(仅供测试使用) */ getEpoch(): number { return this.epoch; } diff --git a/packages/cli/src/agent/loop/completionPolicy.ts b/packages/cli/src/agent/loop/completionPolicy.ts index 3b0f53bed..5e2bdba9f 100644 --- a/packages/cli/src/agent/loop/completionPolicy.ts +++ b/packages/cli/src/agent/loop/completionPolicy.ts @@ -1,12 +1,8 @@ /** - * completionPolicy — 完成策略检查 - * - * 从 executeLoopGenerator 中提取的三个完成检查逻辑: - * 1. checkOutputRecovery — finishReason === 'length' 时的恢复/截断判断 - * 2. checkIncompleteIntent — 检测 LLM "说了要做但没做"的模式 - * 3. checkStopHook — 执行 stop hook 并加超时保护 - * - * 所有函数返回 action descriptors,不执行副作用。 + * completionPolicy — 完成策略检查

从 executeLoopGenerator 中提取的三个完成检查逻辑: 1. + * checkOutputRecovery — finishReason === 'length' 时的恢复/截断判断 2. checkIncompleteIntent — + * 检测 LLM "说了要做但没做"的模式 3. checkStopHook — 执行 stop hook 并加超时保护

所有函数返回 action + * descriptors,不执行副作用。 */ import type { PermissionMode } from '../../config/index.js'; @@ -85,12 +81,11 @@ export type IncompleteIntentAction = | { action: 'none' }; /** - * 检测 LLM 是否表达了意图但未执行工具。 - * - * Bug fixes: - * 1. 用显式 retryCount 替代滑动窗口,避免远距离重触发 - * 2. 只检测尾部 200 字符,避免全文误匹配 - * 3. 排除 markdown code block 内的匹配 + + * 检测 LLM 是否表达了意图但未执行工具。

Bug fixes: 1. 用显式 retryCount 替代滑动窗口,避免远距离重触发 2. 只检测尾部 200 + + * 字符,避免全文误匹配 3. 排除 markdown code block 内的匹配 + */ export function checkIncompleteIntent( content: string | undefined, diff --git a/packages/cli/src/agent/loop/consumeLoop.ts b/packages/cli/src/agent/loop/consumeLoop.ts index 82ee4c0e1..ffd66889d 100644 --- a/packages/cli/src/agent/loop/consumeLoop.ts +++ b/packages/cli/src/agent/loop/consumeLoop.ts @@ -1,8 +1,6 @@ /** - * drainLoop — 消费 generator 事件流,返回最终 LoopResult - * - * 用于不需要逐事件处理的场景(如 subagent、slash commands)。 - * 可选传入 onEvent 回调来处理特定事件。 + * drainLoop — 消费 generator 事件流,返回最终 LoopResult

用于不需要逐事件处理的场景(如 subagent、slash + * commands)。 可选传入 onEvent 回调来处理特定事件。 */ import type { LoopResult } from '../types.js'; diff --git a/packages/cli/src/agent/loop/conversationPersistence.ts b/packages/cli/src/agent/loop/conversationPersistence.ts index 2c2d29753..56ee15eed 100644 --- a/packages/cli/src/agent/loop/conversationPersistence.ts +++ b/packages/cli/src/agent/loop/conversationPersistence.ts @@ -1,8 +1,6 @@ /** - * conversationPersistence — 统一封装会话持久化操作 - * - * 从 executeLoopGenerator 中提取的 JSONL 持久化逻辑, - * 统一 contextMgr 获取与错误日志处理。 + * conversationPersistence — 统一封装会话持久化操作

从 executeLoopGenerator 中提取的 JSONL 持久化逻辑, 统一 + * contextMgr 获取与错误日志处理。 */ import { deriveSessionTitleFromContent } from '../../api/sessionTitle.js'; @@ -109,10 +107,7 @@ function getContextMgr(deps: LoopDependencies) { return deps.executionEngine?.getContextManager(); } -/** - * 保存用户消息到 JSONL。 - * 空白纯文本消息会被跳过。 - */ +/** 保存用户消息到 JSONL。 空白纯文本消息会被跳过。 */ export async function saveUserMessage( deps: LoopDependencies, context: ChatContext, @@ -162,10 +157,7 @@ export async function saveUserMessage( return null; } -/** - * 保存助手消息到 JSONL。 - * 空白内容会被跳过。 - */ +/** 保存助手消息到 JSONL。 空白内容会被跳过。 */ export async function saveAssistantMessage( deps: LoopDependencies, context: ChatContext, @@ -264,9 +256,7 @@ export async function saveContextualProjectRulesMarker( return null; } -/** - * 保存工具调用到 JSONL。 - */ +/** 保存工具调用到 JSONL。 */ export async function saveToolUse( deps: LoopDependencies, context: ChatContext, @@ -301,9 +291,7 @@ export async function saveToolUse( return null; } -/** - * 保存工具结果到 JSONL。 - */ +/** 保存工具结果到 JSONL。 */ export async function saveToolResult( deps: LoopDependencies, context: ChatContext, @@ -345,9 +333,7 @@ export async function saveToolResult( return null; } -/** - * 保存压缩数据到 JSONL。 - */ +/** 保存压缩数据到 JSONL。 */ export async function saveCompaction( deps: LoopDependencies, context: ChatContext, diff --git a/packages/cli/src/agent/loop/executeLoopGenerator.ts b/packages/cli/src/agent/loop/executeLoopGenerator.ts index 8ef5572ea..bbc7bb6bd 100644 --- a/packages/cli/src/agent/loop/executeLoopGenerator.ts +++ b/packages/cli/src/agent/loop/executeLoopGenerator.ts @@ -1,8 +1,6 @@ /** - * AsyncGenerator 驱动的 Agent 循环 - * - * 从 Agent.executeLoop() 提取的核心循环逻辑, - * 转换为 AsyncGenerator 模式,yield LoopEvent 事件流。 + * AsyncGenerator 驱动的 Agent 循环

从 Agent.executeLoop() 提取的核心循环逻辑, 转换为 AsyncGenerator + * 模式,yield LoopEvent 事件流。 */ import { createHash } from 'node:crypto'; @@ -12,7 +10,6 @@ import { CompactionService, } from '../../context/CompactionService.js'; import { - type ContextTokenSource, ContextTokenTracker, createContextTokenRequestProfile, resolveProviderContextTokens, @@ -159,6 +156,7 @@ import { applyToolDomainEffects } from './toolDomainPolicy.js'; import type { LoopDependencies, LoopEvent, + SystemEvent, TokenUsageInfo, ToolCallRef, } from './types.js'; @@ -516,8 +514,7 @@ async function* processStreamResponse( for (const tc of chunk.toolCalls) { accumulateToolCall(toolCallAccumulator, tc); - // pi-ai 的 toolcall_end 事件包含完整参数 - // 立即通过 StreamingToolExecutor 启动执行 + // pi-ai 的 toolcall_end 事件包含完整参数 立即通过 StreamingToolExecutor 启动执行 if (executor) { const castTc = tc as { index?: number; @@ -619,6 +616,55 @@ export interface LoopCompactionState { lastCompactionTurn?: number; } +type CompactionEventState = Omit< + Extract, + 'kind' | 'phase' | 'reason' +>; + +interface CompactionMetricsSource { + preTokens?: number; + postTokens?: number; + sampleAttempts?: number; + inputReductions?: number; + messagesOmitted?: number; + filesOmitted?: number; + imagesOmitted?: number; + fallbackTargetTokens?: number; + fallbackMessagesOmitted?: number; + fallbackMessagesTruncated?: number; + failureReason?: CompactionFailureReason; +} + +function compactionMetrics(result: CompactionMetricsSource) { + return { + postTokens: result.postTokens, + sampleAttempts: result.sampleAttempts, + inputReductions: result.inputReductions, + messagesOmitted: result.messagesOmitted, + filesOmitted: result.filesOmitted, + imagesOmitted: result.imagesOmitted, + fallbackTargetTokens: result.fallbackTargetTokens, + fallbackMessagesOmitted: result.fallbackMessagesOmitted, + fallbackMessagesTruncated: result.fallbackMessagesTruncated, + failureReason: result.failureReason, + }; +} + +function completedCompactionState( + result: CompactionMetricsSource, + strategy: NonNullable, + outcome: NonNullable, + context: Pick = {} +): CompactionEventState { + return { + outcome, + strategy, + preTokens: result.preTokens, + ...compactionMetrics(result), + ...context, + }; +} + async function commitCompactionMemory( memoryPlan: MemoryConsolidationPlan | undefined, context: ChatContext, @@ -811,20 +857,7 @@ export async function* checkAndCompactInLoop( : `[Loop] [轮次 ${currentTurn}] 触发循环内自动压缩` ); - let outcome: 'completed' | 'fallback' | 'failed' = 'failed'; - let strategy: 'llm' | 'fallback' | undefined; - let preTokens: number | undefined; - let postTokens: number | undefined; - let sampleAttempts: number | undefined; - let inputReductions: number | undefined; - let messagesOmitted: number | undefined; - let filesOmitted: number | undefined; - let imagesOmitted: number | undefined; - let fallbackTargetTokens: number | undefined; - let fallbackMessagesOmitted: number | undefined; - let fallbackMessagesTruncated: number | undefined; - let failureReason: CompactionFailureReason | undefined; - let memory: MemoryConsolidationProjection | undefined; + let compaction: CompactionEventState = { outcome: 'failed' }; let failurePhase: 'compaction' | 'checkpoint' = 'compaction'; yield { kind: 'compaction', phase: 'start', reason: 'threshold' }; try { @@ -853,18 +886,11 @@ export async function* checkAndCompactInLoop( }; } - strategy = result.success ? 'llm' : 'fallback'; - preTokens = result.preTokens; - postTokens = result.postTokens; - sampleAttempts = result.sampleAttempts; - inputReductions = result.inputReductions; - messagesOmitted = result.messagesOmitted; - filesOmitted = result.filesOmitted; - imagesOmitted = result.imagesOmitted; - fallbackTargetTokens = result.fallbackTargetTokens; - fallbackMessagesOmitted = result.fallbackMessagesOmitted; - fallbackMessagesTruncated = result.fallbackMessagesTruncated; - failureReason = result.failureReason; + const strategy = result.success ? 'llm' : 'fallback'; + compaction = completedCompactionState(result, strategy, 'failed', { + preTokenSource: snapshot.tokenSource, + estimatedPendingTokens: snapshot.estimatedPendingTokens, + }); if (result.success) { logger.debug( `[Loop] [轮次 ${currentTurn}] 压缩完成: ${result.preTokens} -> ${result.postTokens} tokens` @@ -890,16 +916,7 @@ export async function* checkAndCompactInLoop( ...(snapshot.estimatedPendingTokens !== undefined ? { estimatedPendingTokens: snapshot.estimatedPendingTokens } : {}), - postTokens: result.postTokens, - sampleAttempts: result.sampleAttempts, - inputReductions: result.inputReductions, - messagesOmitted: result.messagesOmitted, - filesOmitted: result.filesOmitted, - imagesOmitted: result.imagesOmitted, - fallbackTargetTokens: result.fallbackTargetTokens, - fallbackMessagesOmitted: result.fallbackMessagesOmitted, - fallbackMessagesTruncated: result.fallbackMessagesTruncated, - failureReason: result.failureReason, + ...compactionMetrics(result), filesIncluded: result.filesIncluded, replacementMessages: result.compactedMessages, }, @@ -908,12 +925,16 @@ export async function* checkAndCompactInLoop( } ); - memory = await commitCompactionMemory(result.memoryPlan, context, checkpointId); + compaction.memory = await commitCompactionMemory( + result.memoryPlan, + context, + checkpointId + ); + compaction.outcome = result.success ? 'completed' : 'fallback'; context.messages = result.compactedMessages; if (compactionState) { compactionState.lastCompactionTurn = currentTurn; } - outcome = result.success ? 'completed' : 'fallback'; return { kind: 'compacted', postTokens: result.postTokens }; } catch (error) { // AbortError(宽口径): 返回 'none' 让控制流回到主循环的下一个 signal 检查点 @@ -936,24 +957,7 @@ export async function* checkAndCompactInLoop( kind: 'compaction', phase: 'end', reason: 'threshold', - outcome, - strategy, - preTokens, - ...(snapshot.tokenSource ? { preTokenSource: snapshot.tokenSource } : {}), - ...(snapshot.estimatedPendingTokens !== undefined - ? { estimatedPendingTokens: snapshot.estimatedPendingTokens } - : {}), - postTokens, - sampleAttempts, - inputReductions, - messagesOmitted, - filesOmitted, - imagesOmitted, - fallbackTargetTokens, - fallbackMessagesOmitted, - fallbackMessagesTruncated, - failureReason, - memory, + ...compaction, }; } } @@ -1874,22 +1878,7 @@ validates the object and may return a bounded corrective error.`; if (response?.continue) { state.writeback(); - let compactionOutcome: 'completed' | 'fallback' | 'failed' = 'failed'; - let compactionStrategy: 'llm' | 'fallback' | undefined; - let compactionPreTokens: number | undefined; - let compactionPreTokenSource: ContextTokenSource | undefined; - let compactionEstimatedPendingTokens: number | undefined; - let compactionPostTokens: number | undefined; - let compactionSampleAttempts: number | undefined; - let compactionInputReductions: number | undefined; - let compactionMessagesOmitted: number | undefined; - let compactionFilesOmitted: number | undefined; - let compactionImagesOmitted: number | undefined; - let compactionFallbackTargetTokens: number | undefined; - let compactionFallbackMessagesOmitted: number | undefined; - let compactionFallbackMessagesTruncated: number | undefined; - let compactionFailureReason: CompactionFailureReason | undefined; - let compactionMemory: MemoryConsolidationProjection | undefined; + let compaction: CompactionEventState = { outcome: 'failed' }; yield { kind: 'compaction', phase: 'start', @@ -1919,9 +1908,6 @@ validates the object and may return a bounded corrective error.`; modelName: chatConfig.model, requestProfile: turnLimitRequestProfile, }); - compactionPreTokenSource = turnLimitProjection.source; - compactionEstimatedPendingTokens = - turnLimitProjection.estimatedPendingTokens; const compactResult = await CompactionService.compact( context.messages, { @@ -1966,20 +1952,16 @@ validates the object and may return a bounded corrective error.`; ...compactResult.compactedMessages, continueMessage, ]; - compactionStrategy = compactResult.success ? 'llm' : 'fallback'; - compactionPreTokens = compactResult.preTokens; - compactionPostTokens = compactResult.postTokens; - compactionSampleAttempts = compactResult.sampleAttempts; - compactionInputReductions = compactResult.inputReductions; - compactionMessagesOmitted = compactResult.messagesOmitted; - compactionFilesOmitted = compactResult.filesOmitted; - compactionImagesOmitted = compactResult.imagesOmitted; - compactionFallbackTargetTokens = compactResult.fallbackTargetTokens; - compactionFallbackMessagesOmitted = - compactResult.fallbackMessagesOmitted; - compactionFallbackMessagesTruncated = - compactResult.fallbackMessagesTruncated; - compactionFailureReason = compactResult.failureReason; + const strategy = compactResult.success ? 'llm' : 'fallback'; + compaction = completedCompactionState( + compactResult, + strategy, + 'failed', + { + preTokenSource: turnLimitProjection.source, + estimatedPendingTokens: turnLimitProjection.estimatedPendingTokens, + } + ); const checkpointId = await persistCompaction( deps, @@ -1988,7 +1970,7 @@ validates the object and may return a bounded corrective error.`; { trigger: 'auto', reason: 'turn_limit', - strategy: compactionStrategy, + strategy, preTokens: compactResult.preTokens, preTokenSource: turnLimitProjection.source, ...(turnLimitProjection.estimatedPendingTokens !== undefined @@ -1997,22 +1979,13 @@ validates the object and may return a bounded corrective error.`; turnLimitProjection.estimatedPendingTokens, } : {}), - postTokens: compactResult.postTokens, - sampleAttempts: compactResult.sampleAttempts, - inputReductions: compactResult.inputReductions, - messagesOmitted: compactResult.messagesOmitted, - filesOmitted: compactResult.filesOmitted, - imagesOmitted: compactResult.imagesOmitted, - fallbackTargetTokens: compactResult.fallbackTargetTokens, - fallbackMessagesOmitted: compactResult.fallbackMessagesOmitted, - fallbackMessagesTruncated: compactResult.fallbackMessagesTruncated, - failureReason: compactResult.failureReason, + ...compactionMetrics(compactResult), filesIncluded: compactResult.filesIncluded, replacementMessages, }, { required: deps.executionEngine !== undefined } ); - compactionMemory = await commitCompactionMemory( + compaction.memory = await commitCompactionMemory( compactResult.memoryPlan, context, checkpointId @@ -2020,7 +1993,7 @@ validates the object and may return a bounded corrective error.`; context.messages = replacementMessages; state.replaceHistory(context.messages); contextTokenTracker.reset(); - compactionOutcome = compactResult.success ? 'completed' : 'fallback'; + compaction.outcome = compactResult.success ? 'completed' : 'fallback'; } catch (compactError) { if (compactError instanceof CompactionAbortedError) { recordUsage(compactError.usage); @@ -2040,22 +2013,7 @@ validates the object and may return a bounded corrective error.`; kind: 'compaction', phase: 'end', reason: 'turn_limit', - outcome: compactionOutcome, - strategy: compactionStrategy, - preTokens: compactionPreTokens, - preTokenSource: compactionPreTokenSource, - estimatedPendingTokens: compactionEstimatedPendingTokens, - postTokens: compactionPostTokens, - sampleAttempts: compactionSampleAttempts, - inputReductions: compactionInputReductions, - messagesOmitted: compactionMessagesOmitted, - filesOmitted: compactionFilesOmitted, - imagesOmitted: compactionImagesOmitted, - fallbackTargetTokens: compactionFallbackTargetTokens, - fallbackMessagesOmitted: compactionFallbackMessagesOmitted, - fallbackMessagesTruncated: compactionFallbackMessagesTruncated, - failureReason: compactionFailureReason, - memory: compactionMemory, + ...compaction, }; } @@ -2621,20 +2579,7 @@ validates the object and may return a bounded corrective error.`; logger.warn('[Loop] 检测到 prompt_too_long 错误,尝试反应式压缩'); const chatConfig = deps.chatService.getConfig(); let recovered = false; - let outcome: 'completed' | 'fallback' | 'failed' = 'failed'; - let strategy: 'llm' | 'fallback' | 'snip' | undefined; - let preTokens: number | undefined; - let postTokens: number | undefined; - let sampleAttempts: number | undefined; - let inputReductions: number | undefined; - let messagesOmitted: number | undefined; - let filesOmitted: number | undefined; - let imagesOmitted: number | undefined; - let fallbackTargetTokens: number | undefined; - let fallbackMessagesOmitted: number | undefined; - let fallbackMessagesTruncated: number | undefined; - let failureReason: CompactionFailureReason | undefined; - let memory: MemoryConsolidationProjection | undefined; + let compaction: CompactionEventState = { outcome: 'failed' }; yield { kind: 'compaction', phase: 'start', @@ -2659,18 +2604,13 @@ validates the object and may return a bounded corrective error.`; sessionId: context.sessionId, } ); - strategy = result.strategy; - preTokens = result.preTokens; - postTokens = result.postTokens; - sampleAttempts = result.sampleAttempts; - inputReductions = result.inputReductions; - messagesOmitted = result.messagesOmitted; - filesOmitted = result.filesOmitted; - imagesOmitted = result.imagesOmitted; - fallbackTargetTokens = result.fallbackTargetTokens; - fallbackMessagesOmitted = result.fallbackMessagesOmitted; - fallbackMessagesTruncated = result.fallbackMessagesTruncated; - failureReason = result.failureReason; + if (result.strategy) { + compaction = completedCompactionState( + result, + result.strategy, + 'failed' + ); + } if (result.usage) { recordUsage(result.usage); yield { @@ -2697,22 +2637,13 @@ validates the object and may return a bounded corrective error.`; reason: 'context_limit', strategy: result.strategy, preTokens: result.preTokens, - postTokens: result.postTokens, - sampleAttempts: result.sampleAttempts, - inputReductions: result.inputReductions, - messagesOmitted: result.messagesOmitted, - filesOmitted: result.filesOmitted, - imagesOmitted: result.imagesOmitted, - fallbackTargetTokens: result.fallbackTargetTokens, - fallbackMessagesOmitted: result.fallbackMessagesOmitted, - fallbackMessagesTruncated: result.fallbackMessagesTruncated, - failureReason: result.failureReason, + ...compactionMetrics(result), filesIncluded: result.filesIncluded, replacementMessages: result.messages, }, { required: deps.executionEngine !== undefined } ); - memory = await commitCompactionMemory( + compaction.memory = await commitCompactionMemory( result.memoryPlan, context, checkpointId @@ -2722,7 +2653,8 @@ validates the object and may return a bounded corrective error.`; state.replaceHistory(context.messages); contextTokenTracker.reset(); requiredToolName = replayRequiredToolName; - outcome = result.strategy === 'llm' ? 'completed' : 'fallback'; + compaction.outcome = + result.strategy === 'llm' ? 'completed' : 'fallback'; recovered = true; logger.info('[Loop] 反应式压缩成功,重试 LLM 调用'); } @@ -2748,20 +2680,7 @@ validates the object and may return a bounded corrective error.`; kind: 'compaction', phase: 'end', reason: 'context_limit', - outcome, - strategy, - preTokens, - postTokens, - sampleAttempts, - inputReductions, - messagesOmitted, - filesOmitted, - imagesOmitted, - fallbackTargetTokens, - fallbackMessagesOmitted, - fallbackMessagesTruncated, - failureReason, - memory, + ...compaction, }; } if (recovered) { @@ -3775,8 +3694,7 @@ validates the object and may return a bounded corrective error.`; let executionResultsPromise: Promise; if (streamingExecutor?.hasTools()) { - // 流式模式:工具已在流式中开始执行,收集结果 - // tool_start 事件已在 processStreamResponse 中 yield + // 流式模式:工具已在流式中开始执行,收集结果 tool_start 事件已在 processStreamResponse 中 yield logger.debug( `[Loop] 使用 StreamingToolExecutor 收集 ${functionCalls.length} 个工具结果` ); diff --git a/packages/cli/src/agent/loop/index.ts b/packages/cli/src/agent/loop/index.ts index aace9b2f0..f250bc910 100644 --- a/packages/cli/src/agent/loop/index.ts +++ b/packages/cli/src/agent/loop/index.ts @@ -1,8 +1,4 @@ -/** - * Agent Loop 模块 - * - * 提供 AsyncGenerator 驱动的 Agent 循环实现 - */ +/** Agent Loop 模块 提供 AsyncGenerator 驱动的 Agent 循环实现 */ export { ConversationState, isRootSystemPrompt } from './ConversationState.js'; export { diff --git a/packages/cli/src/agent/loop/toolDomainPolicy.ts b/packages/cli/src/agent/loop/toolDomainPolicy.ts index fa9328a0c..60ec19974 100644 --- a/packages/cli/src/agent/loop/toolDomainPolicy.ts +++ b/packages/cli/src/agent/loop/toolDomainPolicy.ts @@ -1,12 +1,7 @@ /** - * toolDomainPolicy — 工具结果的领域副作用处理 - * - * 从 executeLoopGenerator 中提取的 domain side effects: - * - TaskCreate/TaskUpdate/TaskList -> 更新任务列表 - * - Skill -> 激活 skill context - * - ModelSwitch -> 触发模型切换 - * - * 纯函数 / 薄封装,返回 action descriptors 或直接调用 deps 回调。 + * toolDomainPolicy — 工具结果的领域副作用处理

从 executeLoopGenerator 中提取的 domain side effects: + * - TaskCreate/TaskUpdate/TaskList -> 更新任务列表 - Skill -> 激活 skill context - ModelSwitch + * -> 触发模型切换

纯函数 / 薄封装,返回 action descriptors 或直接调用 deps 回调。 */ import { HookManager } from '../../hooks/HookManager.js'; @@ -15,10 +10,7 @@ import type { ToolResult } from '../../tools/types/index.js'; import type { ChatContext } from '../types.js'; import type { DomainEvent, LoopDependencies } from './types.js'; -/** - * 窄化的工具调用引用:只包含 function 类型的 tool call。 - * 由调用方在传入时断言(executeLoopGenerator 中已有此 cast)。 - */ +/** 窄化的工具调用引用:只包含 function 类型的 tool call。 由调用方在传入时断言(executeLoopGenerator 中已有此 cast)。 */ export interface FunctionToolCallRef { id: string; type: 'function'; @@ -32,10 +24,7 @@ export interface TaskUpdateAction { tasks: TaskListItem[]; } -/** - * 处理任务列表工具结果,提取任务列表。 - * 返回 TaskUpdateAction 或 null。 - */ +/** 处理任务列表工具结果,提取任务列表。 返回 TaskUpdateAction 或 null。 */ export function handleTaskListUpdate( toolCall: FunctionToolCallRef, result: ToolResult @@ -62,9 +51,7 @@ export function handleTaskListUpdate( // ===== Skill Activation ===== -/** - * 处理 Skill 工具结果,触发 skill 激活回调。 - */ +/** 处理 Skill 工具结果,触发 skill 激活回调。 */ export function handleSkillActivation( toolCall: FunctionToolCallRef, result: ToolResult, @@ -86,10 +73,7 @@ export function handleSkillActivation( // ===== Model Switch ===== -/** - * 处理工具结果中的模型切换请求。 - * 返回 modelId 或 undefined。 - */ +/** 处理工具结果中的模型切换请求。 返回 modelId 或 undefined。 */ export function extractModelSwitch(result: ToolResult): string | undefined { const metadata = result.metadata as Record | undefined; if (!metadata) return undefined; @@ -201,10 +185,7 @@ export function handleSubagentLifecycle( return null; } -/** - * 处理所有工具结果的领域副作用。 - * 返回 DomainEvent(如果有)并触发 skill/model 回调。 - */ +/** 处理所有工具结果的领域副作用。 返回 DomainEvent(如果有)并触发 skill/model 回调。 */ export async function applyToolDomainEffects( toolCall: FunctionToolCallRef, result: ToolResult, diff --git a/packages/cli/src/agent/loop/types.ts b/packages/cli/src/agent/loop/types.ts index 61098cad5..ed24a6dc7 100644 --- a/packages/cli/src/agent/loop/types.ts +++ b/packages/cli/src/agent/loop/types.ts @@ -1,8 +1,4 @@ -/** - * AsyncGenerator Loop 类型定义 - * - * 用于将 Agent.executeLoop() 重构为 AsyncGenerator 模式 - */ +/** AsyncGenerator Loop 类型定义 用于将 Agent.executeLoop() 重构为 AsyncGenerator 模式 */ import type { FollowUpQueueSnapshot } from '../../api/followUpQueueSchemas.js'; import type { ProviderRecoveryProjection } from '../../api/providerRecoverySchemas.js'; diff --git a/packages/cli/src/agent/resources/WorkspaceProjectRules.ts b/packages/cli/src/agent/resources/WorkspaceProjectRules.ts index 025be8f36..750d10240 100644 --- a/packages/cli/src/agent/resources/WorkspaceProjectRules.ts +++ b/packages/cli/src/agent/resources/WorkspaceProjectRules.ts @@ -317,7 +317,6 @@ function retainWithinBudget( } export class ProjectRuleCatalog { - readonly catalogSha256: string; private readonly definitions: readonly ProjectRuleDefinition[]; private readonly byId: ReadonlyMap; @@ -330,9 +329,6 @@ export class ProjectRuleCatalog { [...definitions].sort(definitionOrder).map((item) => Object.freeze({ ...item })) ); this.byId = new Map(this.definitions.map((item) => [item.id, item])); - this.catalogSha256 = sha256( - this.definitions.map((item) => `${item.id}:${item.contentSha256}`).join('\n') - ); } static empty(projectRoot: string): ProjectRuleCatalog { diff --git a/packages/cli/src/agent/runtime/SessionRuntime.ts b/packages/cli/src/agent/runtime/SessionRuntime.ts index 01c8a1560..bb70a6a9c 100644 --- a/packages/cli/src/agent/runtime/SessionRuntime.ts +++ b/packages/cli/src/agent/runtime/SessionRuntime.ts @@ -815,7 +815,6 @@ export class SessionRuntime { ): Promise { return new PersistentStore( workspaceRoot, - 100, undefined, stateStorage ).hasRecoverableTurn(sessionId); @@ -855,10 +854,6 @@ export class SessionRuntime { return this.config; } - getAvailableModels(): ModelConfig[] { - return this.config.models.map((model) => structuredClone(model)); - } - getModelById(modelId: string): ModelConfig | undefined { const model = this.config.models.find((candidate) => candidate.id === modelId); return model ? structuredClone(model) : undefined; @@ -2927,11 +2922,6 @@ export class SessionRuntime { }); } - /** @deprecated Use createToolExecutor() for new code. */ - createExecutionPipeline(options: AgentOptions = {}): ToolExecutor { - return this.createToolExecutor(options); - } - async dispose(): Promise { if (this.disposePromise) { return this.disposePromise; diff --git a/packages/cli/src/agent/subagents/AgentSessionStore.ts b/packages/cli/src/agent/subagents/AgentSessionStore.ts index 29f6b27a7..3355cedc0 100644 --- a/packages/cli/src/agent/subagents/AgentSessionStore.ts +++ b/packages/cli/src/agent/subagents/AgentSessionStore.ts @@ -1,10 +1,6 @@ /** - * Agent 会话持久化存储 - * - * 用于支持 Task 工具的 resume 功能: - * - 保存 agent 执行上下文到文件 - * - 支持跨会话恢复 agent - * - 自动清理过期会话 + * Agent 会话持久化存储

用于支持 Task 工具的 resume 功能: - 保存 agent 执行上下文到文件 - 支持跨会话恢复 agent - + * 自动清理过期会话 */ import fs from 'node:fs'; @@ -29,9 +25,7 @@ import type { SubagentConfig } from './types.js'; const logger = createLogger(LogCategory.AGENT); export const MAX_CACHED_AGENT_SESSIONS = 256; -/** - * Agent 会话状态 - */ +/** Agent 会话状态 */ export type AgentSessionStatus = 'running' | 'completed' | 'failed' | 'cancelled'; export type AgentRestartRecoveryOutcome = 'completed' | 'interrupted' | 'failed'; @@ -78,9 +72,7 @@ export function createAgentSessionConfigSnapshot( }; } -/** - * Agent 会话数据 - */ +/** Agent 会话数据 */ export interface AgentSession { /** 持久化 schema。旧 sidecar 在读取时规范化为 v2。 */ schemaVersion: 2; @@ -280,9 +272,7 @@ export class AgentSessionStore { return AgentSessionStore.instance; } - /** - * 确保存储目录存在 - */ + /** 确保存储目录存在 */ private ensureDirectory(): void { if (!fs.existsSync(this.sessionsDir)) { fs.mkdirSync(this.sessionsDir, { recursive: true, mode: 0o700 }); @@ -307,17 +297,13 @@ export class AgentSessionStore { } } - /** - * 获取会话文件路径 - */ + /** 获取会话文件路径 */ private getSessionPath(agentId: string): string { assertValidSessionId(agentId); return join(this.sessionsDir, `${agentId}.json`); } - /** - * 保存会话 - */ + /** 保存会话 */ saveSession(session: AgentSession): void { const normalized = this.normalizeSession(session, session.id); this.ensureDirectory(); @@ -332,9 +318,7 @@ export class AgentSessionStore { logger.debug(`Session saved: ${normalized.id}`); } - /** - * 加载会话 - */ + /** 加载会话 */ loadSession(agentId: string): AgentSession | undefined { // 先检查缓存 const cached = this.cache.get(agentId); @@ -471,9 +455,7 @@ export class AgentSessionStore { }; } - /** - * 更新会话状态 - */ + /** 更新会话状态 */ updateSession( agentId: string, updates: Partial @@ -493,23 +475,7 @@ export class AgentSessionStore { return updatedSession; } - /** - * 追加消息到会话 - */ - appendMessages(agentId: string, messages: Message[]): AgentSession | undefined { - const session = this.loadSession(agentId); - if (!session) { - return undefined; - } - - return this.updateSession(agentId, { - messages: [...session.messages, ...messages], - }); - } - - /** - * 标记会话完成 - */ + /** 标记会话完成 */ markCompleted( agentId: string, result: { @@ -531,9 +497,7 @@ export class AgentSessionStore { }); } - /** - * 删除会话 - */ + /** 删除会话 */ deleteSession(agentId: string): boolean { try { const filePath = this.getSessionPath(agentId); @@ -548,9 +512,7 @@ export class AgentSessionStore { } } - /** - * 列出所有会话 - */ + /** 列出所有会话 */ listSessions(): AgentSession[] { try { const files = fs.readdirSync(this.sessionsDir); @@ -578,44 +540,7 @@ export class AgentSessionStore { } } - /** - * 列出运行中的会话 - */ - listRunningSessions(): AgentSession[] { - return this.listSessions().filter((s) => s.status === 'running'); - } - - /** - * 清理过期会话 - * @param maxAgeMs 最大保留时间(毫秒),默认 7 天 - */ - cleanupExpiredSessions(maxAgeMs: number = 7 * 24 * 60 * 60 * 1000): number { - const now = Date.now(); - const sessions = this.listSessions(); - let cleaned = 0; - - for (const session of sessions) { - // 只清理已完成的会话 - if (session.status === 'running') continue; - - const age = now - session.lastActiveAt; - if (age > maxAgeMs) { - if (this.deleteSession(session.id)) { - cleaned++; - } - } - } - - if (cleaned > 0) { - logger.info(`Cleaned up ${cleaned} expired agent sessions`); - } - - return cleaned; - } - - /** - * 清空缓存 - */ + /** 清空缓存 */ clearCache(): void { this.cache.clear(); } diff --git a/packages/cli/src/agent/subagents/BackgroundAgentManager.ts b/packages/cli/src/agent/subagents/BackgroundAgentManager.ts index 2ac450ead..2dbd54c38 100644 --- a/packages/cli/src/agent/subagents/BackgroundAgentManager.ts +++ b/packages/cli/src/agent/subagents/BackgroundAgentManager.ts @@ -1,11 +1,4 @@ -/** - * 后台 Agent 管理器 - * - * 管理在后台运行的 subagent: - * - 启动后台 agent - * - 跟踪状态和输出 - * - 支持等待完成、恢复、终止 - */ +/** 后台 Agent 管理器 管理在后台运行的 subagent: - 启动后台 agent - 跟踪状态和输出 - 支持等待完成、恢复、终止 */ import { stat } from 'node:fs/promises'; import type { @@ -146,9 +139,7 @@ function lastAssistantText(messages: readonly Message[]): string { return ''; } -/** - * 后台 Agent 运行时信息 - */ +/** 后台 Agent 运行时信息 */ interface BackgroundAgentRuntime { /** Agent ID */ id: string; @@ -163,9 +154,7 @@ interface BackgroundAgentRuntime { startTime: number; } -/** - * 启动后台 Agent 的选项 - */ +/** 启动后台 Agent 的选项 */ export interface StartBackgroundAgentOptions { /** Subagent 配置 */ config: SubagentConfig; @@ -262,9 +251,7 @@ export interface ResumeAgentResult { source: AgentSession; } -/** - * 后台 Agent 管理器 - */ +/** 后台 Agent 管理器 */ export class BackgroundAgentManager { private static instance: BackgroundAgentManager | null = null; @@ -584,9 +571,7 @@ export class BackgroundAgentManager { return id; } - /** - * 执行 Agent(内部方法) - */ + /** 执行 Agent(内部方法) */ private async executeAgent( agentId: string, config: SubagentConfig, @@ -896,9 +881,7 @@ export class BackgroundAgentManager { } } - /** - * 获取 Agent 状态 - */ + /** 获取 Agent 状态 */ getAgent( agentId: string, owner?: AgentSessionOwner | string @@ -911,9 +894,7 @@ export class BackgroundAgentManager { return isAgentSessionOwnedBy(session, owner) ? session : undefined; } - /** - * 检查 Agent 是否正在运行 - */ + /** 检查 Agent 是否正在运行 */ isRunning(agentId: string): boolean { return this.runningAgents.has(agentId); } @@ -1063,9 +1044,7 @@ export class BackgroundAgentManager { return { agentId: resumedId, source: session }; } - /** - * 取消/终止 Agent - */ + /** 取消/终止 Agent */ killAgent(agentId: string, owner?: AgentSessionOwner): boolean { if (owner && !this.getAgent(agentId, owner)) return false; const runtime = this.runningAgents.get(agentId); @@ -1090,9 +1069,7 @@ export class BackgroundAgentManager { return true; } - /** - * 列出所有后台 Agent - */ + /** 列出所有后台 Agent */ listAll(): AgentSession[] { return this.sessionStore.listSessions(); } @@ -1107,36 +1084,18 @@ export class BackgroundAgentManager { ); } - /** - * 列出运行中的 Agent - */ - listRunning(): AgentSession[] { - return this.sessionStore.listRunningSessions(); - } - - /** - * 获取运行中 Agent 的数量 - */ + /** 获取运行中 Agent 的数量 */ getRunningCount(): number { return this.runningAgents.size; } - /** - * 终止所有运行中的 Agent - */ + /** 终止所有运行中的 Agent */ killAll(): void { for (const [agentId] of this.runningAgents) { this.killAgent(agentId); } } - /** - * 清理过期会话 - */ - cleanupExpiredSessions(maxAgeMs?: number): number { - return this.sessionStore.cleanupExpiredSessions(maxAgeMs); - } - cleanupExpiredSessionsForParent( owner: AgentSessionOwner | string, maxAgeMs: number = 7 * 24 * 60 * 60 * 1000 diff --git a/packages/cli/src/agent/subagents/SubagentExecutor.ts b/packages/cli/src/agent/subagents/SubagentExecutor.ts index b7e9d4a45..847abb617 100644 --- a/packages/cli/src/agent/subagents/SubagentExecutor.ts +++ b/packages/cli/src/agent/subagents/SubagentExecutor.ts @@ -28,15 +28,7 @@ import { } from './builtinVerificationAgent.js'; import type { SubagentConfig, SubagentContext, SubagentResult } from './types.js'; -/** - * Subagent 执行器 - * - * 职责: - * - 创建子 Agent 实例 - * - 配置工具白名单 - * - 执行任务并返回结果 - * - 将子代理对话流写入独立 JSONL 文件 - */ +/** Subagent 执行器 职责: - 创建子 Agent 实例 - 配置工具白名单 - 执行任务并返回结果 - 将子代理对话流写入独立 JSONL 文件 */ export class SubagentExecutor { constructor( private config: SubagentConfig, @@ -130,9 +122,7 @@ export class SubagentExecutor { signal: context.signal, }; - /** - * Phase 4: 统一通过 onEvent 转发所有 LoopEvent - */ + /** Phase 4: 统一通过 onEvent 转发所有 LoopEvent */ const onEvent = async (event: LoopEvent) => { if (event.kind === 'tool_result' && 'function' in event.toolCall) { recordModifiedFiles( diff --git a/packages/cli/src/agent/subagents/SubagentRegistry.ts b/packages/cli/src/agent/subagents/SubagentRegistry.ts index 6de090eac..1dbcd0d44 100644 --- a/packages/cli/src/agent/subagents/SubagentRegistry.ts +++ b/packages/cli/src/agent/subagents/SubagentRegistry.ts @@ -19,9 +19,7 @@ const RESERVED_BUILTIN_SUBAGENTS = new Set([ GOAL_VERIFICATION_SUBAGENT_TYPE, ]); -/** - * 配置来源类型(不包含动态的 plugin:xxx 格式) - */ +/** 配置来源类型(不包含动态的 plugin:xxx 格式) */ type ConfigSource = | 'builtin' | 'claude-code-user' @@ -32,12 +30,11 @@ type ConfigSource = | 'plugin'; /** - * Subagent 注册表 - * - * 职责: - * - 注册和发现 subagents - * - 解析 Markdown + YAML frontmatter 配置 - * - 生成 LLM 可读的描述 + + * Subagent 注册表

职责: - 注册和发现 subagents - 解析 Markdown + YAML frontmatter 配置 - 生成 LLM + + * 可读的描述 + */ export class SubagentRegistry { private static instances = new Map(); @@ -90,30 +87,22 @@ export class SubagentRegistry { } } - /** - * 获取指定 subagent - */ + /** 获取指定 subagent */ getSubagent(name: string): SubagentConfig | undefined { return this.subagents.get(name); } - /** - * 获取所有 subagent 名称 - */ + /** 获取所有 subagent 名称 */ getAllNames(): string[] { return Array.from(this.subagents.keys()); } - /** - * 获取所有 subagent 配置 - */ + /** 获取所有 subagent 配置 */ getAllSubagents(): SubagentConfig[] { return Array.from(this.subagents.values()); } - /** - * 生成 LLM 可读的 subagent 描述(用于系统提示) - */ + /** 生成 LLM 可读的 subagent 描述(用于系统提示) */ getDescriptionsForPrompt(): string { const subagents = this.getAllSubagents(); if (subagents.length === 0) { @@ -301,9 +290,7 @@ export class SubagentRegistry { return count; } - /** - * 加载内置 subagent 配置 - */ + /** 加载内置 subagent 配置 */ loadBuiltinAgents(): void { for (const agent of builtinAgents) { // 使用 set 而非 register,允许被后续配置覆盖 @@ -316,9 +303,7 @@ export class SubagentRegistry { logger.debug(`Loaded ${builtinAgents.length} builtin subagents`); } - /** - * 清空所有注册的 subagents(用于测试) - */ + /** 清空所有注册的 subagents(用于测试) */ clear(): void { this.subagents.clear(); } @@ -338,10 +323,7 @@ export class SubagentRegistry { return snapshot; } - /** - * 获取按来源分组的 subagents - * 用于 UI 展示和调试 - */ + /** 获取按来源分组的 subagents 用于 UI 展示和调试 */ getSubagentsBySource(): Record { const result: Record = { builtin: [], @@ -365,10 +347,7 @@ export class SubagentRegistry { return result; } - /** - * 清除所有插件代理 - * Called when refreshing plugins - */ + /** 清除所有插件代理 Called when refreshing plugins */ clearPluginAgents(): void { const toDelete: string[] = []; for (const [name, config] of this.subagents.entries()) { @@ -380,44 +359,11 @@ export class SubagentRegistry { this.subagents.delete(name); } } - - /** - * 获取 Claude Code 配置目录路径 - * 用于 UI 展示 - */ - static getClaudeCodeAgentsDir(type: 'user' | 'project'): string { - if (type === 'user') { - return path.join(os.homedir(), '.claude', 'agents'); - } - return path.join(getCwd(), '.claude', 'agents'); - } - - /** - * 获取 Blade 配置目录路径 - * 用于 UI 展示 - */ - static getBladeAgentsDir(type: 'user' | 'project'): string { - if (type === 'user') { - return path.join(os.homedir(), '.blade', 'agents'); - } - return path.join(getCwd(), '.blade', 'agents'); - } } -/** - * 全局单例 - */ +/** 全局单例 */ export function getSubagentRegistry( workspaceRoot: string = getCwd() ): SubagentRegistry { return SubagentRegistry.getInstance(workspaceRoot); } - -/** @deprecated Use getSubagentRegistry(workspaceRoot). */ -export const subagentRegistry = new Proxy({} as SubagentRegistry, { - get(_target, property) { - const registry = getSubagentRegistry(); - const value = Reflect.get(registry, property, registry) as unknown; - return typeof value === 'function' ? value.bind(registry) : value; - }, -}); diff --git a/packages/cli/src/agent/subagents/builtinAgents.ts b/packages/cli/src/agent/subagents/builtinAgents.ts index aefaa86b9..603efdd38 100644 --- a/packages/cli/src/agent/subagents/builtinAgents.ts +++ b/packages/cli/src/agent/subagents/builtinAgents.ts @@ -1,8 +1,6 @@ /** - * 内置 Subagent 配置 - * - * 这些 agent 是 Blade 默认提供的,与 Claude Code 保持一致。 - * 用户可以通过 ~/.blade/agents/ 或 .blade/agents/ 扩展更多 agent。 + * 内置 Subagent 配置

这些 agent 是 Blade 默认提供的,与 Claude Code 保持一致。 用户可以通过 ~/.blade/agents/ + * 或 .blade/agents/ 扩展更多 agent。 */ import { goalVerificationAgentConfig } from './builtinGoalVerificationAgent.js'; @@ -10,10 +8,7 @@ import { reviewAgentConfig } from './builtinReviewAgent.js'; import { verificationAgentConfig } from './builtinVerificationAgent.js'; import type { SubagentConfig } from './types.js'; -/** - * 内置 Subagent 列表(4 个核心 agent) - * - */ +/** 内置 Subagent 列表(4 个核心 agent) */ export const builtinAgents: SubagentConfig[] = [ { name: 'general-purpose', diff --git a/packages/cli/src/agent/subagents/builtinGoalVerificationAgent.ts b/packages/cli/src/agent/subagents/builtinGoalVerificationAgent.ts index 9176ca4cd..dec0c6f92 100644 --- a/packages/cli/src/agent/subagents/builtinGoalVerificationAgent.ts +++ b/packages/cli/src/agent/subagents/builtinGoalVerificationAgent.ts @@ -124,51 +124,7 @@ export function goalVerificationVerdictFromOutput( return goalVerificationOutputFromValue(value)?.verdict; } -const GOAL_VERIFICATION_SYSTEM_PROMPT = `# Goal Completion Verifier - -You are a fresh, independent, adversarial verifier. The parent Agent has claimed -that a persisted goal is complete. Your job is to refute that claim unless the -current workspace provides direct evidence for every explicit requirement. - -## Authority and constraints - -1. READ-ONLY. You have Read, Glob, Grep, and read-only Bash only. Never modify - files or external state. -2. NO DELEGATION. Do not call Task or any other agent. -3. OBJECTIVE IS AUTHORITATIVE. Enumerate every explicit requirement in the - block before deciding. -4. CURRENT EVIDENCE ONLY. Inspect the current workspace. Do not trust the parent - summary, claimed test output, or a previous verifier verdict. -5. MATCH VERIFICATION TO THE GOAL. Run configured tests, lint, type-check, or - build commands only when they are relevant to the objective or changed - implementation. A small artifact goal does not fail merely because the - workspace has no unrelated project checks. -6. NAMED ARTIFACTS MUST BE INSPECTED. If the objective names a file, command, - document, output, or observable behavior, verify it directly. -7. MISSING OR INDIRECT EVIDENCE IS NOT PASS. Use PARTIAL when the implementation - may be correct but a requirement cannot be proved. Use FAIL for a concrete - contradiction, failed check, defect, or missing required artifact. -8. SAFE FEEDBACK. Keep summary and findings concise, use workspace-relative - file locations, and never include credentials or secret values. -9. HOST CONTROL PLANE. This reserved verifier runs only after the host durably - accepts the parent Agent's UpdateGoal complete call. During verification, the - Goal intentionally remains status=verifying with - completionVerification.status=pending. Treat that state as an awaiting - verdict, never as evidence that UpdateGoal was omitted. Do not require - status=complete or a PASS verdict before issuing your own verdict because - only your PASS permits the host to commit them. This proves only the - completion-candidate control action, not the requested deliverables. - -## Verdict - -Submit the host-requested structured final object with: - -- verdict: pass, fail, or partial -- summary: concise requirement-by-requirement conclusion -- findings: concrete gaps with file or command locators - -PASS is allowed only when every requirement is directly proven and no relevant -check fails.`; +import GOAL_VERIFICATION_SYSTEM_PROMPT from './goal-verification.md?raw'; export const goalVerificationAgentConfig: SubagentConfig = { name: GOAL_VERIFICATION_SUBAGENT_TYPE, diff --git a/packages/cli/src/agent/subagents/builtinReviewAgent.ts b/packages/cli/src/agent/subagents/builtinReviewAgent.ts index e87be2f12..8bdc9649a 100644 --- a/packages/cli/src/agent/subagents/builtinReviewAgent.ts +++ b/packages/cli/src/agent/subagents/builtinReviewAgent.ts @@ -1,64 +1,8 @@ import { PermissionMode } from '../../config/types.js'; import { REVIEW_SUBAGENT_TYPE } from '../../utils/shell/readOnlyAudit.js'; +import REVIEW_SYSTEM_PROMPT from './review.md?raw'; import type { SubagentConfig } from './types.js'; -const REVIEW_SYSTEM_PROMPT = `# Code Review Agent - -You are an independent senior code reviewer. Find actionable defects introduced -by the requested change. Do not implement fixes and do not praise the author. - -## Security boundary - -- You are strictly read-only. Never modify files, Git state, configuration, or - dependencies. -- Do not use network tools or delegate to another agent. -- Use Git and file-reading tools to inspect the exact target supplied by the - host. Do not silently expand the review to unrelated pre-existing code. -- A command may run only when it is read-only or an existing verification - command. The runtime enforces this independently of these instructions. - -## Finding rules - -Report a finding only when all of these are true: - -1. It is caused by the reviewed change. -2. It has a concrete correctness, security, reliability, performance, or - maintainability impact. -3. The author can act on it independently. -4. You can cite the smallest relevant changed line range. - -Priority: - -- 0: release blocker or broadly exploitable issue. -- 1: high-impact defect that should be fixed before merge. -- 2: normal defect worth fixing. -- 3: low-impact issue; omit pure style preferences. - -## Output contract - -Your final response must be exactly one JSON object and no Markdown fence: - -{ - "overall_explanation": "Brief assessment grounded in evidence.", - "findings": [ - { - "title": "[P1] Imperative title, at most 80 characters", - "body": "Why this is a defect, the triggering scenario, and impact.", - "priority": 1, - "confidence_score": 0.98, - "code_location": { - "path": "relative/path/to/file.ts", - "line_start": 10, - "line_end": 12 - } - } - ] -} - -Use workspace-relative paths. Keep line ranges within 10 lines and overlapping -the reviewed diff. Return an empty findings array when no qualifying defect is -found.`; - export const reviewAgentConfig: SubagentConfig = { name: REVIEW_SUBAGENT_TYPE, description: diff --git a/packages/cli/src/agent/subagents/builtinVerificationAgent.ts b/packages/cli/src/agent/subagents/builtinVerificationAgent.ts index 9ac7636f6..458e9595d 100644 --- a/packages/cli/src/agent/subagents/builtinVerificationAgent.ts +++ b/packages/cli/src/agent/subagents/builtinVerificationAgent.ts @@ -1,9 +1,4 @@ -/** - * 内置验证 Subagent 配置 - * - * 独立验证 Agent,用于在实现完成后进行质量评估。 - * 严格只读 — 不能修改代码,只能运行构建、测试、lint 和对抗性检查。 - */ +/** 内置验证 Subagent 配置 独立验证 Agent,用于在实现完成后进行质量评估。 严格只读 — 不能修改代码,只能运行构建、测试、lint 和对抗性检查。 */ import type { JsonObject } from '../../store/types.js'; import type { SubagentConfig } from './types.js'; @@ -46,121 +41,7 @@ export function independentVerificationVerdictFromOutput( : undefined; } -/** - * 验证 Agent 系统提示 - */ -const VERIFICATION_SYSTEM_PROMPT = `# Verification Agent - -You are an **independent verification engineer**. Your sole purpose \ -is to find problems — not to praise or reassure. You are the last \ -line of defense before code ships. - -## Constraints - -1. **READ-ONLY**: You have NO write tools (no Edit, Write, ApplyPatch, or \ -NotebookEdit). You cannot modify files. If you discover issues, \ -report them — do not attempt to fix them. -2. **NO SUB-AGENTS**: You must not delegate to other agents or use \ -the Task tool. Execute all verification steps yourself using your \ -tools directly. -3. **TOOL-BASED EVIDENCE ONLY**: Every claim must be backed by \ -actual tool output. Never say "looks correct" or "should work" — \ -run the command and prove it. -4. **NO ASSUMPTIONS**: Do not assume tests pass. Do not assume types \ -are correct. Run the checks. -5. **CONVERGE**: Never repeat a file read, search, or verification \ -command after it has produced conclusive evidence. Once the configured \ -checks and changed-file review are complete, emit the verdict immediately. - -## Verification Workflow - -Execute these phases in order. Do NOT skip any phase. - -### Phase 1: Project Setup Detection - -1. Use Glob to find project config files: \`package.json\`, \ -\`tsconfig.json\`, \`biome.json\`, \`.eslintrc.*\`, \ -\`vitest.config.*\`, \`jest.config.*\`, \`Makefile\`, \ -\`Cargo.toml\`, \`go.mod\`, etc. -2. Use Read to examine them and determine: - - Package manager (bun/npm/pnpm/yarn) - - Available scripts (test, lint, type-check, build) - - Project language and framework -3. Identify which checks are available for this project. - -### Phase 2: Automated Checks - -Run all applicable checks. Capture full output. - -| Check | Typical Command | Priority | -|-------|----------------|----------| -| **Type checking** | \`bun run type-check\` or \`npx tsc --noEmit\` | HIGH | -| **Tests** | \`bun run test:all\` or \`npm test\` | HIGH | -| **Linting** | \`bun run lint\` or \`npx biome check\` | HIGH | -| **Build** | Read-only equivalent such as \`npx tsc --noEmit\` | MEDIUM | - -- If a command fails, record the exact error output. -- If a command succeeds, record confirmation. -- Set reasonable timeouts (use Bash timeout parameter). -- The audit workspace is intentionally read-only. Do not install dependencies or run - build commands that emit artifacts into the workspace. Inspect the configured build - script and use a no-write equivalent when available. If no safe equivalent exists, - report the build as not rerun; do not treat the sandbox write denial itself as a - product failure. -- Output may be bounded with a single numeric \`head\` or \`tail\` pipeline. Do not use - \`tee\`, redirects, or chained output filters. -- The Bash tool preserves the verification command's exit status when applying that - output bound. Do not append a shell status probe or any additional command. - -### Phase 3: Code Review of Changed Files - -1. Treat the changed-file list supplied by the parent as authoritative. \ -Use \`git status --short\` and \`git diff --name-only\` only to discover \ -additional uncommitted changes. Do not use \`HEAD~1\` as a substitute for \ -the supplied scope. -2. Read each changed file and review for: - - **Logic errors**: off-by-one, null/undefined handling, race \ -conditions - - **Type safety**: any casts, type assertions, missing null checks - - **Error handling**: uncaught exceptions, missing error paths - - **Edge cases**: empty arrays, empty strings, boundary values - - **Security**: injection risks, credential exposure, unsafe eval - - **Code style**: naming conventions, dead code, commented-out code - -### Phase 4: Adversarial Analysis - -Think like an attacker or a hostile user: - -1. **Input validation**: Are all inputs validated? What happens with \ -malformed data? -2. **Boundary conditions**: What happens at limits? (max length, \ -zero, negative) -3. **Concurrency**: Are there race conditions or shared mutable \ -state issues? -4. **Dependency risks**: Are new dependencies trustworthy? Pinned \ -versions? -5. **Regression potential**: Could these changes break existing \ -functionality? - -## Output Format - -Reserve your final model turn for the host-requested structured output object. -Submit exactly these fields: - -- verdict: pass, fail, or partial -- summary: concise automated-check and code-review conclusion -- findings: concrete findings with file, command, or output evidence - -### Verdict Rules - -- **PASS**: All automated checks pass AND no HIGH severity issues \ -found. -- **FAIL**: Any automated check fails OR any HIGH severity issue \ -found. -- **PARTIAL**: All automated checks pass BUT MEDIUM severity issues \ -exist. - -Be thorough. Be skeptical. Find the bugs.`; +import VERIFICATION_SYSTEM_PROMPT from './verification.md?raw'; /** * 验证 Agent 配置 diff --git a/packages/cli/src/agent/subagents/goal-verification.md b/packages/cli/src/agent/subagents/goal-verification.md new file mode 100644 index 000000000..5e9da38dd --- /dev/null +++ b/packages/cli/src/agent/subagents/goal-verification.md @@ -0,0 +1,45 @@ +# Goal Completion Verifier + +You are a fresh, independent, adversarial verifier. The parent Agent has claimed +that a persisted goal is complete. Your job is to refute that claim unless the +current workspace provides direct evidence for every explicit requirement. + +## Authority and constraints + +1. READ-ONLY. You have Read, Glob, Grep, and read-only Bash only. Never modify + files or external state. +2. NO DELEGATION. Do not call Task or any other agent. +3. OBJECTIVE IS AUTHORITATIVE. Enumerate every explicit requirement in the + block before deciding. +4. CURRENT EVIDENCE ONLY. Inspect the current workspace. Do not trust the parent + summary, claimed test output, or a previous verifier verdict. +5. MATCH VERIFICATION TO THE GOAL. Run configured tests, lint, type-check, or + build commands only when they are relevant to the objective or changed + implementation. A small artifact goal does not fail merely because the + workspace has no unrelated project checks. +6. NAMED ARTIFACTS MUST BE INSPECTED. If the objective names a file, command, + document, output, or observable behavior, verify it directly. +7. MISSING OR INDIRECT EVIDENCE IS NOT PASS. Use PARTIAL when the implementation + may be correct but a requirement cannot be proved. Use FAIL for a concrete + contradiction, failed check, defect, or missing required artifact. +8. SAFE FEEDBACK. Keep summary and findings concise, use workspace-relative + file locations, and never include credentials or secret values. +9. HOST CONTROL PLANE. This reserved verifier runs only after the host durably + accepts the parent Agent's UpdateGoal complete call. During verification, the + Goal intentionally remains status=verifying with + completionVerification.status=pending. Treat that state as an awaiting + verdict, never as evidence that UpdateGoal was omitted. Do not require + status=complete or a PASS verdict before issuing your own verdict because + only your PASS permits the host to commit them. This proves only the + completion-candidate control action, not the requested deliverables. + +## Verdict + +Submit the host-requested structured final object with: + +- verdict: pass, fail, or partial +- summary: concise requirement-by-requirement conclusion +- findings: concrete gaps with file or command locators + +PASS is allowed only when every requirement is directly proven and no relevant +check fails. \ No newline at end of file diff --git a/packages/cli/src/agent/subagents/review.md b/packages/cli/src/agent/subagents/review.md new file mode 100644 index 000000000..2e652b358 --- /dev/null +++ b/packages/cli/src/agent/subagents/review.md @@ -0,0 +1,56 @@ +# Code Review Agent + +You are an independent senior code reviewer. Find actionable defects introduced +by the requested change. Do not implement fixes and do not praise the author. + +## Security boundary + +- You are strictly read-only. Never modify files, Git state, configuration, or + dependencies. +- Do not use network tools or delegate to another agent. +- Use Git and file-reading tools to inspect the exact target supplied by the + host. Do not silently expand the review to unrelated pre-existing code. +- A command may run only when it is read-only or an existing verification + command. The runtime enforces this independently of these instructions. + +## Finding rules + +Report a finding only when all of these are true: + +1. It is caused by the reviewed change. +2. It has a concrete correctness, security, reliability, performance, or + maintainability impact. +3. The author can act on it independently. +4. You can cite the smallest relevant changed line range. + +Priority: + +- 0: release blocker or broadly exploitable issue. +- 1: high-impact defect that should be fixed before merge. +- 2: normal defect worth fixing. +- 3: low-impact issue; omit pure style preferences. + +## Output contract + +Your final response must be exactly one JSON object and no Markdown fence: + +{ + "overall_explanation": "Brief assessment grounded in evidence.", + "findings": [ + { + "title": "[P1] Imperative title, at most 80 characters", + "body": "Why this is a defect, the triggering scenario, and impact.", + "priority": 1, + "confidence_score": 0.98, + "code_location": { + "path": "relative/path/to/file.ts", + "line_start": 10, + "line_end": 12 + } + } + ] +} + +Use workspace-relative paths. Keep line ranges within 10 lines and overlapping +the reviewed diff. Return an empty findings array when no qualifying defect is +found. \ No newline at end of file diff --git a/packages/cli/src/agent/subagents/types.ts b/packages/cli/src/agent/subagents/types.ts index a991dc95d..b84fdc2ea 100644 --- a/packages/cli/src/agent/subagents/types.ts +++ b/packages/cli/src/agent/subagents/types.ts @@ -1,7 +1,3 @@ -/** - * Subagent 系统类型定义 - */ - import { type CommunicationStyleSelection, PermissionMode, @@ -14,11 +10,6 @@ import type { WorktreeSession } from '../../worktree/WorktreeManager.js'; import type { VerificationVerdict } from '../loop/independentVerification.js'; import type { LoopEvent } from '../loop/types.js'; import type { SubagentIsolationMode } from './SubagentWorktreeLifecycle.js'; - -/** - * Claude Code permissionMode 类型 - * 参考: https://code.claude.com/docs/en/sub-agents - */ export type ClaudeCodePermissionMode = | 'default' | 'acceptEdits' @@ -58,9 +49,6 @@ export function mapClaudeCodePermissionMode( } } -/** - * Subagent 背景颜色 - */ export type SubagentColor = | 'red' | 'blue' @@ -70,30 +58,13 @@ export type SubagentColor = | 'orange' | 'pink' | 'cyan'; - -/** - * Subagent 配置 - */ export interface SubagentConfig { - /** Subagent 唯一标识符 */ name: string; - - /** 描述(给 LLM 看的能力说明) */ description: string; - - /** 系统提示模板(可选,支持变量替换) */ systemPrompt?: string; - - /** 允许的工具列表(空数组 = 所有工具) */ tools?: string[]; - - /** 禁止的工具列表(优先于允许列表) */ disallowedTools?: string[]; - - /** UI 背景颜色(可选,用于视觉区分) */ color?: SubagentColor; - - /** 配置文件路径(用于调试) */ configPath?: string; /** @@ -102,20 +73,10 @@ export interface SubagentConfig { * - 注意:Blade 目前不支持多模型,此字段仅用于兼容 Claude Code 配置 */ model?: 'sonnet' | 'opus' | 'haiku' | 'inherit' | string; - - /** 权限模式(已映射为 Blade PermissionMode) */ permissionMode?: PermissionMode; - - /** 最大对话轮次 */ maxTurns?: number; - - /** 自动加载的 skills 列表 */ skills?: string[]; - - /** 默认文件系统隔离模式 */ isolation?: SubagentIsolationMode; - - /** 配置来源(用于调试和优先级) */ source?: | 'builtin' | 'claude-code-user' @@ -134,35 +95,18 @@ export interface SubagentConfig { * - Phase 4 完成:旧命名回调已删除,统一走 onEvent */ export interface SubagentContext { - /** 任务提示 */ prompt: string; - - /** 父 Agent 的会话 ID(可选,用于追溯) */ parentSessionId?: string; /** Root Session owning Provider request admission for the full child tree. */ providerAdmissionOwnerId?: string; - - /** 父 Agent 的消息 ID(可选) */ parentMessageId?: string; - - /** 父 Agent 的权限模式(继承给子 Agent) */ permissionMode?: PermissionMode; - - /** 父 Session 当前的 durable reasoning 策略 */ modelId?: string; reasoningEffort?: ReasoningEffortSelection; - - /** 父 Session 当前的 provider service tier */ serviceTier?: ServiceTierSelection; - - /** 父 Session 当前的 response verbosity */ responseVerbosity?: ResponseVerbositySelection; - - /** 父 Session 当前的 communication style */ communicationStyle?: CommunicationStyleSelection; - - /** 子代理会话 ID(用于与主会话关联) */ subagentSessionId?: string; /** Foreground cancellation boundary owned by the invoking surface. */ @@ -176,8 +120,6 @@ export interface SubagentContext { /** Resume depth from the root */ resumeDepth?: number; - - /** 子代理执行目录(默认继承父 Agent) */ workspaceRoot?: string; /** 子代理是否已位于预创建的 managed worktree */ @@ -185,37 +127,18 @@ export interface SubagentContext { /** Resume 时继承的完整模型历史 */ existingMessages?: Message[]; - - /** - * 统一事件回调 - * SubagentExecutor 直接转发所有 LoopEvent。 - */ onEvent?: (event: LoopEvent) => void | Promise; } -/** - * Subagent 执行结果 - */ export interface SubagentResult { - /** 执行是否成功 */ success: boolean; - - /** 结果消息 */ message: string; - - /** 错误信息(如果失败) */ error?: string; - - /** 子代理会话 ID(用于关联独立 JSONL 文件) */ agentId?: string; /** 执行结束后的完整模型历史,用于 durable resume */ messages?: Message[]; - - /** 保留的隔离 worktree 路径(无改动自动清理时为空) */ worktreePath?: string; - - /** 保留的隔离 worktree 分支 */ worktreeBranch?: string; /** 用于后台 resume 的完整 worktree lease */ @@ -229,19 +152,10 @@ export interface SubagentResult { /** Goal verifier 的有界、脱敏修复反馈 */ verificationFeedback?: string; - - /** 子代理执行期间成功修改的文件路径 */ modifiedFiles?: string[]; - - /** 执行统计 */ stats?: { - /** Token 使用量 */ tokens?: number; - - /** 工具调用次数 */ toolCalls?: number; - - /** 执行时长(毫秒) */ duration?: number; }; } @@ -258,18 +172,12 @@ export interface SubagentResult { export interface SubagentFrontmatter { name: string; description: string; - /** 工具列表(逗号分隔字符串或数组),不指定则继承所有工具 */ tools?: string[] | string; - /** UI 背景颜色 */ color?: SubagentColor; /** 模型别名(sonnet/opus/haiku)或 'inherit' */ model?: 'sonnet' | 'opus' | 'haiku' | 'inherit' | string; - /** 权限模式(Claude Code 格式,将被映射为 Blade PermissionMode) */ permissionMode?: ClaudeCodePermissionMode; - /** 自动加载的 skills 列表(逗号分隔字符串或数组) */ skills?: string[] | string; - /** 默认文件系统隔离模式 */ isolation?: SubagentIsolationMode; - /** 许可证信息(Claude Code skills 格式) */ license?: string; } diff --git a/packages/cli/src/agent/subagents/verification.md b/packages/cli/src/agent/subagents/verification.md new file mode 100644 index 000000000..58a0ca97c --- /dev/null +++ b/packages/cli/src/agent/subagents/verification.md @@ -0,0 +1,86 @@ +# Verification Agent + +You are an **independent verification engineer**. Your sole purpose is to find problems — not to praise or reassure. You are the last line of defense before code ships. + +## Constraints + +1. **READ-ONLY**: You have NO write tools (no Edit, Write, ApplyPatch, or NotebookEdit). You cannot modify files. If you discover issues, report them — do not attempt to fix them. +2. **NO SUB-AGENTS**: You must not delegate to other agents or use the Task tool. Execute all verification steps yourself using your tools directly. +3. **TOOL-BASED EVIDENCE ONLY**: Every claim must be backed by actual tool output. Never say "looks correct" or "should work" — run the command and prove it. +4. **NO ASSUMPTIONS**: Do not assume tests pass. Do not assume types are correct. Run the checks. +5. **CONVERGE**: Never repeat a file read, search, or verification command after it has produced conclusive evidence. Once the configured checks and changed-file review are complete, emit the verdict immediately. + +## Verification Workflow + +Execute these phases in order. Do NOT skip any phase. + +### Phase 1: Project Setup Detection + +1. Use Glob to find project config files: `package.json`, `tsconfig.json`, `biome.json`, `.eslintrc.*`, `vitest.config.*`, `jest.config.*`, `Makefile`, `Cargo.toml`, `go.mod`, etc. +2. Use Read to examine them and determine: + - Package manager (bun/npm/pnpm/yarn) + - Available scripts (test, lint, type-check, build) + - Project language and framework +3. Identify which checks are available for this project. + +### Phase 2: Automated Checks + +Run all applicable checks. Capture full output. + +| Check | Typical Command | Priority | +|-------|----------------|----------| +| **Type checking** | `bun run type-check` or `npx tsc --noEmit` | HIGH | +| **Tests** | `bun run test:all` or `npm test` | HIGH | +| **Linting** | `bun run lint` or `npx biome check` | HIGH | +| **Build** | Read-only equivalent such as `npx tsc --noEmit` | MEDIUM | + +- If a command fails, record the exact error output. +- If a command succeeds, record confirmation. +- Set reasonable timeouts (use Bash timeout parameter). +- The audit workspace is intentionally read-only. Do not install dependencies or run + build commands that emit artifacts into the workspace. Inspect the configured build + script and use a no-write equivalent when available. If no safe equivalent exists, + report the build as not rerun; do not treat the sandbox write denial itself as a + product failure. +- Output may be bounded with a single numeric `head` or `tail` pipeline. Do not use + `tee`, redirects, or chained output filters. +- The Bash tool preserves the verification command's exit status when applying that + output bound. Do not append a shell status probe or any additional command. + +### Phase 3: Code Review of Changed Files + +1. Treat the changed-file list supplied by the parent as authoritative. Use `git status --short` and `git diff --name-only` only to discover additional uncommitted changes. Do not use `HEAD~1` as a substitute for the supplied scope. +2. Read each changed file and review for: + - **Logic errors**: off-by-one, null/undefined handling, race conditions + - **Type safety**: any casts, type assertions, missing null checks + - **Error handling**: uncaught exceptions, missing error paths + - **Edge cases**: empty arrays, empty strings, boundary values + - **Security**: injection risks, credential exposure, unsafe eval + - **Code style**: naming conventions, dead code, commented-out code + +### Phase 4: Adversarial Analysis + +Think like an attacker or a hostile user: + +1. **Input validation**: Are all inputs validated? What happens with malformed data? +2. **Boundary conditions**: What happens at limits? (max length, zero, negative) +3. **Concurrency**: Are there race conditions or shared mutable state issues? +4. **Dependency risks**: Are new dependencies trustworthy? Pinned versions? +5. **Regression potential**: Could these changes break existing functionality? + +## Output Format + +Reserve your final model turn for the host-requested structured output object. +Submit exactly these fields: + +- verdict: pass, fail, or partial +- summary: concise automated-check and code-review conclusion +- findings: concrete findings with file, command, or output evidence + +### Verdict Rules + +- **PASS**: All automated checks pass AND no HIGH severity issues found. +- **FAIL**: Any automated check fails OR any HIGH severity issue found. +- **PARTIAL**: All automated checks pass BUT MEDIUM severity issues exist. + +Be thorough. Be skeptical. Find the bugs. \ No newline at end of file diff --git a/packages/cli/src/agent/teams/TeamCoordinator.ts b/packages/cli/src/agent/teams/TeamCoordinator.ts index 67d06b1e6..59b430a2b 100644 --- a/packages/cli/src/agent/teams/TeamCoordinator.ts +++ b/packages/cli/src/agent/teams/TeamCoordinator.ts @@ -33,9 +33,4 @@ export class TeamCoordinator { ); return { completedTaskIds, unblockedTasks }; } - - async isComplete(): Promise { - const tasks = await this.taskGraph.listTasks(); - return tasks.length > 0 && tasks.every((task) => task.status === 'completed'); - } } diff --git a/packages/cli/src/agent/types.ts b/packages/cli/src/agent/types.ts index fb5d9f488..1e188f43a 100644 --- a/packages/cli/src/agent/types.ts +++ b/packages/cli/src/agent/types.ts @@ -1,6 +1,4 @@ -/** - * Agent核心类型定义 - */ +/** Agent核心类型定义 */ import type { FollowUpQueueSnapshot } from '../api/followUpQueueSchemas.js'; import type { PermissionConfig } from '../config/types.js'; @@ -28,15 +26,10 @@ import type { import type { TaskAdmissionHandle } from './runtime/TaskRunScheduler.js'; import type { SubagentConfig } from './subagents/types.js'; -/** - * 用户消息内容类型 - * 支持纯文本或多模态内容(文本 + 图片) - */ +/** 用户消息内容类型 支持纯文本或多模态内容(文本 + 图片) */ export type UserMessageContent = string | ContentPart[]; -/** - * 子代理信息(用于 JSONL 写入) - */ +/** 子代理信息(用于 JSONL 写入) */ export interface SubagentInfoForContext { parentSessionId: string; providerAdmissionOwnerId?: string; @@ -48,13 +41,11 @@ export interface SubagentInfoForContext { } /** - * 聊天上下文接口 - * - * 职责:保存会话相关的数据和状态 - * - 消息历史、会话标识、用户标识等数据 - * - 会话级别的 UI 交互处理器(如 confirmationHandler) - * - * 不包含:循环过程中的事件回调(这些应该放在 LoopOptions) + + * 聊天上下文接口

职责:保存会话相关的数据和状态 - 消息历史、会话标识、用户标识等数据 - 会话级别的 UI 交互处理器(如 + + * confirmationHandler)

不包含:循环过程中的事件回调(这些应该放在 LoopOptions) + */ export interface ChatContext { messages: Message[]; @@ -77,10 +68,7 @@ export interface ChatContext { subagentInfo?: SubagentInfoForContext; // 子代理信息(用于 JSONL 写入) } -/** - * Agent 创建选项 - 仅包含运行时参数 - * Agent 的配置来自 Store (通过 getConfig() 获取 BladeConfig) - */ +/** Agent 创建选项 - 仅包含运行时参数 Agent 的配置来自 Store (通过 getConfig() 获取 BladeConfig) */ export interface AgentOptions { sessionId?: string; // 运行时参数 @@ -116,16 +104,15 @@ export interface AgentResponse { // ===== Agentic Loop Types ===== /** - * Agentic Loop 选项 - * - * 职责:控制循环行为 - * - 循环控制参数(maxTurns, autoCompact 等) - * - 行为回调(onToolApprove, onToolResult, onTurnLimitReached) - * - * 设计原则: - * - Phase 4 完成:事件通知回调已移除,消费者通过 chatStream() + LoopEvent 获取事件 - * - 保留的回调都是 behavioral(影响循环控制流),不是 notification - * - 和 ChatContext 职责分离:LoopOptions = 行为控制,ChatContext = 数据状态 + + * Agentic Loop 选项

职责:控制循环行为 - 循环控制参数(maxTurns, autoCompact 等) - + + * 行为回调(onToolApprove, onToolResult, onTurnLimitReached)

设计原则: - Phase 4 + + * 完成:事件通知回调已移除,消费者通过 chatStream() + LoopEvent 获取事件 - 保留的回调都是 behavioral(影响循环控制流),不是 + + * notification - 和 ChatContext 职责分离:LoopOptions = 行为控制,ChatContext = 数据状态 + */ export interface LoopOptions { // 循环控制参数 @@ -197,9 +184,7 @@ export interface LoopOptions { onTurnLimitReached?: (data: { turnsCount: number }) => Promise; } -/** - * 轮次限制响应 - */ +/** 轮次限制响应 */ export interface TurnLimitResponse { continue: boolean; reason?: string; diff --git a/packages/cli/src/api/sessionTitle.ts b/packages/cli/src/api/sessionTitle.ts index f6ef3a845..e25387b65 100644 --- a/packages/cli/src/api/sessionTitle.ts +++ b/packages/cli/src/api/sessionTitle.ts @@ -1,12 +1,10 @@ /** - * Shared session-title derivation. - * - * Production coding agents (Codex, Grok Build, Claude Code) name a session - * after its opening intent rather than a timestamp. We take the deterministic - * route: derive a concise, human-scannable title from the first user message. - * Deterministic derivation has zero latency, needs no LLM round-trip, and works - * identically across CLI, Web, and ACP — the single source of truth lives here - * so all three surfaces stay consistent. + * Shared session-title derivation.

Production coding agents (Codex, Grok Build, + * Claude Code) name a session after its opening intent rather than a timestamp. We take + * the deterministic route: derive a concise, human-scannable title from the first user + * message. Deterministic derivation has zero latency, needs no LLM round-trip, and + * works identically across CLI, Web, and ACP — the single source of truth lives here so + * all three surfaces stay consistent. */ const MAX_TITLE_LENGTH = 60; @@ -75,9 +73,7 @@ export function deriveSessionTitle(raw: string): string { return `${base.trimEnd()}…`; } -/** - * Derive a title directly from a message content value, flattening it first. - */ +/** Derive a title directly from a message content value, flattening it first. */ export function deriveSessionTitleFromContent(content: unknown): string { return deriveSessionTitle(flattenMessageText(content)); } diff --git a/packages/cli/src/api/sideConversation.ts b/packages/cli/src/api/sideConversation.ts index 5103260f1..068750032 100644 --- a/packages/cli/src/api/sideConversation.ts +++ b/packages/cli/src/api/sideConversation.ts @@ -1,4 +1,3 @@ -export const SIDE_CONVERSATION_COMMAND = 'btw'; export const MAX_SIDE_QUESTION_CHARS = 16 * 1024; export interface ParsedSideConversationCommand { diff --git a/packages/cli/src/bootstrap/state.ts b/packages/cli/src/bootstrap/state.ts index 009384902..1975448aa 100644 --- a/packages/cli/src/bootstrap/state.ts +++ b/packages/cli/src/bootstrap/state.ts @@ -1,10 +1,7 @@ /** - * 全局 CWD 状态单例 - * - * 参考 Claude Code 的 bootstrap/state.ts 设计: - * - cwd: 当前工作目录,可被 worktree 等场景改变 - * - originalCwd: 进程启动时的原始目录 - * - projectRoot: 稳定的项目根目录,用于项目标识(history, skills, sessions),启动后不变 + * 全局 CWD 状态单例

参考 Claude Code 的 bootstrap/state.ts 设计: - cwd: 当前工作目录,可被 worktree + * 等场景改变 - originalCwd: 进程启动时的原始目录 - projectRoot: 稳定的项目根目录,用于项目标识(history, skills, + * sessions),启动后不变 */ import { realpathSync } from 'fs'; @@ -65,10 +62,6 @@ export function setOriginalCwd(newCwd: string): void { getState().originalCwd = newCwd.normalize('NFC'); } -export function getProjectRoot(): string { - return getState().projectRoot; -} - export function setProjectRoot(root: string): void { getState().projectRoot = root.normalize('NFC'); } diff --git a/packages/cli/src/browser/BrowserProcessPool.ts b/packages/cli/src/browser/BrowserProcessPool.ts index 0aff92fe1..fc5430df3 100644 --- a/packages/cli/src/browser/BrowserProcessPool.ts +++ b/packages/cli/src/browser/BrowserProcessPool.ts @@ -263,9 +263,3 @@ export function getBrowserProcessPool(): BrowserProcessPool { defaultPool ??= new BrowserProcessPool(); return defaultPool; } - -export async function disposeBrowserProcessPool(): Promise { - const pool = defaultPool; - defaultPool = undefined; - await pool?.dispose(); -} diff --git a/packages/cli/src/browser/SessionBrowserRuntime.ts b/packages/cli/src/browser/SessionBrowserRuntime.ts index f3db2cce4..d627787e9 100644 --- a/packages/cli/src/browser/SessionBrowserRuntime.ts +++ b/packages/cli/src/browser/SessionBrowserRuntime.ts @@ -571,14 +571,20 @@ export class SessionBrowserRuntime { const generation = this.runtimeGeneration; let actionFailed = false; let actionError: unknown; - const downloadPromise = - options.action.kind === 'click' || coordinateAction - ? state.page - .waitForEvent('download', { - timeout: Math.min(timeout, BROWSER_CLICK_SETTLE_TIMEOUT_MS), - }) - .catch(() => undefined) - : undefined; + const observesClickSideEffects = + options.action.kind === 'click' || coordinateAction; + const clickSettleTimeout = Math.min(timeout, BROWSER_CLICK_SETTLE_TIMEOUT_MS); + const downloadPromise = observesClickSideEffects + ? state.page + .waitForEvent('download', { timeout: clickSettleTimeout }) + .catch(() => undefined) + : undefined; + const popupRegistrationPromise = observesClickSideEffects + ? state.page + .waitForEvent('popup', { timeout: clickSettleTimeout }) + .then((popup) => this.trackDiscoveredPage(popup, state.id)) + .catch(() => undefined) + : undefined; try { if (options.action.kind === 'click') { state.nextDialogAction = options.action.dialog?.action; @@ -586,9 +592,10 @@ export class SessionBrowserRuntime { state.nextDialogAction = coordinateAction.dialog?.action; } await this.executeAction(state.page, locator, options.action, timeout, signal); - const download = downloadPromise - ? await raceWithAbort(downloadPromise, signal) - : undefined; + const [download] = await raceWithAbort( + Promise.all([downloadPromise, popupRegistrationPromise]), + signal + ); if (download) { this.trackDownloadCancellation(state, download); } @@ -963,25 +970,6 @@ export class SessionBrowserRuntime { }, options.signal); } - stats(): { - pages: number; - pending: number; - active: boolean; - generation: number; - hasContext: boolean; - disposed: boolean; - } { - const gate = this.gate.stats(); - return { - pages: this.pages.size, - pending: gate.pending, - active: gate.active, - generation: this.runtimeGeneration, - hasContext: this.context !== undefined, - disposed: this.disposed, - }; - } - async dispose(): Promise { if (this.disposed) return; this.disposed = true; diff --git a/packages/cli/src/browser/index.ts b/packages/cli/src/browser/index.ts deleted file mode 100644 index 4843ddc47..000000000 --- a/packages/cli/src/browser/index.ts +++ /dev/null @@ -1,9 +0,0 @@ -export * from './BrowserArtifactStore.js'; -export * from './BrowserInstallation.js'; -export * from './BrowserOperationGate.js'; -export * from './BrowserProcessPool.js'; -export * from './BrowserSecurity.js'; -export * from './BrowserSnapshotAuthority.js'; -export * from './constants.js'; -export * from './SessionBrowserRuntime.js'; -export * from './types.js'; diff --git a/packages/cli/src/cli/config.ts b/packages/cli/src/cli/config.ts index 34e28273b..37ba64d42 100644 --- a/packages/cli/src/cli/config.ts +++ b/packages/cli/src/cli/config.ts @@ -1,7 +1,4 @@ -/** - * Yargs 配置文件 - * 定义所有全局选项和命令结构 - */ +/** Yargs 配置文件 定义所有全局选项和命令结构 */ import type { Options } from 'yargs'; import { getDescription, getVersion } from '../utils/packageInfo.js'; diff --git a/packages/cli/src/cli/middleware.ts b/packages/cli/src/cli/middleware.ts index 0eec40434..88dff5f1a 100644 --- a/packages/cli/src/cli/middleware.ts +++ b/packages/cli/src/cli/middleware.ts @@ -5,18 +5,13 @@ import type { GlobalOptions } from './types.js'; const logger = createLogger(LogCategory.GENERAL); -/** - * Yargs 中间件 - * 处理全局逻辑,如权限验证、配置加载等 - */ +/** Yargs 中间件 处理全局逻辑,如权限验证、配置加载等 */ import type { MiddlewareFunction } from 'yargs'; import { parseCliAgents } from './agents.js'; import { applyCliSettingsToArguments, loadCliSettings } from './settings.js'; -/** - * 权限验证中间件 - */ +/** 权限验证中间件 */ export const validatePermissions: MiddlewareFunction = (argv) => { // 处理 --yolo 快捷方式 if (argv.yolo) { @@ -61,10 +56,7 @@ export const validatePermissions: MiddlewareFunction = (argv) => { } }; -/** - * 配置加载中间件 - * 所有命令(包括 UI 模式)都会执行,负责初始化 ConfigManager 和 Store - */ +/** 配置加载中间件 所有命令(包括 UI 模式)都会执行,负责初始化 ConfigManager 和 Store */ export const loadConfiguration: MiddlewareFunction = async (argv) => { const cliSettings = await loadCliSettings( typeof argv.settings === 'string' ? argv.settings : undefined @@ -119,9 +111,7 @@ function validateSessionOptions(argv: Record): void { } } -/** - * 输出格式验证中间件 - */ +/** 输出格式验证中间件 */ export const validateOutput: MiddlewareFunction = (argv) => { if (argv.jsonSchema && argv.outputSchema) { throw new Error('--json-schema cannot be combined with --output-schema'); diff --git a/packages/cli/src/cli/types.ts b/packages/cli/src/cli/types.ts index 526c90b41..63e80cf3e 100644 --- a/packages/cli/src/cli/types.ts +++ b/packages/cli/src/cli/types.ts @@ -1,6 +1,4 @@ -/** - * Yargs CLI 类型定义 - */ +/** Yargs CLI 类型定义 */ export interface GlobalOptions { debug?: string; diff --git a/packages/cli/src/commands/doctor.ts b/packages/cli/src/commands/doctor.ts index 1ebf58451..d17b15ee1 100644 --- a/packages/cli/src/commands/doctor.ts +++ b/packages/cli/src/commands/doctor.ts @@ -1,6 +1,4 @@ -/** - * Doctor 命令 - Yargs 版本 - */ +/** Doctor 命令 - Yargs 版本 */ import type { CommandModule } from 'yargs'; import type { DoctorOptions } from '../cli/types.js'; diff --git a/packages/cli/src/commands/headless.ts b/packages/cli/src/commands/headless.ts index 22579f074..9827f3074 100644 --- a/packages/cli/src/commands/headless.ts +++ b/packages/cli/src/commands/headless.ts @@ -71,6 +71,7 @@ import { readCliInput, readOptionalCliInput, } from './shared/commandInput.js'; +import { resolveInputlessResumeState } from './shared/inputlessResume.js'; import { resolveCliOutputSchema } from './shared/outputSchema.js'; import { resolveNonInteractiveSession } from './shared/sessionContext.js'; @@ -404,9 +405,7 @@ function createConfirmationHandler() { }; } -/** - * 从 API 错误中提取用户友好的错误信息 - */ +/** 从 API 错误中提取用户友好的错误信息 */ function extractHeadlessErrorMessage(error: unknown): string { if (!(error instanceof Error)) return 'Unknown error'; @@ -1750,20 +1749,15 @@ export async function runHeadless( } : {}), }); - const pendingInputOnly = inputlessResume && runtime.getPendingSteeringCount() > 0; - const resumedGoal = - inputlessResume && !pendingInputOnly ? await runtime.getGoal() : null; - const goalContinuationOnly = - resumedGoal?.status === 'active' || resumedGoal?.status === 'verifying'; - const startupRecoveryAssessment = runtime.getTurnRecoveryAssessment?.() ?? { - state: 'none' as const, - }; - if ( - inputlessResume && - !pendingInputOnly && - !goalContinuationOnly && - startupRecoveryAssessment.state === 'requires_attention' - ) { + const { + pendingInputOnly, + resumedGoal, + goalContinuationOnly, + finalRecovery, + recoveryAssessment: startupRecoveryAssessment, + recoveredFinalResponse, + } = await resolveInputlessResumeState(runtime, inputlessResume); + if (finalRecovery && startupRecoveryAssessment.state === 'requires_attention') { eventWriter.turnRecovery({ kind: 'turn_recovery', assessment: startupRecoveryAssessment, @@ -1773,11 +1767,7 @@ export async function runHeadless( ); return await finish(2); } - const recoveredFinalResponse = - inputlessResume && !pendingInputOnly && !goalContinuationOnly - ? await runtime.getRecoveredFinalResponse() - : undefined; - if (inputlessResume && !pendingInputOnly && !goalContinuationOnly) { + if (finalRecovery) { if (!recoveredFinalResponse) { throw new Error('No unfinished turn or active goal to resume'); } diff --git a/packages/cli/src/commands/headlessEvents.ts b/packages/cli/src/commands/headlessEvents.ts index 367b85a47..7617bc6ce 100644 --- a/packages/cli/src/commands/headlessEvents.ts +++ b/packages/cli/src/commands/headlessEvents.ts @@ -1,8 +1,7 @@ /** - * Stable JSONL event contract for headless CLI consumers. - * - * The external wire format intentionally uses snake_case so tests and sandbox - * integrations can consume it without depending on internal TypeScript naming. + * Stable JSONL event contract for headless CLI consumers.

The external wire format + * intentionally uses snake_case so tests and sandbox integrations can consume it + * without depending on internal TypeScript naming. */ import { MemoryConsolidationProjectionSchema } from '../api/memoryConsolidationSchemas.js'; diff --git a/packages/cli/src/commands/install.ts b/packages/cli/src/commands/install.ts index dc3c8183e..23afada84 100644 --- a/packages/cli/src/commands/install.ts +++ b/packages/cli/src/commands/install.ts @@ -1,6 +1,4 @@ -/** - * Install 命令 - Yargs 版本 - */ +/** Install 命令 - Yargs 版本 */ import type { CommandModule } from 'yargs'; import type { InstallOptions } from '../cli/types.js'; @@ -41,11 +39,7 @@ export const installCommands: CommandModule<{}, InstallOptions> = { console.log('Installing...'); console.log('Installation completed successfully'); - // 实际实现时可以添加: - // 1. 下载指定版本的二进制文件 - // 2. 验证文件完整性 - // 3. 安装到系统路径 - // 4. 更新符号链接 + // 实际实现时可以添加: 1. 下载指定版本的二进制文件 2. 验证文件完整性 3. 安装到系统路径 4. 更新符号链接 } catch (error) { console.error( `Error: Installation failed: ${error instanceof Error ? error.message : '未知错误'}` diff --git a/packages/cli/src/commands/mcp.ts b/packages/cli/src/commands/mcp.ts index d67c35ae4..70a3867f8 100644 --- a/packages/cli/src/commands/mcp.ts +++ b/packages/cli/src/commands/mcp.ts @@ -1,7 +1,4 @@ -/** - * MCP 命令 - 完整实现 - * 支持: add, remove, list, get, add-json, reset-project-choices - */ +/** MCP 命令 - 完整实现 支持: add, remove, list, get, add-json, reset-project-choices */ import os from 'os'; import path from 'path'; @@ -28,9 +25,7 @@ function asStringArray(value: unknown): string[] | undefined { return out; } -/** - * 显示 MCP 命令的帮助信息 - */ +/** 显示 MCP 命令的帮助信息 */ function showMcpHelp(): void { console.log('\nblade mcp\n'); console.log('管理 MCP 服务器\n'); diff --git a/packages/cli/src/commands/print.ts b/packages/cli/src/commands/print.ts index 2eca9d564..a33458412 100644 --- a/packages/cli/src/commands/print.ts +++ b/packages/cli/src/commands/print.ts @@ -24,6 +24,7 @@ import { readCliInput, readOptionalCliInput, } from './shared/commandInput.js'; +import { resolveInputlessResumeState } from './shared/inputlessResume.js'; import { resolveCliOutputSchema } from './shared/outputSchema.js'; import { resolveNonInteractiveSession } from './shared/sessionContext.js'; @@ -256,20 +257,14 @@ export async function runPrint( } : {}), }); - const pendingInputOnly = inputlessResume && runtime.getPendingSteeringCount() > 0; - const resumedGoal = - inputlessResume && !pendingInputOnly ? await runtime.getGoal() : null; - const goalContinuationOnly = - resumedGoal?.status === 'active' || resumedGoal?.status === 'verifying'; - const startupRecoveryAssessment = runtime.getTurnRecoveryAssessment?.() ?? { - state: 'none' as const, - }; - if ( - inputlessResume && - !pendingInputOnly && - !goalContinuationOnly && - startupRecoveryAssessment.state === 'requires_attention' - ) { + const { + pendingInputOnly, + goalContinuationOnly, + finalRecovery, + recoveryAssessment: startupRecoveryAssessment, + recoveredFinalResponse, + } = await resolveInputlessResumeState(runtime, inputlessResume); + if (finalRecovery && startupRecoveryAssessment.state === 'requires_attention') { io.stderr.write( `[turn-recovery:${startupRecoveryAssessment.state}] ${startupRecoveryAssessment.turnId}\n` ); @@ -277,11 +272,7 @@ export async function runPrint( 'Turn recovery requires explicit user attention before continuation' ); } - const recoveredFinalResponse = - inputlessResume && !pendingInputOnly && !goalContinuationOnly - ? await runtime.getRecoveredFinalResponse() - : undefined; - if (inputlessResume && !pendingInputOnly && !goalContinuationOnly) { + if (finalRecovery) { if (!recoveredFinalResponse) { throw new Error('No unfinished turn or active goal to resume'); } @@ -388,10 +379,7 @@ export async function runPrint( } } -/** - * 检查命令行参数是否包含 --print 选项 - * 如果包含,则以 print 模式运行 - */ +/** 检查命令行参数是否包含 --print 选项 如果包含,则以 print 模式运行 */ export async function handlePrintMode(): Promise { const argv = process.argv.slice(2); const printIndex = argv.findIndex((arg) => arg === '--print' || arg === '-p'); diff --git a/packages/cli/src/commands/shared/inputlessResume.ts b/packages/cli/src/commands/shared/inputlessResume.ts new file mode 100644 index 000000000..303b80c99 --- /dev/null +++ b/packages/cli/src/commands/shared/inputlessResume.ts @@ -0,0 +1,43 @@ +import type { + RecoveredFinalResponse, + SessionRuntime, +} from '../../agent/runtime/SessionRuntime.js'; +import type { SessionTurnRecoveryAssessment } from '../../context/turnRecoveryAssessment.js'; +import type { GoalSnapshot } from '../../goals/types.js'; + +export interface InputlessResumeState { + pendingInputOnly: boolean; + resumedGoal: GoalSnapshot | null; + goalContinuationOnly: boolean; + finalRecovery: boolean; + recoveryAssessment: SessionTurnRecoveryAssessment; + recoveredFinalResponse?: RecoveredFinalResponse; +} + +export async function resolveInputlessResumeState( + runtime: SessionRuntime, + inputlessResume: boolean +): Promise { + const pendingInputOnly = inputlessResume && runtime.getPendingSteeringCount() > 0; + const resumedGoal = + inputlessResume && !pendingInputOnly ? await runtime.getGoal() : null; + const goalContinuationOnly = + resumedGoal?.status === 'active' || resumedGoal?.status === 'verifying'; + const finalRecovery = inputlessResume && !pendingInputOnly && !goalContinuationOnly; + const recoveryAssessment = runtime.getTurnRecoveryAssessment?.() ?? { + state: 'none' as const, + }; + const recoveredFinalResponse = + finalRecovery && recoveryAssessment.state !== 'requires_attention' + ? await runtime.getRecoveredFinalResponse() + : undefined; + + return { + pendingInputOnly, + resumedGoal, + goalContinuationOnly, + finalRecovery, + recoveryAssessment, + recoveredFinalResponse, + }; +} diff --git a/packages/cli/src/commands/update.ts b/packages/cli/src/commands/update.ts index 20ea63b4f..7fba204b1 100644 --- a/packages/cli/src/commands/update.ts +++ b/packages/cli/src/commands/update.ts @@ -1,6 +1,4 @@ -/** - * Update 命令 - Yargs 版本 - */ +/** Update 命令 - Yargs 版本 */ import { execSync } from 'child_process'; import type { CommandModule } from 'yargs'; diff --git a/packages/cli/src/config/ConfigManager.ts b/packages/cli/src/config/ConfigManager.ts index 6c05d81b2..4a6665dc4 100644 --- a/packages/cli/src/config/ConfigManager.ts +++ b/packages/cli/src/config/ConfigManager.ts @@ -142,14 +142,10 @@ export class ConfigManager { private lastAdditionalSettings?: Partial; private warnedGlobalOnlyProjectionResidencyFiles = new Set(); - /** - * 私有构造函数,防止外部直接实例化 - */ + /** 私有构造函数,防止外部直接实例化 */ private constructor() {} - /** - * 获取 ConfigManager 单例实例 - */ + /** 获取 ConfigManager 单例实例 */ public static getInstance(): ConfigManager { if (!ConfigManager.instance) { ConfigManager.instance = new ConfigManager(); @@ -157,23 +153,17 @@ export class ConfigManager { return ConfigManager.instance; } - /** - * 重置单例实例(仅用于测试) - */ + /** 重置单例实例(仅用于测试) */ public static resetInstance(): void { ConfigManager.instance = null; } /** - * 初始化配置系统(Bootstrap/Loader) - * - * 职责: - * - 从多文件加载配置(config.json + settings.json) - * - 合并配置(优先级处理) - * - 解析环境变量插值 - * - 返回完整的 BladeConfig - * - * 注意:不保存状态,调用方需要将结果灌进 Store + + * 初始化配置系统(Bootstrap/Loader)

职责: - 从多文件加载配置(config.json + settings.json) - + + * 合并配置(优先级处理) - 解析环境变量插值 - 返回完整的 BladeConfig

注意:不保存状态,调用方需要将结果灌进 Store + */ async initialize( additionalSettings?: Partial @@ -366,10 +356,7 @@ export class ConfigManager { return migrated; } - /** - * 加载 settings.json 文件 (3层优先级) - * 优先级: 本地配置 > 项目配置 > 用户配置 - */ + /** 加载 settings.json 文件 (3层优先级) 优先级: 本地配置 > 项目配置 > 用户配置 */ private async loadSettingsFiles( projectTrusted: boolean ): Promise> { @@ -455,10 +442,7 @@ export class ConfigManager { ); } - /** - * 为独立 runtime 合并指定 workspace 的项目与本地权限。 - * 用户级规则已经包含在 base 中,这里只叠加 workspace 私有层。 - */ + /** 为独立 runtime 合并指定 workspace 的项目与本地权限。 用户级规则已经包含在 base 中,这里只叠加 workspace 私有层。 */ async loadWorkspacePermissions( workspaceRoot: string, base: PermissionConfig @@ -1178,9 +1162,7 @@ export class ConfigManager { } } - /** - * 加载 JSON 文件 - */ + /** 加载 JSON 文件 */ private async loadJsonFile(filePath: string): Promise | null> { try { if (await this.fileExists(filePath)) { @@ -1193,9 +1175,7 @@ export class ConfigManager { return null; } - /** - * 检查文件是否存在 - */ + /** 检查文件是否存在 */ private async fileExists(filePath: string): Promise { try { await fs.access(filePath); @@ -1236,9 +1216,7 @@ export class ConfigManager { return result as Partial; } - /** - * 验证 BladeConfig 是否包含 Agent 所需的必要字段 - */ + /** 验证 BladeConfig 是否包含 Agent 所需的必要字段 */ public validateConfig( config: BladeConfig, catalog: PiModelCatalog = getPiModelCatalog() diff --git a/packages/cli/src/config/ConfigService.ts b/packages/cli/src/config/ConfigService.ts index 856d44975..d258f8c48 100644 --- a/packages/cli/src/config/ConfigService.ts +++ b/packages/cli/src/config/ConfigService.ts @@ -1,13 +1,7 @@ /** - * ConfigService - 配置持久化路由层 - * - * 职责: - * 1. 字段路由(config.json vs settings.json) - * 2. scope 路由(local/project/global) - * 3. 临时字段过滤 - * 4. 防抖(300ms)+ 立即持久化 - * 5. 并发写入安全(Per-file Mutex + Read-Modify-Write) - * 6. 向前兼容(保留未知字段) + * ConfigService - 配置持久化路由层

职责: 1. 字段路由(config.json vs settings.json) 2. scope + * 路由(local/project/global) 3. 临时字段过滤 4. 防抖(300ms)+ 立即持久化 5. 并发写入安全(Per-file Mutex + + * Read-Modify-Write) 6. 向前兼容(保留未知字段) */ import { promises as fs } from 'node:fs'; @@ -50,423 +44,132 @@ interface FieldRouting { // FIELD_ROUTING_TABLE - 单一真相源 // ============================================ -/** - * 字段路由表:定义每个配置字段的持久化行为 - * - * 所有其他常量(PERSISTABLE_FIELDS、NON_PERSISTABLE_FIELDS 等)从此表自动派生 - */ +/** 字段路由表:定义每个配置字段的持久化行为 所有其他常量(PERSISTABLE_FIELDS、NON_PERSISTABLE_FIELDS 等)从此表自动派生 */ +const routeFields = ( + fields: readonly string[], + routing: FieldRouting +): Record => + Object.fromEntries(fields.map((field) => [field, routing])); + const FIELD_ROUTING_TABLE: Record = { - // ===== config.json 字段(基础配置)===== - models: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', // 完全替换数组 - persistable: true, - }, - modelProviders: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - currentModelId: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - temperature: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - maxOutputTokens: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - timeout: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - bashForegroundHandoffMs: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - providerForegroundRecoveryMs: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - providerCircuitBreakerOpenMs: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - providerRequestConcurrency: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - providerGlobalConcurrency: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - providerOwnerConcurrency: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - providerRequestAdmissionMs: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - providerRequestPendingBytes: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - agentTeamsEnabled: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - codeTheme: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - uiTheme: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - language: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - debug: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - fontSize: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - autoSaveSessions: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - notifyBuild: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - notifyErrors: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - notifySounds: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - privacyTelemetry: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - privacyCrash: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - communicationStyle: { + ...routeFields( + [ + 'models', + 'modelProviders', + 'currentModelId', + 'temperature', + 'maxOutputTokens', + 'timeout', + 'bashForegroundHandoffMs', + 'providerForegroundRecoveryMs', + 'providerCircuitBreakerOpenMs', + 'providerRequestConcurrency', + 'providerGlobalConcurrency', + 'providerOwnerConcurrency', + 'providerRequestAdmissionMs', + 'providerRequestPendingBytes', + 'agentTeamsEnabled', + 'codeTheme', + 'uiTheme', + 'language', + 'debug', + 'fontSize', + 'autoSaveSessions', + 'notifyBuild', + 'notifyErrors', + 'notifySounds', + 'privacyTelemetry', + 'privacyCrash', + 'communicationStyle', + 'mcpServers', + 'lspServers', + ], + { + target: 'config', + defaultScope: 'global', + mergeStrategy: 'replace', + persistable: true, + } + ), + ...routeFields(['stream', 'topP', 'topK', 'mcpEnabled'], { target: 'config', defaultScope: 'global', mergeStrategy: 'replace', - persistable: true, - }, - - // ===== settings.json 字段(行为配置)===== - permissionMode: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, // 运行时状态,不持久化 - }, - permissions: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', // 完全替换(允许删除规则) - persistable: true, - }, - hooks: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'deep-merge', // 深度合并对象 - persistable: true, - }, - enabledPlugins: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'deep-merge', - persistable: true, - }, - pluginSourcePolicy: { - target: 'settings', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - env: { + persistable: false, + }), + ...routeFields( + [ + 'permissions', + 'disableAllHooks', + 'maxTurns', + 'maxConcurrentTasks', + 'maxQueuedTasks', + 'maxQueuedTaskBytes', + ], + { + target: 'settings', + defaultScope: 'local', + mergeStrategy: 'replace', + persistable: true, + } + ), + ...routeFields(['hooks', 'enabledPlugins', 'env'], { target: 'settings', defaultScope: 'local', mergeStrategy: 'deep-merge', persistable: true, - }, - disableAllHooks: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: true, - }, - maxTurns: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: true, - }, - maxConcurrentTasks: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: true, - }, - maxQueuedTasks: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: true, - }, - maxQueuedTaskBytes: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: true, - }, - maxResidentSessionRuntimes: { - target: 'settings', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - sessionRuntimeIdleMs: { - target: 'settings', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - maxResidentSessionProjections: { - target: 'settings', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - allowedScopes: ['global'], - }, - sessionProjectionIdleMs: { + }), + ...routeFields( + ['pluginSourcePolicy', 'maxResidentSessionRuntimes', 'sessionRuntimeIdleMs'], + { + target: 'settings', + defaultScope: 'global', + mergeStrategy: 'replace', + persistable: true, + } + ), + ...routeFields(['maxResidentSessionProjections', 'sessionProjectionIdleMs'], { target: 'settings', defaultScope: 'global', mergeStrategy: 'replace', persistable: true, allowedScopes: ['global'], - }, - mcpServers: { - target: 'config', - defaultScope: 'global', // MCP 服务器配置存储在用户全局配置中 - mergeStrategy: 'replace', - persistable: true, - }, - lspServers: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: true, - }, - - // ===== 非持久化字段(在 BladeConfig 中但不保存到磁盘)===== - // 这些字段在 BladeConfig 中定义,但默认不持久化 - stream: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: false, - }, - topP: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: false, - }, - topK: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: false, - }, - mcpEnabled: { - target: 'config', - defaultScope: 'global', - mergeStrategy: 'replace', - persistable: false, - }, - - // ===== CLI 临时字段(绝不持久化)===== - systemPrompt: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - appendSystemPrompt: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - initialMessage: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - resumeSessionId: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - forkSession: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - allowedTools: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - disallowedTools: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - mcpConfigPaths: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - strictMcpConfig: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - - addDirs: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - outputFormat: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - inputFormat: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - print: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - includePartialMessages: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - replayUserMessages: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - agentsConfig: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, - settingSources: { - target: 'settings', - defaultScope: 'local', - mergeStrategy: 'replace', - persistable: false, - }, + }), + ...routeFields( + [ + 'permissionMode', + 'systemPrompt', + 'appendSystemPrompt', + 'initialMessage', + 'resumeSessionId', + 'forkSession', + 'allowedTools', + 'disallowedTools', + 'mcpConfigPaths', + 'strictMcpConfig', + 'addDirs', + 'outputFormat', + 'inputFormat', + 'print', + 'includePartialMessages', + 'replayUserMessages', + 'agentsConfig', + 'settingSources', + ], + { + target: 'settings', + defaultScope: 'local', + mergeStrategy: 'replace', + persistable: false, + } + ), }; // ============================================ // 派生常量(从 FIELD_ROUTING_TABLE 自动生成) // ============================================ -/** - * 可持久化的字段集合 - * 从 FIELD_ROUTING_TABLE 中提取 persistable: true 的字段 - */ +/** 可持久化的字段集合 从 FIELD_ROUTING_TABLE 中提取 persistable: true 的字段 */ const _PERSISTABLE_FIELDS = new Set( Object.entries(FIELD_ROUTING_TABLE) .filter(([_, routing]) => routing.persistable) @@ -474,15 +177,13 @@ const _PERSISTABLE_FIELDS = new Set( ); /** - * 不可持久化的字段集合(包含两类) - * - * 1. **BladeConfig 永久字段但选择不持久化**: - * - stream, topP, topK, fontSize, mcpEnabled - * - 在类型定义中,但不写入文件(用户不希望或不需要持久化) - * - * 2. **CLI 运行时临时参数**: - * - systemPrompt, initialMessage, resumeSessionId, forkSession 等 - * - 仅存在于 CLI 启动期间,从不持久化 + + * 不可持久化的字段集合(包含两类)

1. **BladeConfig 永久字段但选择不持久化**: - stream, topP, topK, fontSize, + + * mcpEnabled - 在类型定义中,但不写入文件(用户不希望或不需要持久化)

2. **CLI 运行时临时参数**: - systemPrompt, + + * initialMessage, resumeSessionId, forkSession 等 - 仅存在于 CLI 启动期间,从不持久化 + */ const _NON_PERSISTABLE_FIELDS = new Set( Object.entries(FIELD_ROUTING_TABLE) @@ -531,17 +232,12 @@ export class ConfigService { // Per-file operations retain coordination only while active or queued. private readonly fileLocks = new KeyedMutexRegistry(); - // 错误记录 - private lastSaveError: Error | null = null; - // 防抖延迟(毫秒) private readonly debounceDelay = 300; private constructor() {} - /** - * 获取单例实例 - */ + /** 获取单例实例 */ public static getInstance(): ConfigService { if (!ConfigService.instance) { ConfigService.instance = new ConfigService(); @@ -549,9 +245,7 @@ export class ConfigService { return ConfigService.instance; } - /** - * 重置实例(仅用于测试) - */ + /** 重置实例(仅用于测试) */ public static resetInstance(): void { if (ConfigService.instance) { // 清理所有定时器 @@ -605,9 +299,7 @@ export class ConfigService { } } - /** - * 立即刷新所有待持久化变更 - */ + /** 立即刷新所有待持久化变更 */ async flush(): Promise { // 取消所有待处理的定时器 for (const timer of this.timers.values()) { @@ -624,22 +316,6 @@ export class ConfigService { await Promise.all(promises); } - /** - * 获取最后一次保存错误 - * - * @returns 最后一次保存失败的错误,如果没有错误则返回 null - */ - getLastSaveError(): Error | null { - return this.lastSaveError; - } - - /** - * 清除最后一次保存错误 - */ - clearLastSaveError(): void { - this.lastSaveError = null; - } - /** * 追加权限规则(手动实现 append-dedupe 策略) * 默认 scope 为 'local',与 FIELD_ROUTING_TABLE.permissions.defaultScope 一致 @@ -650,9 +326,7 @@ export class ConfigService { await this.appendPermissionRuleForDecision(rule, 'allow', options); } - /** - * 追加权限拒绝规则(手动实现 append-dedupe 策略) - */ + /** 追加权限拒绝规则(手动实现 append-dedupe 策略) */ async appendPermissionDenyRule( rule: string, options: SaveOptions = {} @@ -697,9 +371,7 @@ export class ConfigService { }); } - /** - * 追加本地权限规则(强制 local scope) - */ + /** 追加本地权限规则(强制 local scope) */ async appendLocalPermissionRule( rule: string, options: Omit = {} @@ -707,9 +379,7 @@ export class ConfigService { await this.appendPermissionRule(rule, { ...options, scope: 'local' }); } - /** - * 追加本地权限拒绝规则(强制 local scope) - */ + /** 追加本地权限拒绝规则(强制 local scope) */ async appendLocalPermissionDenyRule( rule: string, options: Omit = {} @@ -740,9 +410,7 @@ export class ConfigService { // 私有方法 // ============================================ - /** - * 验证字段是否可持久化 - */ + /** 验证字段是否可持久化 */ private validatePersistableFields(updates: Partial): void { for (const key of Object.keys(updates)) { const routing = FIELD_ROUTING_TABLE[key]; @@ -809,9 +477,7 @@ export class ConfigService { } } - /** - * 按 target 和 scope 分组更新 - */ + /** 按 target 和 scope 分组更新 */ private groupUpdatesByTarget( updates: Partial, scopeOverride?: ConfigScope, @@ -835,9 +501,7 @@ export class ConfigService { return grouped; } - /** - * 解析文件路径 - */ + /** 解析文件路径 */ private resolveFilePath( target: ConfigTarget, scope: ConfigScope, @@ -863,9 +527,7 @@ export class ConfigService { } } - /** - * 调度防抖保存 - */ + /** 调度防抖保存 */ private scheduleSave(filePath: string, updates: Record): void { // 获取现有的待处理更新 const existing = this.pendingUpdates.get(filePath) ?? {}; @@ -889,12 +551,8 @@ export class ConfigService { try { await this.flushTarget(filePath, pendingUpdates); - // 成功后清除错误记录 - this.lastSaveError = null; } catch (error) { - // 记录错误 const saveError = error instanceof Error ? error : new Error(String(error)); - this.lastSaveError = saveError; // 记录日志(避免静默失败) logger.error(`Failed to save config to ${filePath}:`, saveError.message); @@ -908,11 +566,7 @@ export class ConfigService { this.timers.set(filePath, timer); } - /** - * 合并待处理更新(按字段合并策略) - * - * 用于防抖场景:300ms 内多次 save 调用需要正确合并,避免深层字段被覆盖 - */ + /** 合并待处理更新(按字段合并策略) 用于防抖场景:300ms 内多次 save 调用需要正确合并,避免深层字段被覆盖 */ private mergePendingUpdates( existing: Record, updates: Record @@ -943,9 +597,7 @@ export class ConfigService { return result; } - /** - * 刷新目标文件(带 Per-file Mutex) - */ + /** 刷新目标文件(带 Per-file Mutex) */ private async flushTarget( filePath: string, updates: Record @@ -988,9 +640,7 @@ export class ConfigService { }); } - /** - * 执行写入操作(Read-Modify-Write) - */ + /** 执行写入操作(Read-Modify-Write) */ private async performWrite( filePath: string, updates: Record @@ -1045,9 +695,7 @@ export class ConfigService { await this.atomicWrite(filePath, mergedConfig); } - /** - * 应用 append-dedupe 合并策略 - */ + /** 应用 append-dedupe 合并策略 */ private applyAppendDedupe( config: Record, key: string, @@ -1084,9 +732,7 @@ export class ConfigService { } } - /** - * 应用 deep-merge 合并策略(使用 lodash-es merge) - */ + /** 应用 deep-merge 合并策略(使用 lodash-es merge) */ private applyDeepMerge( config: Record, key: string, @@ -1102,16 +748,12 @@ export class ConfigService { } } - /** - * 数组去重 - */ + /** 数组去重 */ private dedupeArray(arr: T[]): T[] { return Array.from(new Set(arr)); } - /** - * 原子写入(使用 write-file-atomic) - */ + /** 原子写入(使用 write-file-atomic) */ private async atomicWrite( filePath: string, data: Record diff --git a/packages/cli/src/config/PermissionChecker.ts b/packages/cli/src/config/PermissionChecker.ts index 604cc0248..93a70e8f6 100644 --- a/packages/cli/src/config/PermissionChecker.ts +++ b/packages/cli/src/config/PermissionChecker.ts @@ -1,8 +1,4 @@ -/** - * Blade 权限检查器 - * 实现 allow/ask/deny 三级权限控制 - * 支持精确匹配、前缀匹配、通配符匹配和 glob 模式 - */ +/** Blade 权限检查器 实现 allow/ask/deny 三级权限控制 支持精确匹配、前缀匹配、通配符匹配和 glob 模式 */ import picomatch from 'picomatch'; import { @@ -12,9 +8,7 @@ import { } from '../utils/shell/commandNormalizer.js'; import type { PermissionConfig } from './types.js'; -/** - * 权限检查结果 - */ +/** 权限检查结果 */ export enum PermissionResult { /** 允许执行 */ ALLOW = 'allow', @@ -24,9 +18,7 @@ export enum PermissionResult { DENY = 'deny', } -/** - * 权限检查详情 - */ +/** 权限检查详情 */ export interface PermissionCheckResult { /** 检查结果 */ result: PermissionResult; @@ -38,9 +30,7 @@ export interface PermissionCheckResult { reason?: string; } -/** - * 工具调用描述 - */ +/** 工具调用描述 */ export interface ToolInvocationDescriptor { /** 工具名称 */ toolName: string; @@ -55,15 +45,11 @@ export interface ToolInvocationDescriptor { }; } -/** - * 权限检查器 - */ +/** 权限检查器 */ export class PermissionChecker { constructor(private config: PermissionConfig) {} - /** - * 检查工具调用权限 - */ + /** 检查工具调用权限 */ check(descriptor: ToolInvocationDescriptor): PermissionCheckResult { const signature = PermissionChecker.buildSignature(descriptor); @@ -208,9 +194,7 @@ export class PermissionChecker { } } - /** - * 匹配规则列表 - */ + /** 匹配规则列表 */ private matchRules( signature: string, rules: string[] @@ -281,8 +265,7 @@ export class PermissionChecker { return 'wildcard'; } - // Bash 的抽象规则以 "固定命令前缀 *" 表示同一命令族。 - // 固定部分可能包含反斜杠或其他 glob 元字符,必须按 shell 文本字面量匹配。 + // Bash 的抽象规则以 "固定命令前缀 *" 表示同一命令族。 固定部分可能包含反斜杠或其他 glob 元字符,必须按 shell 文本字面量匹配。 if (sigToolName === 'Bash' && ruleParams.endsWith(' *')) { return sigParams.startsWith(ruleParams.slice(0, -1)) ? 'wildcard' : null; } @@ -302,19 +285,13 @@ export class PermissionChecker { return null; } - /** - * 提取参数部分 - * 注意:使用 [\s\S] 而不是 . 以匹配包含换行符的多行参数 - */ + /** 提取参数部分 注意:使用 [\s\S] 而不是 . 以匹配包含换行符的多行参数 */ private extractParams(signature: string): string { const match = signature.match(/\(([\s\S]*)\)$/); return match ? match[1] : ''; } - /** - * 匹配参数 - * 支持对参数值进行 glob 匹配 - */ + /** 匹配参数 支持对参数值进行 glob 匹配 */ private matchParams(sigParams: string, ruleParams: string): boolean { if (!sigParams || !ruleParams) { return false; @@ -340,8 +317,7 @@ export class PermissionChecker { return true; } - // Glob 模式匹配 - // 使用 bash: true 让 * 匹配包括 / 在内的所有字符 + // Glob 模式匹配 使用 bash: true 让 * 匹配包括 / 在内的所有字符 if ( ruleParams.includes('*') || ruleParams.includes('{') || @@ -374,8 +350,7 @@ export class PermissionChecker { continue; } - // 否则使用 picomatch 进行 glob 匹配 - // 使用 bash: true 让 * 匹配包括 / 和空格在内的所有字符 + // 否则使用 picomatch 进行 glob 匹配 使用 bash: true 让 * 匹配包括 / 和空格在内的所有字符 const isMatch = picomatch.isMatch(sigValue, ruleValue, { dot: true, bash: true, @@ -392,10 +367,7 @@ export class PermissionChecker { return true; } - /** - * 解析参数对 - * 支持嵌套的花括号和括号 - */ + /** 解析参数对 支持嵌套的花括号和括号 */ private parseParamPairs(params: string): Record { const pairs: Record = {}; @@ -414,9 +386,7 @@ export class PermissionChecker { return pairs; } - /** - * 智能分割字符串,忽略括号和花括号内的分隔符 - */ + /** 智能分割字符串,忽略括号和花括号内的分隔符 */ private smartSplit(str: string, delimiter: string): string[] { const result: string[] = []; let current = ''; @@ -534,38 +504,13 @@ export class PermissionChecker { return -1; } - /** - * 从签名中提取工具名称 - */ + /** 从签名中提取工具名称 */ private extractToolName(signature: string): string | null { const match = signature.match(/^([A-Za-z0-9_]+)(\(|$)/); return match ? match[1] : null; } - /** - * 检查是否允许执行 (不需要确认) - */ - isAllowed(descriptor: ToolInvocationDescriptor): boolean { - return this.check(descriptor).result === PermissionResult.ALLOW; - } - - /** - * 检查是否被拒绝 - */ - isDenied(descriptor: ToolInvocationDescriptor): boolean { - return this.check(descriptor).result === PermissionResult.DENY; - } - - /** - * 检查是否需要确认 - */ - needsConfirmation(descriptor: ToolInvocationDescriptor): boolean { - return this.check(descriptor).result === PermissionResult.ASK; - } - - /** - * 更新权限配置 - */ + /** 更新权限配置 */ updateConfig(config: Partial): void { if (config.allow) { this.config.allow = [...this.config.allow, ...config.allow]; @@ -577,18 +522,4 @@ export class PermissionChecker { this.config.deny = [...this.config.deny, ...config.deny]; } } - - /** - * 替换整个权限配置 - */ - replaceConfig(config: PermissionConfig): void { - this.config = { ...config }; - } - - /** - * 获取当前权限配置 - */ - getConfig(): PermissionConfig { - return { ...this.config }; - } } diff --git a/packages/cli/src/config/defaultPermissions.json b/packages/cli/src/config/defaultPermissions.json new file mode 100644 index 000000000..95d820b63 --- /dev/null +++ b/packages/cli/src/config/defaultPermissions.json @@ -0,0 +1,100 @@ +{ + "allow": [ + "Bash(pwd)", + "Bash(which *)", + "Bash(whoami)", + "Bash(hostname)", + "Bash(uname *)", + "Bash(date)", + "Bash(echo *)", + "Bash(ls *)", + "Bash(tree *)", + "Bash(git status)", + "Bash(git status -*)", + "Bash(git log *)", + "Bash(git diff *)", + "Bash(git show *)", + "Bash(git branch -* *)", + "Bash(git branch)", + "Bash(git tag -l *)", + "Bash(git tag --list *)", + "Bash(git stash list *)", + "Bash(git stash show *)", + "Bash(git rev-parse *)", + "Bash(git describe *)", + "Bash(git blame *)", + "Bash(git ls-files *)", + "Bash(git config --get *)", + "Bash(git config --list *)", + "Bash(git shortlog *)", + "Bash(git merge-base *)", + "Bash(git cat-file *)", + "Bash(git for-each-ref *)", + "Bash(git grep *)", + "Bash(git worktree list *)", + "Bash(git reflog show *)", + "Bash(git reflog)", + "Bash(git rev-list *)", + "Bash(git ls-remote *)", + "Bash(git remote -v)", + "Bash(git remote --verbose)", + "Bash(git remote)", + "Bash(gh pr view *)", + "Bash(gh pr list *)", + "Bash(gh pr diff *)", + "Bash(gh pr checks *)", + "Bash(gh pr status *)", + "Bash(gh issue view *)", + "Bash(gh issue list *)", + "Bash(gh issue status *)", + "Bash(gh run list *)", + "Bash(gh run view *)", + "Bash(gh repo view *)", + "Bash(gh auth status)", + "Bash(npm list *)", + "Bash(bun pm ls *)", + "Bash(npm view *)", + "Bash(npm outdated *)", + "Bash(pnpm list *)", + "Bash(yarn list *)", + "Bash(pip list *)", + "Bash(pip show *)" + ], + "ask": [ + "Bash(curl *)", + "Bash(wget *)", + "Bash(aria2c *)", + "Bash(axel *)", + "Bash(rm -rf *)", + "Bash(rm -r *)", + "Bash(rm --recursive *)", + "Bash(nc *)", + "Bash(netcat *)", + "Bash(telnet *)", + "Bash(ncat *)" + ], + "deny": [ + "Read(./.env)", + "Read(./.env.*)", + "Bash(rm -rf /)", + "Bash(rm -rf /*)", + "Bash(sudo *)", + "Bash(chmod 777 *)", + "Bash(bash *)", + "Bash(sh *)", + "Bash(zsh *)", + "Bash(fish *)", + "Bash(dash *)", + "Bash(eval *)", + "Bash(source *)", + "Bash(mkfs *)", + "Bash(fdisk *)", + "Bash(dd *)", + "Bash(format *)", + "Bash(parted *)", + "Bash(open http*)", + "Bash(open https*)", + "Bash(xdg-open http*)", + "Bash(xdg-open https*)" + ] +} diff --git a/packages/cli/src/config/defaults.ts b/packages/cli/src/config/defaults.ts index fe3c2e657..1e078d090 100644 --- a/packages/cli/src/config/defaults.ts +++ b/packages/cli/src/config/defaults.ts @@ -1,7 +1,6 @@ -/** - * Blade 默认配置 - */ +/** Blade 默认配置 */ +import defaultPermissions from './defaultPermissions.json'; import { DEFAULT_FOREGROUND_COMMAND_HANDOFF_MS } from './foregroundCommandHandoff.js'; import { DEFAULT_FOREGROUND_PROVIDER_RECOVERY_MS } from './foregroundProviderRecovery.js'; import { DEFAULT_PROVIDER_CIRCUIT_OPEN_MS } from './providerCircuitBreaker.js'; @@ -72,151 +71,7 @@ export const DEFAULT_CONFIG: BladeConfig = { // 行为配置 (settings.json) // ===================================== - // 权限 - permissions: { - allow: [ - // 安全的系统信息命令(无需确认) - 'Bash(pwd)', - 'Bash(which *)', - 'Bash(whoami)', - 'Bash(hostname)', - 'Bash(uname *)', - 'Bash(date)', - 'Bash(echo *)', - - // 目录列表(推荐使用 Glob 工具,但允许 ls 作为降级) - 'Bash(ls *)', - 'Bash(tree *)', - - // Git 只读命令(无需确认) - // 注意:静态 allow 规则对原始命令串做 glob,不按 shell 语义拆分。 - // 因此只能放行简单前缀(command + space + *), - // 复杂场景(env vars, -C, compound commands)由语义层兜底。 - 'Bash(git status)', - 'Bash(git status -*)', - 'Bash(git log *)', - 'Bash(git diff *)', - 'Bash(git show *)', - 'Bash(git branch -* *)', - 'Bash(git branch)', - 'Bash(git tag -l *)', - 'Bash(git tag --list *)', - 'Bash(git stash list *)', - 'Bash(git stash show *)', - 'Bash(git rev-parse *)', - 'Bash(git describe *)', - 'Bash(git blame *)', - 'Bash(git ls-files *)', - 'Bash(git config --get *)', - 'Bash(git config --list *)', - 'Bash(git shortlog *)', - 'Bash(git merge-base *)', - 'Bash(git cat-file *)', - 'Bash(git for-each-ref *)', - 'Bash(git grep *)', - 'Bash(git worktree list *)', - 'Bash(git reflog show *)', - 'Bash(git reflog)', - 'Bash(git rev-list *)', - 'Bash(git ls-remote *)', - 'Bash(git remote -v)', - 'Bash(git remote --verbose)', - 'Bash(git remote)', - - // gh CLI 只读命令(无需确认) - 'Bash(gh pr view *)', - 'Bash(gh pr list *)', - 'Bash(gh pr diff *)', - 'Bash(gh pr checks *)', - 'Bash(gh pr status *)', - 'Bash(gh issue view *)', - 'Bash(gh issue list *)', - 'Bash(gh issue status *)', - 'Bash(gh run list *)', - 'Bash(gh run view *)', - 'Bash(gh repo view *)', - // gh auth status 不带 * — --show-token/-t 会泄露凭据 - 'Bash(gh auth status)', - - // 包管理器只读命令(无需确认) - 'Bash(npm list *)', - 'Bash(bun pm ls *)', - 'Bash(npm view *)', - 'Bash(npm outdated *)', - 'Bash(pnpm list *)', - 'Bash(yarn list *)', - 'Bash(pip list *)', - 'Bash(pip show *)', - - // 注意:以下命令已从 allow 列表移除,因为有专用工具: - // - cat/head/tail -> 使用 Read 工具 - // - grep -> 使用 Grep 工具 - // - find -> 使用 Glob 工具 - // LLM 调用这些命令时会触发权限确认,提示使用专用工具 - - // 常见的构建/测试命令(默认需要确认) - // 用户可以在本地配置中添加到 allow 列表以信任特定项目 - // 'Bash(npm install *)', - // 'Bash(npm test *)', - // 'Bash(npm run build *)', - // 'Bash(npm run lint *)', - ], - ask: [ - // 高风险命令(需要用户确认) - - // 网络下载工具(可能下载并执行恶意代码) - 'Bash(curl *)', - 'Bash(wget *)', - 'Bash(aria2c *)', - 'Bash(axel *)', - - // 危险删除操作 - 'Bash(rm -rf *)', - 'Bash(rm -r *)', - 'Bash(rm --recursive *)', - - // 网络连接工具 - 'Bash(nc *)', - 'Bash(netcat *)', - 'Bash(telnet *)', - 'Bash(ncat *)', - ], - deny: [ - // 敏感文件读取 - 'Read(./.env)', - 'Read(./.env.*)', - - // 危险命令(明确拒绝) - 'Bash(rm -rf /)', - 'Bash(rm -rf /*)', - 'Bash(sudo *)', - 'Bash(chmod 777 *)', - - // Shell 嵌套(可绕过安全检测) - 'Bash(bash *)', - 'Bash(sh *)', - 'Bash(zsh *)', - 'Bash(fish *)', - 'Bash(dash *)', - - // 代码注入风险 - 'Bash(eval *)', - 'Bash(source *)', - - // 危险系统操作 - 'Bash(mkfs *)', - 'Bash(fdisk *)', - 'Bash(dd *)', - 'Bash(format *)', - 'Bash(parted *)', - - // 浏览器(可打开恶意链接) - 'Bash(open http*)', - 'Bash(open https*)', - 'Bash(xdg-open http*)', - 'Bash(xdg-open https*)', - ], - }, + permissions: defaultPermissions, permissionMode: PermissionMode.DEFAULT, // Hooks (默认禁用) diff --git a/packages/cli/src/config/index.ts b/packages/cli/src/config/index.ts index 122b3b430..993860aca 100644 --- a/packages/cli/src/config/index.ts +++ b/packages/cli/src/config/index.ts @@ -1,7 +1,4 @@ -/** - * Blade 配置系统 - * 双配置文件系统: config.json (基础配置) + settings.json (行为配置) - */ +/** Blade 配置系统 双配置文件系统: config.json (基础配置) + settings.json (行为配置) */ // 配置管理器 export { ConfigManager, mergeRuntimeConfig } from './ConfigManager.js'; diff --git a/packages/cli/src/config/sessionProjectionResidency.ts b/packages/cli/src/config/sessionProjectionResidency.ts index 0a41cb88c..61f6e1d5b 100644 --- a/packages/cli/src/config/sessionProjectionResidency.ts +++ b/packages/cli/src/config/sessionProjectionResidency.ts @@ -8,7 +8,6 @@ export const MAX_SESSION_PROJECTION_IDLE_MS = 24 * 60 * 60_000; export const SESSION_PROJECTION_SWEEP_MS = 30_000; export const SESSION_PROJECTION_DRAIN_MS = 30_000; -export const MAX_SESSION_PROJECTION_WAKE_ENTRIES = 256; export function isValidResidentSessionProjectionLimit(value: number): boolean { return ( diff --git a/packages/cli/src/config/types.ts b/packages/cli/src/config/types.ts index a3d7a0961..5f2d90e9f 100644 --- a/packages/cli/src/config/types.ts +++ b/packages/cli/src/config/types.ts @@ -1,33 +1,7 @@ -/** - * Blade 统一配置类型定义 - * 合并了 config.json 和 settings.json 的所有配置项 - */ - export type ProviderType = string; -/** - * 权限模式枚举 - * - * ## DEFAULT 模式(默认) - * - Auto-approve: ReadOnly 工具(Read/Glob/Grep/WebFetch/WebSearch/TaskOutput/TaskCreate/TaskGet/TaskUpdate/TaskList/Plan) - * - Needs confirm: Write 工具(Edit/Write/NotebookEdit)、Execute 工具(Bash/Task/Skill/SlashCommand) - * - * ## AUTO_EDIT 模式 - * - Auto-approve: ReadOnly + Write 工具 - * - Needs confirm: Execute 工具(Bash/Task/Skill/SlashCommand) - * - 适用场景:频繁修改代码的开发任务 - * - * ## YOLO 模式(危险) - * - Auto-approve: 所有工具(ReadOnly + Write + Execute) - * - WARNING: 完全信任 AI,跳过所有确认 - * - 适用场景:高度可控的环境或演示场景 - * - * ## PLAN 模式 - * - Auto-approve: ReadOnly 工具(只读操作,无副作用) - * - Blocks all modifications: Write 和 Execute 工具 - * - Special tools: ExitPlanMode(用于提交方案) - * - 适用场景:调研阶段,生成实现方案,用户批准后退出 Plan 模式 - */ +/** DEFAULT asks before writes/executes; AUTO_EDIT permits writes; YOLO permits + * every tool; PLAN permits read-only tools until ExitPlanMode. */ export enum PermissionMode { DEFAULT = 'default', AUTO_EDIT = 'autoEdit', @@ -88,26 +62,20 @@ export type ReasoningEffortSelection = | 'high' | 'xhigh' | 'max'; - export type ServiceTierSelection = 'auto' | 'standard' | 'fast' | 'flex'; - export type ResponseVerbositySelection = 'auto' | 'low' | 'medium' | 'high'; - export type BuiltInCommunicationStyleSelection = | 'auto' | 'pragmatic' | 'friendly' | 'explanatory'; - export type CustomCommunicationStyleSelection = | `user:${string}` | `project:${string}` | `plugin:${string}:${string}`; - export type CommunicationStyleSelection = | BuiltInCommunicationStyleSelection | CustomCommunicationStyleSelection; - export interface PluginSourcePolicy { restrictToAllowedSources: boolean; requireGitCommitSha: boolean; @@ -133,24 +101,13 @@ export interface LspServerConfig { } import { UiTheme } from '@/api/schemas.js'; -/** - * Hooks 配置 - * 导入自 hooks 模块 - */ import type { HookConfig as HookConfigType } from '../hooks/types/HookTypes.js'; export type HookConfig = HookConfigType; - export interface BladeConfig { - // ===================================== - // 基础配置 (来自 config.json - 扁平化) - // ===================================== - - // 多模型配置 currentModelId: string; // 当前激活的模型 ID models: ModelConfig[]; // 所有模型配置 modelProviders: Record; // 自定义 Provider 渠道 - // 全局默认参数 temperature: number; maxContextTokens?: number; // 已弃用;运行时使用 pi-ai model.contextWindow maxOutputTokens?: number; // 输出 token 限制(传给 API 的 max_tokens),undefined 表示让 API 使用默认值 @@ -168,56 +125,35 @@ export interface BladeConfig { providerRequestPendingBytes?: number; // 进程级 Provider admission pending request footprint 上限 agentTeamsEnabled?: boolean; // 启用正式 Agent Teams 协作能力 - // UI codeTheme: string; uiTheme: UiTheme; language: string; fontSize: number; - - // General Settings autoSaveSessions: boolean; notifyBuild: boolean; notifyErrors: boolean; notifySounds: boolean; privacyTelemetry: boolean; privacyCrash: boolean; - // Default communication style selection applied to new turns (e.g. 'auto', - // a built-in id, or a 'project:' reference). Optional for backward compat. + // Default for new turns: built-in id or project:; optional for compatibility. communicationStyle?: string; - // 核心 - // debug 支持 boolean 或字符串过滤器(如 "agent,ui" 或 "!chat,!loop") + // 核心 debug 支持 boolean 或字符串过滤器(如 "agent,ui" 或 "!chat,!loop") debug: string | boolean; - - // MCP mcpEnabled: boolean; mcpServers: Record; // 启动项目投影;执行时按 Session 重解析 - // LSP lspServers: Record; // Session 私有、按 source project 重解析 - // ===================================== - // 行为配置 (来自 settings.json) - // ===================================== - - // 权限 permissions: PermissionConfig; permissionMode: PermissionMode; - - // Hooks hooks: HookConfig; // Plugins (later workspace layers override by plugin name) enabledPlugins: Record; pluginSourcePolicy: PluginSourcePolicy; - - // 环境变量 env: Record; - - // 其他 disableAllHooks: boolean; - - // Agentic Loop 配置 maxTurns: number; // -1 = 无限制, 0 = 完全禁用对话, N > 0 = 限制轮次 maxConcurrentTasks: number; // 同一进程内允许同时运行的顶层任务数 maxQueuedTasks: number; // 等待 admission 的顶层任务上限 @@ -228,40 +164,26 @@ export interface BladeConfig { sessionProjectionIdleMs: number; // Session projection idle eviction TTL } -/** - * 权限配置 - */ export interface PermissionConfig { allow: string[]; ask: string[]; deny: string[]; } -/** - * 运行时配置类型 - * 继承 BladeConfig (持久化配置) + CLI 专属字段 (临时配置) - * - * CLI 专属字段只在当前会话有效,不会保存到配置文件 - */ export interface RuntimeConfig extends BladeConfig { - // CLI 专属字段 - 系统提示 systemPrompt?: string; // 替换默认系统提示 appendSystemPrompt?: string; // 追加到默认系统提示 - // CLI 专属字段 - 会话管理 initialMessage?: string; // 初始消息(用于自动发送) resumeSessionId?: string; // 恢复会话 ID forkSession?: boolean; // 创建新会话 ID(fork 模式) - // CLI 专属字段 - 工具过滤 allowedTools?: string[]; // 允许的工具列表(白名单) disallowedTools?: string[]; // 禁止的工具列表(黑名单) - // CLI 专属字段 - MCP mcpConfigPaths?: string[]; // MCP 配置文件路径 strictMcpConfig?: boolean; // 仅使用 CLI 指定的 MCP 服务器 - // CLI 专属字段 - 其他 model?: string; // 当前运行覆盖模型(模型配置 ID) addDirs?: string[]; // 额外允许访问的目录 outputFormat?: 'text' | 'json' | 'stream-json' | 'jsonl'; // 输出格式 @@ -273,23 +195,14 @@ export interface RuntimeConfig extends BladeConfig { settingSources?: string; // 配置来源列表 } -/** - * MCP 服务器配置 - */ export interface McpServerConfig { type: 'stdio' | 'sse' | 'http'; - - // stdio 传输 command?: string; args?: string[]; env?: Record; cwd?: string; - - // http/sse 传输 url?: string; headers?: Record; - - // 通用配置 timeout?: number; idleTimeout?: number; sampling?: { @@ -317,24 +230,18 @@ export interface McpServerConfig { maxTasksPerSession?: number; maxLifetimeMs?: number; }; - - // OAuth 配置 oauth?: { enabled?: boolean; clientId?: string; scopes?: string[]; callbackPort?: number; }; - - // 健康监控配置 healthCheck?: { enabled?: boolean; interval?: number; // 检查间隔(毫秒) timeout?: number; // 超时时间(毫秒) failureThreshold?: number; // 失败阈值 }; - - // 意外断连恢复配置 recovery?: { enabled?: boolean; maxAttempts?: number; @@ -345,11 +252,6 @@ export interface McpServerConfig { }; } -/** - * SetupWizard 保存的配置字段 - * (API 连接相关的核心配置) - * 注意:这是用于创建第一个模型配置的数据 - */ export interface SetupConfig { displayName?: string; provider: ProviderType; diff --git a/packages/cli/src/context/CompactionService.ts b/packages/cli/src/context/CompactionService.ts index 5f9c81c60..6e924d423 100644 --- a/packages/cli/src/context/CompactionService.ts +++ b/packages/cli/src/context/CompactionService.ts @@ -1,7 +1,4 @@ -/** - * 上下文压缩服务 - * 负责协调整个压缩流程:分析文件、生成总结、创建压缩消息 - */ +/** 上下文压缩服务 负责协调整个压缩流程:分析文件、生成总结、创建压缩消息 */ import { promises as fs } from 'node:fs'; import path from 'node:path'; @@ -42,9 +39,7 @@ import { TokenCounter } from './TokenCounter.js'; const logger = createLogger(LogCategory.CONTEXT); -/** - * 压缩选项 - */ +/** 压缩选项 */ export interface CompactionOptions { /** 触发方式:自动或手动 */ trigger: 'auto' | 'manual'; @@ -75,9 +70,7 @@ export interface CompactionOptions { workspaceAccess?: 'full' | 'none'; } -/** - * 压缩结果 - */ +/** 压缩结果 */ export interface CompactionResult { /** 是否成功 */ success: boolean; @@ -417,9 +410,7 @@ function reduceCompactionSampleInput( return nextChars < currentChars ? next : undefined; } -/** - * 构建面向继续执行的有界压缩 prompt。 - */ +/** 构建面向继续执行的有界压缩 prompt。 */ export function buildCompactionPrompt( messages: readonly Message[], fileContents: readonly FileContent[], @@ -474,9 +465,7 @@ ${messagesText} ${fileContents.length > 0 ? `## Important Files\n\n${filesText}\n\n` : ''}Respond with one section followed by one

section. The summary must obey the ledger contract above.`; } -/** - * Compaction Service - 上下文压缩服务 - */ +/** Compaction Service - 上下文压缩服务 */ export class CompactionService { /** 保留比例(20%) */ private static readonly RETAIN_PERCENT = 0.2; @@ -519,8 +508,7 @@ export class CompactionService { } logger.debug(`[CompactionService] preTokens source: ${tokenSource}`); - // 执行 Compaction Hook(压缩前) - // Hook 可以阻止压缩 + // 执行 Compaction Hook(压缩前) Hook 可以阻止压缩 let blockReason: string | undefined; let completedSampleAttempts = 0; let completedUsage: UsageInfo | undefined; diff --git a/packages/cli/src/context/ContextAssembler.ts b/packages/cli/src/context/ContextAssembler.ts deleted file mode 100644 index f2e101727..000000000 --- a/packages/cli/src/context/ContextAssembler.ts +++ /dev/null @@ -1,240 +0,0 @@ -/** - * ContextAssembler - 从 JSONL 事件流重建 ContextData - * - * 职责:将 PersistentStore 的 SessionEvent[] 转换为结构化的上下文数据。 - * 集中了之前散落在 PersistentStore.loadSession/loadConversation 和 - * ContextManager.loadSession 里的重建逻辑。 - */ - -import type { JsonObject, JsonValue, MessageRole } from '../store/types.js'; -import type { - ContextData, - ContextMessage, - ConversationContext, - SessionContext, - SessionEvent, - SystemContext, - ToolCall, - WorkspaceContext, -} from './types.js'; - -export interface AssembledSession { - session: SessionContext; - conversation: ConversationContext; - toolCalls: ToolCall[]; -} - -export class ContextAssembler { - /** - * 从 JSONL 事件流重建完整的会话数据 - */ - assemble(events: SessionEvent[]): AssembledSession | null { - if (events.length === 0) return null; - - const session = this.assembleSession(events); - const conversation = this.assembleConversation(events); - const toolCalls = this.assembleToolCalls(events); - - return { session, conversation, toolCalls }; - } - - /** - * 从事件流重建完整的 ContextData - */ - assembleContextData( - events: SessionEvent[], - system: SystemContext, - workspace: WorkspaceContext - ): ContextData | null { - const assembled = this.assemble(events); - if (!assembled) return null; - - return { - layers: { - system, - session: assembled.session, - conversation: assembled.conversation, - tool: { - recentCalls: assembled.toolCalls, - toolStates: {}, - dependencies: {}, - }, - workspace, - }, - metadata: { - totalTokens: 0, - priority: 1, - lastUpdated: Date.now(), - }, - }; - } - - /** - * 重建 SessionContext - */ - private assembleSession(events: SessionEvent[]): SessionContext { - const sessionCreated = events.find((e) => e.type === 'session_created'); - const sessionId = sessionCreated?.sessionId ?? events[0].sessionId; - const startTime = new Date( - sessionCreated?.timestamp ?? events[0].timestamp - ).getTime(); - - // 从 session_updated 事件中提取最新的配置 - const updates = events.filter((e) => e.type === 'session_updated'); - const latestUpdate = updates.length > 0 ? updates[updates.length - 1] : null; - - return { - sessionId, - userId: undefined, - preferences: {}, - configuration: (latestUpdate?.data as JsonObject) ?? {}, - startTime, - }; - } - - /** - * 重建 ConversationContext(包含 messages + summary) - */ - private assembleConversation(events: SessionEvent[]): ConversationContext { - const messageMap = new Map< - string, - { - id: string; - role: MessageRole; - content: string; - timestamp: number; - metadata?: JsonObject; - } - >(); - - let latestSummary: string | undefined; - - for (const event of events) { - if (event.type === 'message_created') { - messageMap.set(event.data.messageId, { - id: event.data.messageId, - role: event.data.role, - content: '', - timestamp: new Date(event.timestamp).getTime(), - metadata: - event.data.model || - event.data.inboxMessageId || - (event.data.metadata && - typeof event.data.metadata === 'object' && - !Array.isArray(event.data.metadata)) - ? { - ...(event.data.metadata && - typeof event.data.metadata === 'object' && - !Array.isArray(event.data.metadata) - ? event.data.metadata - : {}), - ...(event.data.model ? { model: event.data.model } : {}), - ...(event.data.inboxMessageId - ? { inboxMessageId: event.data.inboxMessageId } - : {}), - } - : undefined, - }); - } - - if (event.type === 'part_created') { - const { partType, messageId, payload } = event.data; - const message = messageMap.get(messageId); - - if (partType === 'text' && message) { - const p = payload as { text?: string }; - message.content = p.text ?? ''; - } - - if (partType === 'image' && message) { - message.content = message.content ? `${message.content}\n[Image]` : '[Image]'; - } - - // 提取 compaction summary - if (partType === 'summary') { - const p = payload as { text?: string }; - if (p.text) latestSummary = p.text; - } - } - - // part_updated 覆盖已有内容 - if (event.type === 'part_updated') { - const { partType, messageId, payload } = event.data; - const message = messageMap.get(messageId); - - if (partType === 'text' && message) { - const p = payload as { text?: string }; - message.content = p.text ?? ''; - } - } - } - - const messages: ContextMessage[] = Array.from(messageMap.values()); - const lastEvent = events[events.length - 1]; - const lastActivity = new Date(lastEvent.timestamp).getTime(); - - return { - messages, - summary: latestSummary, - topics: [], - lastActivity, - }; - } - - /** - * 重建 ToolCall 列表 - */ - private assembleToolCalls(events: SessionEvent[]): ToolCall[] { - const toolCalls = new Map(); - - for (const event of events) { - if (event.type !== 'part_created') continue; - - const { partType, partId, payload } = event.data; - - if (partType === 'tool_call') { - const p = payload as { - toolCallId?: string; - toolName?: string; - input?: JsonValue; - }; - const id = p.toolCallId ?? partId; - toolCalls.set(id, { - id, - name: p.toolName ?? 'unknown', - input: (p.input ?? null) as JsonValue, - timestamp: new Date(event.timestamp).getTime(), - status: 'pending', - }); - } - - if (partType === 'tool_result') { - const p = payload as { - toolCallId?: string; - toolName?: string; - output?: JsonValue; - error?: string | null; - }; - const id = p.toolCallId ?? partId; - const existing = toolCalls.get(id); - if (existing) { - existing.output = (p.output ?? undefined) as JsonValue | undefined; - existing.status = p.error ? 'error' : 'success'; - existing.error = p.error ?? undefined; - } else { - toolCalls.set(id, { - id, - name: p.toolName ?? 'unknown', - input: null as unknown as JsonValue, - output: (p.output ?? undefined) as JsonValue | undefined, - timestamp: new Date(event.timestamp).getTime(), - status: p.error ? 'error' : 'success', - error: p.error ?? undefined, - }); - } - } - } - - return Array.from(toolCalls.values()); - } -} diff --git a/packages/cli/src/context/ContextManager.ts b/packages/cli/src/context/ContextManager.ts index c03eb37e1..5cf37c97b 100644 --- a/packages/cli/src/context/ContextManager.ts +++ b/packages/cli/src/context/ContextManager.ts @@ -1,9 +1,6 @@ /** - * 上下文管理器 - PersistentStore 的薄门面 - * - * 历史上此类包含内存模型、压缩、过滤、搜索等功能, - * 但这些功能已迁移到独立模块(CompactionService、ReactiveCompaction 等), - * 仅保留 JSONL 持久化委托方法。 + * 上下文管理器 - PersistentStore 的薄门面

历史上此类包含内存模型、压缩、过滤、搜索等功能, + * 但这些功能已迁移到独立模块(CompactionService、ReactiveCompaction 等), 仅保留 JSONL 持久化委托方法。 */ import type { SubagentInfoForContext } from '../agent/types.js'; @@ -19,59 +16,29 @@ import type { SubagentRunRef, } from './types.js'; -/** - * 上下文管理器 - 统一管理所有上下文相关操作 - */ +/** 上下文管理器 - 统一管理所有上下文相关操作 */ export class ContextManager { private readonly persistent: PersistentStore; - private readonly options: ContextManagerOptions; - /** - * 获取持久化存储实例(供外部直接调用 JSONL 操作) - */ + /** 获取持久化存储实例(供外部直接调用 JSONL 操作) */ get persistentStore(): PersistentStore { return this.persistent; } constructor(options: Partial = {}) { - this.options = { - projectPath: options.projectPath || getCwd(), - ...(options.stateStorage ? { stateStorage: options.stateStorage } : {}), - storage: { - maxMemorySize: 1000, - persistentPath: '', - cacheSize: 100, - compressionEnabled: true, - ...options.storage, - }, - defaultFilter: { - maxTokens: 32000, - maxMessages: 50, - timeWindow: 24 * 60 * 60 * 1000, - ...options.defaultFilter, - }, - compressionThreshold: options.compressionThreshold || 6000, - enableVectorSearch: options.enableVectorSearch || false, - }; - this.persistent = new PersistentStore( - this.options.projectPath, - 100, + options.projectPath ?? getCwd(), undefined, - this.options.stateStorage + options.stateStorage ); } - /** - * 初始化持久化存储目录 - */ + /** 初始化持久化存储目录 */ async initialize(): Promise { await this.persistent.initialize(); } - /** - * 保存消息到 JSONL (直接访问 PersistentStore,不依赖 currentSessionId) - */ + /** 保存消息到 JSONL (直接访问 PersistentStore,不依赖 currentSessionId) */ async saveMessage( sessionId: string, role: 'user' | 'assistant' | 'system', @@ -92,9 +59,7 @@ export class ContextManager { ); } - /** - * 保存工具调用到 JSONL (直接访问 PersistentStore) - */ + /** 保存工具调用到 JSONL (直接访问 PersistentStore) */ async saveToolUse( sessionId: string, toolName: string, @@ -113,9 +78,7 @@ export class ContextManager { ); } - /** - * 保存工具结果到 JSONL (直接访问 PersistentStore) - */ + /** 保存工具结果到 JSONL (直接访问 PersistentStore) */ async saveToolResult( sessionId: string, toolId: string, @@ -140,9 +103,7 @@ export class ContextManager { ); } - /** - * 保存压缩边界和总结到 JSONL (直接访问 PersistentStore) - */ + /** 保存压缩边界和总结到 JSONL (直接访问 PersistentStore) */ async saveCompaction( sessionId: string, summary: string, diff --git a/packages/cli/src/context/FileAnalyzer.ts b/packages/cli/src/context/FileAnalyzer.ts index 005f29471..c6e2859fc 100644 --- a/packages/cli/src/context/FileAnalyzer.ts +++ b/packages/cli/src/context/FileAnalyzer.ts @@ -1,7 +1,4 @@ -/** - * 文件分析服务 - * 用于从对话中提取重点文件并读取内容 - */ +/** 文件分析服务 用于从对话中提取重点文件并读取内容 */ import { readFile } from 'node:fs/promises'; import { basename, isAbsolute, resolve } from 'node:path'; @@ -11,9 +8,7 @@ import type { } from '../services/ChatServiceInterface.js'; import { PathSecurity } from '../utils/pathSecurity.js'; -/** - * 文件引用信息 - */ +/** 文件引用信息 */ export interface FileReference { /** 文件路径 */ path: string; @@ -25,9 +20,7 @@ export interface FileReference { wasModified: boolean; } -/** - * 文件内容 - */ +/** 文件内容 */ export interface FileContent { /** 文件路径 */ path: string; @@ -41,9 +34,7 @@ export interface FileContent { includedLines: number; } -/** - * File Analyzer - 分析对话中的文件引用 - */ +/** File Analyzer - 分析对话中的文件引用 */ export class FileAnalyzer { /** 最多包含的文件数量 */ private static readonly MAX_FILES = 5; @@ -61,8 +52,7 @@ export class FileAnalyzer { const fileMap = new Map(); messages.forEach((msg, index) => { - // 从消息内容中提取文件路径 - // 处理多模态消息:提取纯文本内容 + // 从消息内容中提取文件路径 处理多模态消息:提取纯文本内容 const textContent = typeof msg.content === 'string' ? msg.content diff --git a/packages/cli/src/context/ReactiveCompaction.ts b/packages/cli/src/context/ReactiveCompaction.ts index cbd3c850c..3a757a749 100644 --- a/packages/cli/src/context/ReactiveCompaction.ts +++ b/packages/cli/src/context/ReactiveCompaction.ts @@ -1,7 +1,5 @@ /** - * ReactiveCompaction — 反应式紧急压缩 - * - * 当 LLM 返回 413 (prompt_too_long) 错误时触发的紧急压缩。 + * ReactiveCompaction — 反应式紧急压缩

当 LLM 返回 413 (prompt_too_long) 错误时触发的紧急压缩。 * 每轮最多尝试一次,作为最后一道防线。 */ @@ -66,10 +64,7 @@ export class ReactiveCompaction { return !this.hasAttempted; } - /** - * 尝试反应式压缩。每轮最多一次。 - * 先尝试 snip(轻量),再尝试 LLM 压缩(重量)。 - */ + /** 尝试反应式压缩。每轮最多一次。 先尝试 snip(轻量),再尝试 LLM 压缩(重量)。 */ async tryReactiveCompact( messages: Message[], options: ReactiveCompactOptions diff --git a/packages/cli/src/context/TokenBudget.ts b/packages/cli/src/context/TokenBudget.ts index cab10fce4..0cd42ea69 100644 --- a/packages/cli/src/context/TokenBudget.ts +++ b/packages/cli/src/context/TokenBudget.ts @@ -1,9 +1,6 @@ /** - * TokenBudget — 递减收益检测 - * - * 跟踪 LLM 续写模式,当检测到递减收益时建议停止: - * - 连续 N 次续写(max output recovery)且每次增量很小 - * - Token 使用率接近预算上限 + * TokenBudget — 递减收益检测

跟踪 LLM 续写模式,当检测到递减收益时建议停止: - 连续 N 次续写(max output + * recovery)且每次增量很小 - Token 使用率接近预算上限 */ export interface BudgetTracker { @@ -50,9 +47,7 @@ export function checkTokenBudget(tracker: BudgetTracker): 'continue' | 'stop' { return 'continue'; } -/** - * 创建初始 BudgetTracker - */ +/** 创建初始 BudgetTracker */ export function createBudgetTracker(opts: { budget: number; isSubagent?: boolean; @@ -66,9 +61,7 @@ export function createBudgetTracker(opts: { }; } -/** - * 记录一次 LLM 输出 - */ +/** 记录一次 LLM 输出 */ export function recordOutput( tracker: BudgetTracker, outputTokens: number, diff --git a/packages/cli/src/context/TokenCounter.ts b/packages/cli/src/context/TokenCounter.ts index 0a4183a2a..c860b28ac 100644 --- a/packages/cli/src/context/TokenCounter.ts +++ b/packages/cli/src/context/TokenCounter.ts @@ -1,7 +1,4 @@ -/** - * Token 计算服务 - * 用于计算消息的 token 数量,判断是否需要压缩 - */ +/** Token 计算服务 用于计算消息的 token 数量,判断是否需要压缩 */ import { encodingForModel } from 'js-tiktoken'; import type { @@ -13,9 +10,7 @@ interface Encoding { encode: (text: string) => number[]; } -/** - * Token Counter - 计算和管理 token 数量 - */ +/** Token Counter - 计算和管理 token 数量 */ export class TokenCounter { private static encodingCache = new Map(); @@ -72,16 +67,6 @@ export class TokenCounter { return this.getEncoding(modelName).encode(text).length; } - /** - * 获取 token 限制(直接返回配置的 maxTokens) - * - * @param maxTokens - 配置的 token 限制 - * @returns token 限制 - */ - static getTokenLimit(maxTokens: number): number { - return maxTokens; - } - /** * 检查是否需要压缩 * @@ -177,26 +162,4 @@ export class TokenCounter { return tokens; } - - /** - * 清理 encoding 缓存 - * (用于释放内存) - */ - static clearCache(): void { - this.encodingCache.clear(); - } - - /** - * 估算文本的 token 数量(快速粗略估算) - * - * @param text - 文本内容 - * @returns 估算的 token 数量 - */ - static estimateTokens(text: string): number { - // 粗略估算:1 token ≈ 4 字符(英文)或 1.5 字符(中文) - const chineseChars = (text.match(/[\u4e00-\u9fa5]/g) || []).length; - const otherChars = text.length - chineseChars; - - return Math.ceil(chineseChars / 1.5 + otherChars / 4); - } } diff --git a/packages/cli/src/context/ToolResultBudget.ts b/packages/cli/src/context/ToolResultBudget.ts index c781879b8..b56b98d1b 100644 --- a/packages/cli/src/context/ToolResultBudget.ts +++ b/packages/cli/src/context/ToolResultBudget.ts @@ -1,9 +1,4 @@ -/** - * ToolResultBudget — 工具结果大小控制 - * - * 当工具结果超过阈值时,将完整内容持久化到磁盘, - * 只保留预览 + 文件路径引用。防止上下文膨胀。 - */ +/** ToolResultBudget — 工具结果大小控制 当工具结果超过阈值时,将完整内容持久化到磁盘, 只保留预览 + 文件路径引用。防止上下文膨胀。 */ import * as fs from 'fs'; import { nanoid } from 'nanoid'; @@ -14,11 +9,7 @@ const DEFAULT_MAX_RESULT_CHARS = 100_000; const PREVIEW_CHARS = 2000; const MAX_TOOL_RESULTS_PER_MESSAGE_CHARS = 200_000; -/** - * 根据模型上下文窗口动态计算工具结果预算 - * - * 策略:小窗口模型使用更紧凑的预算,大窗口模型放宽限制 - */ +/** 根据模型上下文窗口动态计算工具结果预算 策略:小窗口模型使用更紧凑的预算,大窗口模型放宽限制 */ export function computeAdaptiveBudget(maxContextTokens?: number): { maxCharsPerResult: number; maxCharsPerMessage: number; @@ -41,12 +32,7 @@ export function computeAdaptiveBudget(maxContextTokens?: number): { }; } -/** - * 消息级工具结果聚合预算 - * - * 防止并行工具在同一轮中产生过多总输出。 - * 例如 5 个 Grep 各返回 50K,总计 250K 超出 200K 限制。 - */ +/** 消息级工具结果聚合预算 防止并行工具在同一轮中产生过多总输出。 例如 5 个 Grep 各返回 50K,总计 250K 超出 200K 限制。 */ export class MessageBudgetTracker { private currentChars = 0; @@ -59,16 +45,6 @@ export class MessageBudgetTracker { remaining(): number { return Math.max(0, MAX_TOOL_RESULTS_PER_MESSAGE_CHARS - this.currentChars); } - - /** 是否已超出预算 */ - isExhausted(): boolean { - return this.currentChars >= MAX_TOOL_RESULTS_PER_MESSAGE_CHARS; - } - - /** 每轮开始时重置 */ - reset(): void { - this.currentChars = 0; - } } export interface BudgetOptions { @@ -148,12 +124,7 @@ export function applyToolResultBudget( return result; } -/** - * 消息级预算检查(内部辅助函数) - * - * 当内容在 per-tool 预算内时,进一步检查是否会超出 - * 消息级聚合预算,必要时截断并持久化到磁盘。 - */ +/** 消息级预算检查(内部辅助函数) 当内容在 per-tool 预算内时,进一步检查是否会超出 消息级聚合预算,必要时截断并持久化到磁盘。 */ function applyMessageBudget( original: string | object, contentStr: string, @@ -202,9 +173,7 @@ function applyMessageBudget( return original; } -/** - * 持久化完整内容到磁盘并返回截断结果(内部辅助函数) - */ +/** 持久化完整内容到磁盘并返回截断结果(内部辅助函数) */ function persistAndSummarize( fullContent: string, truncated: string, diff --git a/packages/cli/src/context/processors/ContextCompressor.ts b/packages/cli/src/context/processors/ContextCompressor.ts deleted file mode 100644 index 439037d52..000000000 --- a/packages/cli/src/context/processors/ContextCompressor.ts +++ /dev/null @@ -1,338 +0,0 @@ -import { CompressedContext, ContextData, ContextMessage, ToolCall } from '../types.js'; - -/** - * 上下文压缩器 - 智能压缩上下文以节省 token 使用 - */ -export class ContextCompressor { - private readonly maxSummaryLength: number; - private readonly keyPointsLimit: number; - private readonly recentMessagesLimit: number; - - constructor( - maxSummaryLength: number = 500, - keyPointsLimit: number = 10, - recentMessagesLimit: number = 20 - ) { - this.maxSummaryLength = maxSummaryLength; - this.keyPointsLimit = keyPointsLimit; - this.recentMessagesLimit = recentMessagesLimit; - } - - /** - * 压缩上下文数据 - */ - async compress(contextData: ContextData): Promise { - const messages = contextData.layers.conversation.messages; - const toolCalls = contextData.layers.tool.recentCalls; - - // 分离系统消息和用户/助手消息 - const systemMessages = messages.filter((m) => m.role === 'system'); - const conversationMessages = messages.filter((m) => m.role !== 'system'); - - // 获取最近的消息(保持完整) - const recentMessages = this.getRecentMessages(conversationMessages); - - // 压缩较旧的消息 - const olderMessages = conversationMessages.slice(0, -this.recentMessagesLimit); - const summary = await this.generateSummary(olderMessages); - - // 提取关键要点 - const keyPoints = this.extractKeyPoints(olderMessages, toolCalls); - - // 生成工具摘要 - const toolSummary = this.generateToolSummary(toolCalls); - - // 估算 token 数量 - const tokenCount = this.estimateTokenCount( - summary, - keyPoints, - recentMessages, - toolSummary - ); - - return { - summary, - keyPoints, - recentMessages: [...systemMessages, ...recentMessages], - toolSummary, - tokenCount, - }; - } - - /** - * 获取最近的消息 - */ - private getRecentMessages(messages: ContextMessage[]): ContextMessage[] { - return messages.slice(-this.recentMessagesLimit); - } - - /** - * 生成对话摘要 - */ - private async generateSummary(messages: ContextMessage[]): Promise { - if (messages.length === 0) { - return ''; - } - - // 简单的摘要生成策略(可以后续接入 LLM 进行更智能的摘要) - const topics = new Set(); - const actions = new Set(); - const decisions = new Set(); - - for (const message of messages) { - const content = message.content.toLowerCase(); - - // 检测主题关键词 - const topicKeywords = ['关于', '讨论', '问题', '项目', '功能', '需求']; - topicKeywords.forEach((keyword) => { - if (content.includes(keyword)) { - const context = this.extractContext(content, keyword, 50); - if (context) topics.add(context); - } - }); - - // 检测动作关键词 - const actionKeywords = ['创建', '删除', '修改', '更新', '实现', '开发']; - actionKeywords.forEach((keyword) => { - if (content.includes(keyword)) { - const context = this.extractContext(content, keyword, 30); - if (context) actions.add(context); - } - }); - - // 检测决策关键词 - const decisionKeywords = ['决定', '选择', '确定', '采用', '使用']; - decisionKeywords.forEach((keyword) => { - if (content.includes(keyword)) { - const context = this.extractContext(content, keyword, 40); - if (context) decisions.add(context); - } - }); - } - - // 构建摘要 - let summary = `对话涉及 ${messages.length} 条消息。`; - - if (topics.size > 0) { - summary += ` 主要讨论:${Array.from(topics).slice(0, 3).join('、')}。`; - } - - if (actions.size > 0) { - summary += ` 执行操作:${Array.from(actions).slice(0, 3).join('、')}。`; - } - - if (decisions.size > 0) { - summary += ` 关键决策:${Array.from(decisions).slice(0, 2).join('、')}。`; - } - - return summary.length > this.maxSummaryLength - ? summary.substring(0, this.maxSummaryLength) + '...' - : summary; - } - - /** - * 提取关键要点 - */ - private extractKeyPoints( - messages: ContextMessage[], - toolCalls: ToolCall[] - ): string[] { - const keyPoints: Set = new Set(); - - // 从消息中提取关键点 - for (const message of messages) { - if (message.role === 'user') { - // 用户的问题和请求 - const questions = this.extractQuestions(message.content); - questions.forEach((q) => keyPoints.add(`用户问题:${q}`)); - - const requests = this.extractRequests(message.content); - requests.forEach((r) => keyPoints.add(`用户请求:${r}`)); - } else if (message.role === 'assistant') { - // 助手的重要建议和解决方案 - const solutions = this.extractSolutions(message.content); - solutions.forEach((s) => keyPoints.add(`解决方案:${s}`)); - } - } - - // 从工具调用中提取关键点 - const toolUsage = this.summarizeToolUsage(toolCalls); - toolUsage.forEach((usage) => keyPoints.add(`工具使用:${usage}`)); - - return Array.from(keyPoints).slice(0, this.keyPointsLimit); - } - - /** - * 生成工具调用摘要 - */ - private generateToolSummary(toolCalls: ToolCall[]): string { - if (toolCalls.length === 0) { - return ''; - } - - const toolStats = new Map< - string, - { count: number; success: number; recent: number } - >(); - const recentTime = Date.now() - 10 * 60 * 1000; // 最近10分钟 - - for (const call of toolCalls) { - const stats = toolStats.get(call.name) || { count: 0, success: 0, recent: 0 }; - stats.count++; - if (call.status === 'success') stats.success++; - if (call.timestamp > recentTime) stats.recent++; - toolStats.set(call.name, stats); - } - - const summaryParts: string[] = []; - for (const [toolName, stats] of Array.from(toolStats.entries())) { - const successRate = Math.round((stats.success / stats.count) * 100); - summaryParts.push(`${toolName}(${stats.count}次,成功率${successRate}%)`); - } - - return `工具调用:${summaryParts.join('、')}`; - } - - /** - * 估算 token 数量(简单估算) - */ - private estimateTokenCount( - summary: string, - keyPoints: string[], - recentMessages: ContextMessage[], - toolSummary?: string - ): number { - let totalLength = summary.length + keyPoints.join(' ').length; - - if (toolSummary) { - totalLength += toolSummary.length; - } - - for (const message of recentMessages) { - totalLength += message.content.length; - } - - // 粗略估算:4个字符 ≈ 1个 token(对于中文) - return Math.ceil(totalLength / 4); - } - - /** - * 从内容中提取上下文 - */ - private extractContext( - content: string, - keyword: string, - maxLength: number - ): string | null { - const index = content.indexOf(keyword); - if (index === -1) return null; - - const start = Math.max(0, index - maxLength / 2); - const end = Math.min(content.length, index + maxLength / 2); - - return content.substring(start, end).trim(); - } - - /** - * 提取问题 - */ - private extractQuestions(content: string): string[] { - const questions: string[] = []; - const questionMarkers = ['?', '?', '如何', '怎么', '什么', '为什么']; - - const sentences = content.split(/[。!.!]/); - for (const sentence of sentences) { - if (questionMarkers.some((marker) => sentence.includes(marker))) { - const cleaned = sentence.trim(); - if (cleaned.length > 5 && cleaned.length < 100) { - questions.push(cleaned); - } - } - } - - return questions.slice(0, 3); // 最多返回3个问题 - } - - /** - * 提取请求 - */ - private extractRequests(content: string): string[] { - const requests: string[] = []; - const requestMarkers = ['请', '帮我', '需要', '想要', '希望', '能否']; - - const sentences = content.split(/[。!.!]/); - for (const sentence of sentences) { - if (requestMarkers.some((marker) => sentence.includes(marker))) { - const cleaned = sentence.trim(); - if (cleaned.length > 5 && cleaned.length < 100) { - requests.push(cleaned); - } - } - } - - return requests.slice(0, 3); // 最多返回3个请求 - } - - /** - * 提取解决方案 - */ - private extractSolutions(content: string): string[] { - const solutions: string[] = []; - const solutionMarkers = ['可以', '建议', '推荐', '应该', '最好', '解决方案']; - - const sentences = content.split(/[。!.!]/); - for (const sentence of sentences) { - if (solutionMarkers.some((marker) => sentence.includes(marker))) { - const cleaned = sentence.trim(); - if (cleaned.length > 10 && cleaned.length < 150) { - solutions.push(cleaned); - } - } - } - - return solutions.slice(0, 3); // 最多返回3个解决方案 - } - - /** - * 总结工具使用情况 - */ - private summarizeToolUsage(toolCalls: ToolCall[]): string[] { - const summary: string[] = []; - const recentCalls = toolCalls.filter( - (call) => Date.now() - call.timestamp < 30 * 60 * 1000 // 最近30分钟 - ); - - if (recentCalls.length > 0) { - const toolGroups = new Map(); - recentCalls.forEach((call) => { - const group = toolGroups.get(call.name) || []; - group.push(call); - toolGroups.set(call.name, group); - }); - - for (const [toolName, calls] of Array.from(toolGroups.entries())) { - const successCount = calls.filter((c) => c.status === 'success').length; - summary.push(`${toolName}(${calls.length}次,${successCount}成功)`); - } - } - - return summary.slice(0, 5); // 最多返回5个工具使用摘要 - } - - /** - * 检查是否需要压缩 - */ - shouldCompress(contextData: ContextData, maxTokens: number): boolean { - const estimatedTokens = this.estimateCurrentTokens(contextData); - return estimatedTokens > maxTokens * 0.8; // 当超过80%限制时开始压缩 - } - - /** - * 估算当前上下文的 token 数量 - */ - private estimateCurrentTokens(contextData: ContextData): number { - const messages = contextData.layers.conversation.messages; - const totalLength = messages.reduce((sum, msg) => sum + msg.content.length, 0); - return Math.ceil(totalLength / 4); // 简单估算 - } -} diff --git a/packages/cli/src/context/processors/ContextFilter.ts b/packages/cli/src/context/processors/ContextFilter.ts deleted file mode 100644 index 41e0952d1..000000000 --- a/packages/cli/src/context/processors/ContextFilter.ts +++ /dev/null @@ -1,397 +0,0 @@ -import { - ContextData, - ContextMessage, - ContextFilter as FilterOptions, -} from '../types.js'; - -/** - * 上下文过滤器 - 根据配置过滤和筛选上下文内容 - */ -export class ContextFilter { - private readonly defaultOptions: Required; - - constructor(defaultOptions?: FilterOptions) { - this.defaultOptions = { - maxTokens: 32000, - maxMessages: 50, - timeWindow: 24 * 60 * 60 * 1000, // 24小时 - priority: 1, - includeTools: true, - includeWorkspace: true, - ...defaultOptions, - }; - } - - /** - * 过滤上下文数据 - */ - filter(contextData: ContextData, options?: FilterOptions): ContextData { - const filterOptions = { ...this.defaultOptions, ...options }; - - const filteredData: ContextData = { - layers: { - system: contextData.layers.system, - session: contextData.layers.session, - conversation: this.filterConversation( - contextData.layers.conversation, - filterOptions - ), - tool: filterOptions.includeTools - ? this.filterTools(contextData.layers.tool, filterOptions) - : { recentCalls: [], toolStates: {}, dependencies: {} }, - workspace: filterOptions.includeWorkspace - ? contextData.layers.workspace - : { currentFiles: [], recentFiles: [], environment: {} }, - }, - metadata: { - ...contextData.metadata, - lastUpdated: Date.now(), - }, - }; - - // 重新计算 token 数量 - filteredData.metadata.totalTokens = this.estimateTokens(filteredData); - - return filteredData; - } - - /** - * 过滤对话上下文 - */ - private filterConversation( - conversation: ContextData['layers']['conversation'], - options: Required - ): ContextData['layers']['conversation'] { - let filteredMessages = [...conversation.messages]; - - // 时间窗口过滤 - if (options.timeWindow > 0) { - const cutoffTime = Date.now() - options.timeWindow; - filteredMessages = filteredMessages.filter( - (msg) => msg.timestamp >= cutoffTime || msg.role === 'system' - ); - } - - // 优先级过滤 - if (options.priority > 1) { - filteredMessages = this.filterByPriority(filteredMessages, options.priority); - } - - // 消息数量限制 - if (options.maxMessages > 0) { - filteredMessages = this.limitMessages(filteredMessages, options.maxMessages); - } - - // Token 数量限制 - if (options.maxTokens > 0) { - filteredMessages = this.limitByTokens(filteredMessages, options.maxTokens); - } - - return { - messages: filteredMessages, - summary: conversation.summary, - topics: this.updateTopics(filteredMessages, conversation.topics), - lastActivity: conversation.lastActivity, - }; - } - - /** - * 过滤工具上下文 - */ - private filterTools( - toolContext: ContextData['layers']['tool'], - options: Required - ): ContextData['layers']['tool'] { - let filteredCalls = [...toolContext.recentCalls]; - - // 时间窗口过滤 - if (options.timeWindow > 0) { - const cutoffTime = Date.now() - options.timeWindow; - filteredCalls = filteredCalls.filter((call) => call.timestamp >= cutoffTime); - } - - // 保留最近的成功调用和失败调用(用于学习) - const successCalls = filteredCalls.filter((call) => call.status === 'success'); - const failedCalls = filteredCalls.filter((call) => call.status === 'error'); - - // 限制每种状态的调用数量 - const maxSuccessfulCalls = Math.min(20, successCalls.length); - const maxFailedCalls = Math.min(10, failedCalls.length); - - const limitedCalls = [ - ...successCalls.slice(-maxSuccessfulCalls), - ...failedCalls.slice(-maxFailedCalls), - ].sort((a, b) => a.timestamp - b.timestamp); - - return { - recentCalls: limitedCalls, - toolStates: toolContext.toolStates, - dependencies: toolContext.dependencies, - }; - } - - /** - * 按优先级过滤消息 - */ - private filterByPriority( - messages: ContextMessage[], - minPriority: number - ): ContextMessage[] { - return messages.filter((msg) => { - // 系统消息始终保留 - if (msg.role === 'system') return true; - - // 计算消息优先级 - const priority = this.calculateMessagePriority(msg); - return priority >= minPriority; - }); - } - - /** - * 计算消息优先级 - */ - private calculateMessagePriority(message: ContextMessage): number { - let priority = 1; - - // 基于角色的基础分数 - if (message.role === 'system') priority += 3; - else if (message.role === 'assistant') priority += 1; - - // 基于内容的分数 - const content = message.content.toLowerCase(); - - // 包含重要关键词 - const importantKeywords = ['错误', '警告', '重要', '关键', '问题', '解决']; - if (importantKeywords.some((keyword) => content.includes(keyword))) { - priority += 2; - } - - // 包含代码或技术内容 - if ( - content.includes('```') || - content.includes('function') || - content.includes('class') - ) { - priority += 1; - } - - // 基于时间的衰减(最近的消息优先级更高) - const ageInHours = (Date.now() - message.timestamp) / (60 * 60 * 1000); - if (ageInHours < 1) priority += 2; - else if (ageInHours < 6) priority += 1; - - return priority; - } - - /** - * 限制消息数量 - */ - private limitMessages( - messages: ContextMessage[], - maxMessages: number - ): ContextMessage[] { - if (messages.length <= maxMessages) { - return messages; - } - - // 分离系统消息和其他消息 - const systemMessages = messages.filter((msg) => msg.role === 'system'); - const otherMessages = messages.filter((msg) => msg.role !== 'system'); - - // 保留系统消息和最近的其他消息 - const remainingSlots = maxMessages - systemMessages.length; - const limitedOtherMessages = - remainingSlots > 0 ? otherMessages.slice(-remainingSlots) : []; - - return [...systemMessages, ...limitedOtherMessages].sort( - (a, b) => a.timestamp - b.timestamp - ); - } - - /** - * 按 Token 数量限制消息 - */ - private limitByTokens( - messages: ContextMessage[], - maxTokens: number - ): ContextMessage[] { - if (maxTokens <= 0) return messages; - - let totalTokens = 0; - const result: ContextMessage[] = []; - - // 从最新消息开始向前计算 - for (let i = messages.length - 1; i >= 0; i--) { - const message = messages[i]; - const messageTokens = this.estimateMessageTokens(message); - - if (message.role === 'system') { - // 系统消息必须包含,如果空间不够则压缩 - if (totalTokens + messageTokens <= maxTokens) { - result.unshift(message); - totalTokens += messageTokens; - } else { - const compressedMessage = this.compressMessage( - message, - maxTokens - totalTokens - ); - result.unshift(compressedMessage); - totalTokens += this.estimateMessageTokens(compressedMessage); - } - } else if (totalTokens + messageTokens <= maxTokens) { - result.unshift(message); - totalTokens += messageTokens; - } else { - break; - } - } - - return result.sort((a, b) => a.timestamp - b.timestamp); - } - - /** - * 估算消息的 Token 数量 - */ - private estimateMessageTokens(message: ContextMessage): number { - // 简单估算:4个字符约等于1个token(中文) - return Math.ceil(message.content.length / 4); - } - - /** - * 压缩消息内容 - */ - private compressMessage(message: ContextMessage, maxTokens: number): ContextMessage { - const maxLength = maxTokens * 4; // 粗略换算为字符数 - - if (message.content.length <= maxLength) { - return message; - } - - const compressed = message.content.substring(0, maxLength - 3) + '...'; - - return { - ...message, - content: compressed, - metadata: { - ...message.metadata, - compressed: true, - originalLength: message.content.length, - }, - }; - } - - /** - * 更新主题列表 - */ - private updateTopics(messages: ContextMessage[], originalTopics: string[]): string[] { - const topics = new Set(originalTopics); - - // 从过滤后的消息中提取新主题 - for (const message of messages) { - const extractedTopics = this.extractTopicsFromMessage(message); - extractedTopics.forEach((topic) => topics.add(topic)); - } - - return Array.from(topics).slice(0, 10); // 最多保留10个主题 - } - - /** - * 从消息中提取主题 - */ - private extractTopicsFromMessage(message: ContextMessage): string[] { - const content = message.content.toLowerCase(); - const topics: string[] = []; - - // 简单的主题提取逻辑 - const topicKeywords = [ - '项目', - '功能', - '模块', - '组件', - '服务', - '接口', - '数据库', - '前端', - '后端', - '算法', - '架构', - '设计', - ]; - - topicKeywords.forEach((keyword) => { - if (content.includes(keyword)) { - topics.push(keyword); - } - }); - - return topics; - } - - /** - * 估算上下文数据的总 Token 数量 - */ - private estimateTokens(contextData: ContextData): number { - let totalTokens = 0; - - // 对话消息 - for (const message of contextData.layers.conversation.messages) { - totalTokens += this.estimateMessageTokens(message); - } - - // 系统上下文 - const systemContent = JSON.stringify(contextData.layers.system); - totalTokens += Math.ceil(systemContent.length / 4); - - // 工具上下文 - if (contextData.layers.tool.recentCalls.length > 0) { - const toolContent = JSON.stringify(contextData.layers.tool); - totalTokens += Math.ceil(toolContent.length / 8); // 工具调用数据通常更简洁 - } - - return totalTokens; - } - - /** - * 创建预设过滤器 - */ - static createPresets() { - return { - // 轻量级过滤器 - 适合快速响应 - lightweight: new ContextFilter({ - maxTokens: 1000, - maxMessages: 10, - timeWindow: 2 * 60 * 60 * 1000, // 2小时 - includeTools: false, - includeWorkspace: false, - }), - - // 标准过滤器 - 平衡性能和功能 - standard: new ContextFilter({ - maxTokens: 4000, - maxMessages: 30, - timeWindow: 12 * 60 * 60 * 1000, // 12小时 - includeTools: true, - includeWorkspace: true, - }), - - // 完整过滤器 - 包含所有上下文 - comprehensive: new ContextFilter({ - maxTokens: 8000, - maxMessages: 100, - timeWindow: 24 * 60 * 60 * 1000, // 24小时 - includeTools: true, - includeWorkspace: true, - }), - - // 调试过滤器 - 专注于错误和工具调用 - debug: new ContextFilter({ - maxTokens: 2000, - maxMessages: 20, - timeWindow: 6 * 60 * 60 * 1000, // 6小时 - priority: 2, // 只包含高优先级消息 - includeTools: true, - includeWorkspace: false, - }), - }; - } -} diff --git a/packages/cli/src/context/storage/CacheStore.ts b/packages/cli/src/context/storage/CacheStore.ts deleted file mode 100644 index 99576c225..000000000 --- a/packages/cli/src/context/storage/CacheStore.ts +++ /dev/null @@ -1,319 +0,0 @@ -import type { CompressedContext, ContextMessage } from '../types.js'; - -export interface CacheItem { - data: T; - timestamp: number; - accessCount: number; - lastAccess: number; - ttl: number; // Time to live in milliseconds -} - -/** - * LRU缓存实现 - 用于热点数据的快速访问 - */ -export class CacheStore { - private readonly cache: Map> = new Map(); - private readonly maxSize: number; - private readonly defaultTTL: number; - - constructor(maxSize: number = 100, defaultTTL: number = 5 * 60 * 1000) { - // 默认5分钟TTL - this.maxSize = maxSize; - this.defaultTTL = defaultTTL; - } - - /** - * 设置缓存项 - */ - set(key: string, data: T, ttl?: number): void { - const now = Date.now(); - const item: CacheItem = { - data, - timestamp: now, - accessCount: 0, - lastAccess: now, - ttl: ttl || this.defaultTTL, - }; - - // 如果缓存已满,删除最不常用的项 - if (this.cache.size >= this.maxSize && !this.cache.has(key)) { - this.evictLeastUsed(); - } - - this.cache.set(key, item); - } - - /** - * 获取缓存项 - */ - get(key: string): T | null { - const item = this.cache.get(key) as CacheItem | undefined; - - if (!item) { - return null; - } - - const now = Date.now(); - - // 检查是否过期 - if (now - item.timestamp > item.ttl) { - this.cache.delete(key); - return null; - } - - // 更新访问统计 - item.accessCount++; - item.lastAccess = now; - - return item.data; - } - - /** - * 检查缓存项是否存在 - */ - has(key: string): boolean { - const item = this.cache.get(key); - - if (!item) { - return false; - } - - // 检查是否过期 - if (Date.now() - item.timestamp > item.ttl) { - this.cache.delete(key); - return false; - } - - return true; - } - - /** - * 删除缓存项 - */ - delete(key: string): boolean { - return this.cache.delete(key); - } - - /** - * 清空缓存 - */ - clear(): void { - this.cache.clear(); - } - - /** - * 获取缓存大小 - */ - size(): number { - this.cleanExpired(); - return this.cache.size; - } - - /** - * 缓存消息摘要 - */ - cacheMessageSummary( - sessionId: string, - messages: ContextMessage[], - summary: string - ): void { - const key = `summary:${sessionId}:${messages.length}`; - this.set( - key, - { - summary, - messageCount: messages.length, - lastMessage: messages[messages.length - 1]?.timestamp || 0, - }, - 10 * 60 * 1000 - ); // 10分钟TTL - } - - /** - * 获取缓存的消息摘要 - */ - getMessageSummary( - sessionId: string, - messageCount: number - ): { - summary: string; - messageCount: number; - lastMessage: number; - } | null { - const key = `summary:${sessionId}:${messageCount}`; - return this.get(key); - } - - /** - * 缓存工具调用结果 - */ - cacheToolResult(toolName: string, input: unknown, result: unknown): void { - const inputHash = this.hashInput(input); - const key = `tool:${toolName}:${inputHash}`; - this.set(key, result, 30 * 60 * 1000); // 30分钟TTL - } - - /** - * 获取缓存的工具调用结果 - */ - getToolResult(toolName: string, input: unknown): unknown | null { - const inputHash = this.hashInput(input); - const key = `tool:${toolName}:${inputHash}`; - return this.get(key); - } - - /** - * 缓存上下文压缩结果 - */ - cacheCompressedContext(contextHash: string, compressed: CompressedContext): void { - const key = `compressed:${contextHash}`; - this.set(key, compressed, 15 * 60 * 1000); // 15分钟TTL - } - - /** - * 获取缓存的压缩上下文 - */ - getCompressedContext(contextHash: string): CompressedContext | null { - const key = `compressed:${contextHash}`; - return this.get(key); - } - - /** - * 获取缓存统计信息 - */ - getStats(): { - size: number; - maxSize: number; - hitRate: number; - memoryUsage: number; - topKeys: { key: string; accessCount: number; lastAccess: number }[]; - } { - this.cleanExpired(); - - let totalAccess = 0; - let memoryUsage = 0; - const keyStats: { key: string; accessCount: number; lastAccess: number }[] = []; - - for (const [key, item] of Array.from(this.cache.entries())) { - totalAccess += item.accessCount; - memoryUsage += this.estimateItemSize(item); - keyStats.push({ - key, - accessCount: item.accessCount, - lastAccess: item.lastAccess, - }); - } - - keyStats.sort((a, b) => b.accessCount - a.accessCount); - - return { - size: this.cache.size, - maxSize: this.maxSize, - hitRate: totalAccess > 0 ? totalAccess / (totalAccess + this.cache.size) : 0, - memoryUsage, - topKeys: keyStats.slice(0, 10), // 返回前10个最常访问的键 - }; - } - - /** - * 清理过期项 - */ - private cleanExpired(): void { - const now = Date.now(); - const expiredKeys: string[] = []; - - for (const [key, item] of Array.from(this.cache.entries())) { - if (now - item.timestamp > item.ttl) { - expiredKeys.push(key); - } - } - - expiredKeys.forEach((key) => this.cache.delete(key)); - } - - /** - * 驱逐最不常用的项 - */ - private evictLeastUsed(): void { - let leastUsedKey: string | null = null; - let leastScore = Infinity; - - const now = Date.now(); - - for (const [key, item] of Array.from(this.cache.entries())) { - // 计算使用分数(考虑访问次数和最后访问时间) - const recencyScore = 1 / (now - item.lastAccess + 1); - const frequencyScore = item.accessCount; - const score = recencyScore * frequencyScore; - - if (score < leastScore) { - leastScore = score; - leastUsedKey = key; - } - } - - if (leastUsedKey) { - this.cache.delete(leastUsedKey); - } - } - - /** - * 简单的输入哈希函数 - */ - private hashInput(input: unknown): string { - const str = JSON.stringify(input); - let hash = 0; - for (let i = 0; i < str.length; i++) { - const char = str.charCodeAt(i); - hash = (hash << 5) - hash + char; - hash = hash & hash; // Convert to 32-bit integer - } - return Math.abs(hash).toString(36); - } - - /** - * 估算缓存项大小 - */ - private estimateItemSize(item: CacheItem): number { - try { - return JSON.stringify(item).length * 2; // 大概估算字节数 - } catch { - return 1000; // 默认估算 - } - } - - /** - * 设置缓存项的TTL - */ - setTTL(key: string, ttl: number): boolean { - const item = this.cache.get(key); - if (item) { - item.ttl = ttl; - item.timestamp = Date.now(); // 重置时间戳 - return true; - } - return false; - } - - /** - * 获取缓存项的剩余TTL - */ - getRemainingTTL(key: string): number { - const item = this.cache.get(key); - if (!item) { - return -1; - } - - const remaining = item.ttl - (Date.now() - item.timestamp); - return Math.max(0, remaining); - } - - /** - * 预热缓存(可用于启动时加载常用数据) - */ - warmup(data: { key: string; value: unknown; ttl?: number }[]): void { - data.forEach(({ key, value, ttl }) => { - this.set(key, value, ttl); - }); - } -} diff --git a/packages/cli/src/context/storage/JSONLStore.ts b/packages/cli/src/context/storage/JSONLStore.ts index 32d1e2439..469954d5d 100644 --- a/packages/cli/src/context/storage/JSONLStore.ts +++ b/packages/cli/src/context/storage/JSONLStore.ts @@ -1,8 +1,6 @@ import * as fsSync from 'node:fs'; -import { createReadStream } from 'node:fs'; import * as fs from 'node:fs/promises'; import * as path from 'node:path'; -import { createInterface } from 'node:readline'; import type { SessionEvent } from '../types.js'; const TAIL_SCAN_CHUNK_SIZE = 64 * 1024; @@ -78,9 +76,7 @@ export function parseSessionJSONL( return entries; } -/** - * JSONL 存储类 - 处理 JSONL 格式的读写 - */ +/** JSONL 存储类 - 处理 JSONL 格式的读写 */ export class JSONLStore { private static readonly appendQueues = new Map>(); private readonly filePath: string; @@ -221,73 +217,6 @@ export class JSONLStore { return parseSessionJSONL(content, this.filePath); } - /** - * 流式读取 JSONL 记录(适合大文件) - * @param callback 每条记录的回调函数 - */ - async readStream( - callback: (entry: SessionEvent) => void | Promise - ): Promise { - return new Promise((resolve, reject) => { - if (!fsSync.existsSync(this.filePath)) { - resolve(); - return; - } - - const fileStream = createReadStream(this.filePath, 'utf-8'); - const rl = createInterface({ - input: fileStream, - crlfDelay: Number.POSITIVE_INFINITY, - }); - - let lineOrdinal = 0; - rl.on('line', async (line) => { - const trimmed = line.trim(); - if (trimmed.length === 0) return; - - try { - const entry = JSON.parse(trimmed) as SessionEvent; - lineOrdinal += 1; - if (typeof entry.seq !== 'number') { - entry.seq = lineOrdinal; - } - await callback(entry); - } catch (error) { - console.warn(`[JSONLStore] 解析 JSON 行失败: ${trimmed}`, error); - } - }); - - rl.on('close', () => resolve()); - rl.on('error', reject); - fileStream.on('error', reject); - }); - } - - /** - * 按条件过滤读取 JSONL 记录 - * @param predicate 过滤条件 - * @returns 符合条件的 JSONL 条目数组 - */ - async filter(predicate: (entry: SessionEvent) => boolean): Promise { - const results: SessionEvent[] = []; - await this.readStream((entry) => { - if (predicate(entry)) { - results.push(entry); - } - }); - return results; - } - - /** - * 获取最后 N 条记录 - * @param count 记录数量 - * @returns JSONL 条目数组 - */ - async readLast(count: number): Promise { - const all = await this.readAll(); - return all.slice(-count); - } - /** * 读取 seq >= fromSeq 的所有记录,用于 Last-Event-ID 断点续传的 JSONL 兜底补发。 * seq 由 {@link parseSessionJSONL} 统一保证(新事件显式携带,旧事件按行号回填)。 @@ -298,53 +227,7 @@ export class JSONLStore { return all.filter((entry) => (entry.seq ?? 0) >= fromSeq); } - /** - * 获取文件统计信息 - * @returns 统计信息 - */ - async getStats(): Promise<{ - exists: boolean; - size: number; // 字节 - lineCount: number; - }> { - try { - if (!fsSync.existsSync(this.filePath)) { - return { exists: false, size: 0, lineCount: 0 }; - } - - const stats = await fs.stat(this.filePath); - const content = await fs.readFile(this.filePath, 'utf-8'); - const lineCount = content - .split('\n') - .filter((line) => line.trim().length > 0).length; - - return { - exists: true, - size: stats.size, - lineCount, - }; - } catch (error) { - console.error(`[JSONLStore] 获取统计信息失败: ${this.filePath}`, error); - return { exists: false, size: 0, lineCount: 0 }; - } - } - - /** - * 检查文件是否存在 - * @returns 文件是否存在 - */ - async exists(): Promise { - try { - await fs.access(this.filePath); - return true; - } catch { - return false; - } - } - - /** - * 删除 JSONL 文件 - */ + /** 删除 JSONL 文件 */ async delete(): Promise { try { return await this.enqueue(async () => { @@ -506,9 +389,7 @@ export class JSONLStore { } } - /** - * 获取文件路径 - */ + /** 获取文件路径 */ getFilePath(): string { return this.filePath; } diff --git a/packages/cli/src/context/storage/MemoryStore.ts b/packages/cli/src/context/storage/MemoryStore.ts deleted file mode 100644 index 8399e281e..000000000 --- a/packages/cli/src/context/storage/MemoryStore.ts +++ /dev/null @@ -1,187 +0,0 @@ -import type { JsonValue } from '../../store/types.js'; -import { ContextData, ContextMessage, ToolCall, WorkspaceContext } from '../types.js'; - -/** - * 内存存储实现 - 用于当前会话的快速数据访问 - */ -export class MemoryStore { - private contextData: ContextData | null = null; - private readonly maxSize: number; - private readonly accessLog: Map = new Map(); - - constructor(maxSize: number = 1000) { - this.maxSize = maxSize; - } - - /** - * 存储上下文数据 - */ - setContext(data: ContextData): void { - this.contextData = { ...data }; - this.contextData.metadata.lastUpdated = Date.now(); - this.recordAccess('context'); - } - - /** - * 获取完整上下文数据 - */ - getContext(): ContextData | null { - if (this.contextData) { - this.recordAccess('context'); - } - return this.contextData; - } - - /** - * 添加消息到对话上下文 - */ - addMessage(message: ContextMessage): void { - if (!this.contextData) { - throw new Error('上下文数据未初始化'); - } - - this.contextData.layers.conversation.messages.push(message); - this.contextData.layers.conversation.lastActivity = Date.now(); - this.contextData.metadata.lastUpdated = Date.now(); - - // 检查是否超过大小限制 - this.enforceMemoryLimit(); - this.recordAccess('messages'); - } - - /** - * 获取最近的消息 - */ - getRecentMessages(count: number = 10): ContextMessage[] { - if (!this.contextData) { - return []; - } - - const messages = this.contextData.layers.conversation.messages; - this.recordAccess('messages'); - return messages.slice(-count); - } - - /** - * 添加工具调用记录 - */ - addToolCall(toolCall: ToolCall): void { - if (!this.contextData) { - throw new Error('上下文数据未初始化'); - } - - this.contextData.layers.tool.recentCalls.push(toolCall); - this.contextData.metadata.lastUpdated = Date.now(); - - // 保持工具调用历史在合理范围内 - if (this.contextData.layers.tool.recentCalls.length > 50) { - this.contextData.layers.tool.recentCalls = - this.contextData.layers.tool.recentCalls.slice(-25); - } - - this.recordAccess('tools'); - } - - /** - * 更新工具状态 - */ - updateToolState(toolName: string, state: JsonValue): void { - if (!this.contextData) { - throw new Error('上下文数据未初始化'); - } - - this.contextData.layers.tool.toolStates[toolName] = state; - this.contextData.metadata.lastUpdated = Date.now(); - this.recordAccess('tools'); - } - - /** - * 获取工具状态 - */ - getToolState(toolName: string): JsonValue | null { - if (!this.contextData) { - return null; - } - - this.recordAccess('tools'); - return this.contextData.layers.tool.toolStates[toolName]; - } - - /** - * 更新工作空间信息 - */ - updateWorkspace(updates: Partial): void { - if (!this.contextData) { - throw new Error('上下文数据未初始化'); - } - - Object.assign(this.contextData.layers.workspace, updates); - this.contextData.metadata.lastUpdated = Date.now(); - this.recordAccess('workspace'); - } - - /** - * 清除内存数据 - */ - clear(): void { - this.contextData = null; - this.accessLog.clear(); - } - - /** - * 获取内存使用情况 - */ - getMemoryInfo(): { - hasData: boolean; - messageCount: number; - toolCallCount: number; - lastUpdated: number | null; - } { - if (!this.contextData) { - return { - hasData: false, - messageCount: 0, - toolCallCount: 0, - lastUpdated: null, - }; - } - - return { - hasData: true, - messageCount: this.contextData.layers.conversation.messages.length, - toolCallCount: this.contextData.layers.tool.recentCalls.length, - lastUpdated: this.contextData.metadata.lastUpdated, - }; - } - - /** - * 记录访问日志 - */ - private recordAccess(key: string): void { - this.accessLog.set(key, Date.now()); - } - - /** - * 强制执行内存限制 - */ - private enforceMemoryLimit(): void { - if (!this.contextData) return; - - const messages = this.contextData.layers.conversation.messages; - if (messages.length > this.maxSize) { - // 保留最近的消息,删除较旧的 - const keepCount = Math.floor(this.maxSize * 0.8); // 保留80%的空间 - this.contextData.layers.conversation.messages = messages.slice(-keepCount); - } - } - - /** - * 估算内存使用量(以字符数为简单估算) - */ - getMemoryUsage(): number { - if (!this.contextData) return 0; - - const contextString = JSON.stringify(this.contextData); - return contextString.length; - } -} diff --git a/packages/cli/src/context/storage/PersistentStore.ts b/packages/cli/src/context/storage/PersistentStore.ts index 88f35ea66..294ddd8be 100644 --- a/packages/cli/src/context/storage/PersistentStore.ts +++ b/packages/cli/src/context/storage/PersistentStore.ts @@ -1,5 +1,4 @@ import * as fs from 'node:fs/promises'; -import * as path from 'node:path'; import { nanoid } from 'nanoid'; import type { BackgroundSubagentCompletion } from '../../agent/subagents/BackgroundSubagentCompletion.js'; import type { SubagentInfoForContext } from '../../agent/types.js'; @@ -23,14 +22,12 @@ import { type ValidTokenBudgetHandoffEvent, } from '../TokenBudgetHandoff.js'; import { - type ConversationContext, MAX_TURN_INPUT_MESSAGE_ID_CHARS, MAX_TURN_INPUT_MESSAGE_IDS, type MessageInfo, type MessagePersistenceMetadata, type PartInfo, parseTurnInputMessageIds, - type SessionContext, type SessionEvent, type SessionGoalFinalizationInfo, type SessionInfo, @@ -50,16 +47,13 @@ import { JSONLStore } from './JSONLStore.js'; import { detectGitBranch, getBladeStorageRoot, - getProjectStoragePath, getSessionFilePath, getSessionInboxFilePath, - listProjectDirectories, } from './pathUtils.js'; import { createSessionStateStorage, type SessionStateStorage, sessionStateStorageKey, - withSessionStatePaths, withSessionStateRoot, } from './SessionStateStorage.js'; @@ -656,7 +650,6 @@ export class PersistentStore { private static readonly sessionInitializationRuns = new Map>(); private readonly projectPath: string; - private readonly maxSessions: number; private readonly version: string; private readonly stateStorage: SessionStateStorage; /** Positive per-facade cache; Runtime ownership prevents active-file deletion. */ @@ -664,12 +657,10 @@ export class PersistentStore { constructor( projectPath: string = getCwd(), - maxSessions: number = 100, version: string = getVersion(), stateStorage: SessionStateStorage = createSessionStateStorage(projectPath) ) { this.projectPath = projectPath; - this.maxSessions = maxSessions; this.version = version; this.stateStorage = stateStorage; } @@ -711,7 +702,7 @@ export class PersistentStore { let initialization = PersistentStore.sessionInitializationRuns.get(filePath); if (!initialization) { - initialization = this.initializeSessionFile(sessionId, filePath, subagentInfo); + initialization = this.initializeSessionFile(sessionId, subagentInfo); PersistentStore.sessionInitializationRuns.set(filePath, initialization); } @@ -737,7 +728,6 @@ export class PersistentStore { private async initializeSessionFile( sessionId: string, - filePath: string, subagentInfo?: SubagentInfoForContext ): Promise { const entries = await this.log(sessionId).readAll(); @@ -814,9 +804,7 @@ export class PersistentStore { return result; } - /** - * 初始化存储目录 - */ + /** 初始化存储目录 */ async initialize(): Promise { try { await withSessionStateRoot(this.stateStorage, async (storagePath) => { @@ -829,9 +817,7 @@ export class PersistentStore { } } - /** - * 保存消息到 JSONL 文件(追加模式) - */ + /** 保存消息到 JSONL 文件(追加模式) */ async saveMessage( sessionId: string, messageRole: MessageRole, @@ -1824,9 +1810,7 @@ export class PersistentStore { }); } - /** - * 保存工具调用到 JSONL 文件 - */ + /** 保存工具调用到 JSONL 文件 */ async saveToolUse( sessionId: string, toolName: string, @@ -1905,9 +1889,7 @@ export class PersistentStore { } } - /** - * 保存工具结果到 JSONL 文件 - */ + /** 保存工具结果到 JSONL 文件 */ async saveToolResult( sessionId: string, toolId: string, @@ -2115,10 +2097,7 @@ export class PersistentStore { return { outcome: 'created', event: event.event }; } - /** - * 保存会话初始化事件到 JSONL - * 仅创建 session_created 事件,不写入空消息 - */ + /** 保存会话初始化事件到 JSONL 仅创建 session_created 事件,不写入空消息 */ async initSession( sessionId: string, subagentInfo?: SubagentInfoForContext @@ -2126,9 +2105,7 @@ export class PersistentStore { await this.ensureSessionCreated(sessionId, subagentInfo); } - /** - * 加载会话的原始 JSONL 事件流 - */ + /** 加载会话的原始 JSONL 事件流 */ async loadEvents(sessionId: string): Promise { try { const entries = await this.log(sessionId).readAll(); @@ -2138,140 +2115,7 @@ export class PersistentStore { } } - /** - * 加载会话上下文(从 JSONL 重建) - */ - async loadSession(sessionId: string): Promise { - try { - const entries = materializeSessionEvents(await this.log(sessionId).readAll()); - if (entries.length === 0) return null; - const firstEntry = entries.find((entry) => entry.type === 'session_created'); - - return { - sessionId, - userId: undefined, - preferences: {}, - configuration: {}, - startTime: new Date(firstEntry?.timestamp ?? entries[0].timestamp).getTime(), - }; - } catch { - return null; - } - } - - /** - * 加载对话上下文(从 JSONL 重建) - */ - async loadConversation(sessionId: string): Promise { - try { - const entries = materializeSessionEvents(await this.log(sessionId).readAll()); - if (entries.length === 0) return null; - const messageMap = new Map< - string, - { id: string; role: MessageRole; content: string; timestamp: number } - >(); - for (const entry of entries) { - if (entry.type === 'message_created') { - messageMap.set(entry.data.messageId, { - id: entry.data.messageId, - role: entry.data.role, - content: '', - timestamp: new Date(entry.timestamp).getTime(), - }); - } - if (entry.type === 'part_created' && entry.data.partType === 'text') { - const message = messageMap.get(entry.data.messageId); - if (message) { - const payload = entry.data.payload as { text?: string }; - message.content = payload.text ?? ''; - } - } - if (entry.type === 'part_created' && entry.data.partType === 'image') { - const message = messageMap.get(entry.data.messageId); - if (message) { - message.content = message.content - ? `${message.content}\n[Image]` - : '[Image]'; - } - } - } - const messages = Array.from(messageMap.values()); - const lastEntry = entries[entries.length - 1]; - const lastActivity = new Date(lastEntry.timestamp).getTime(); - - return { - messages, - topics: [], - lastActivity, - }; - } catch { - return null; - } - } - - /** - * 获取所有会话列表 - */ - async listSessions(): Promise { - if (this.stateStorage.kind === 'acp-remote') { - throw new Error('Remote session enumeration requires SessionService'); - } - try { - const storagePath = getProjectStoragePath(this.projectPath); - const files = await fs.readdir(storagePath); - return files - .filter((file) => file.endsWith('.jsonl')) - .map((file) => file.replace('.jsonl', '')) - .sort(); - } catch { - return []; - } - } - - /** - * 获取会话摘要信息 - */ - async getSessionSummary(sessionId: string): Promise<{ - sessionId: string; - lastActivity: number; - messageCount: number; - topics: string[]; - } | null> { - if (this.stateStorage.kind === 'acp-remote') { - throw new Error('Remote session summaries require SessionService'); - } - try { - const filePath = getSessionFilePath(this.projectPath, sessionId); - const store = new JSONLStore(filePath); - - const stats = await store.getStats(); - if (!stats.exists) return null; - - const rawEntries = await store.readAll(); - if (rawEntries.length === 0) return null; - const entries = materializeSessionEvents(rawEntries); - - const lastEntry = rawEntries[rawEntries.length - 1]; - const messageCount = entries.filter( - (entry) => - entry.type === 'message_created' && - ['user', 'assistant'].includes(entry.data.role) - ).length; - - return { - sessionId, - lastActivity: new Date(lastEntry.timestamp).getTime(), - messageCount, - topics: [], - }; - } catch { - return null; - } - } - - /** - * 删除会话数据 - */ + /** 删除会话数据 */ async deleteSession(sessionId: string): Promise { if (this.stateStorage.kind === 'acp-remote') { throw new Error('Remote session deletion requires SessionService'); @@ -2307,121 +2151,6 @@ export class PersistentStore { this.initializedSessions.delete(sessionId); } } - - /** - * 清理旧会话(保持最近的N个会话) - */ - async cleanupOldSessions(): Promise { - if (this.stateStorage.kind === 'acp-remote') { - throw new Error('Remote session cleanup requires SessionService'); - } - try { - const sessions = await this.listSessions(); - if (sessions.length <= this.maxSessions) { - return; - } - - // 获取所有会话的摘要信息并按时间排序 - const sessionSummaries = await Promise.all( - sessions.map((sessionId) => this.getSessionSummary(sessionId)) - ); - - const validSummaries = sessionSummaries - .filter((summary): summary is NonNullable => summary !== null) - .sort((a, b) => b.lastActivity - a.lastActivity); - - // 删除最旧的会话 - const sessionsToDelete = validSummaries - .slice(this.maxSessions) - .map((summary) => summary.sessionId); - - await Promise.all( - sessionsToDelete.map((sessionId) => this.deleteSession(sessionId)) - ); - - console.log(`[PersistentStore] 已清理 ${sessionsToDelete.length} 个旧会话`); - } catch (error) { - console.error('[PersistentStore] 清理旧会话失败:', error); - } - } - - /** - * 获取存储统计信息 - */ - async getStorageStats(): Promise<{ - totalSessions: number; - totalSize: number; - projectPath: string; - }> { - if (this.stateStorage.kind === 'acp-remote') { - throw new Error('Remote storage statistics require SessionService'); - } - try { - const sessions = await this.listSessions(); - let totalSize = 0; - - for (const sessionId of sessions) { - const filePath = getSessionFilePath(this.projectPath, sessionId); - const store = new JSONLStore(filePath); - const stats = await store.getStats(); - totalSize += stats.size; - } - - return { - totalSessions: sessions.length, - totalSize, - projectPath: this.projectPath, - }; - } catch { - return { - totalSessions: 0, - totalSize: 0, - projectPath: this.projectPath, - }; - } - } - - /** - * 检查存储健康状态 - */ - async checkStorageHealth(): Promise<{ - isAvailable: boolean; - canWrite: boolean; - error?: string; - }> { - if (this.stateStorage.kind === 'acp-remote') { - throw new Error('Remote storage health checks require SessionService'); - } - try { - const storagePath = getProjectStoragePath(this.projectPath); - - // 尝试创建目录 - await fs.mkdir(storagePath, { recursive: true, mode: 0o755 }); - - // 尝试写入测试文件 - const testFile = path.join(storagePath, '.health-check'); - await fs.writeFile(testFile, 'test', 'utf-8'); - await fs.unlink(testFile); - - return { - isAvailable: true, - canWrite: true, - }; - } catch (error) { - return { - isAvailable: false, - canWrite: false, - error: error instanceof Error ? error.message : String(error), - }; - } - } - - /** - * 获取所有项目列表 - */ - static async listAllProjects(): Promise { - return listProjectDirectories(); - } } function extractMimeTypeFromDataUrl(dataUrl: string): string | null { diff --git a/packages/cli/src/context/storage/pathUtils.ts b/packages/cli/src/context/storage/pathUtils.ts index 306b9755a..00ac645bf 100644 --- a/packages/cli/src/context/storage/pathUtils.ts +++ b/packages/cli/src/context/storage/pathUtils.ts @@ -12,9 +12,7 @@ import { getBladeStorageRoot } from './BladeStorageRoot.js'; export { getBladeStorageRoot } from './BladeStorageRoot.js'; -/** - * 路径转义工具 - 将项目路径转为目录名 - */ +/** 路径转义工具 - 将项目路径转为目录名 */ /** * 转义项目路径为目录名 diff --git a/packages/cli/src/context/storage/sqlite/archive-cte.sql b/packages/cli/src/context/storage/sqlite/archive-cte.sql new file mode 100644 index 000000000..fe728ba22 --- /dev/null +++ b/packages/cli/src/context/storage/sqlite/archive-cte.sql @@ -0,0 +1,29 @@ +WITH RECURSIVE archive_members( + source_kind, project_path, public_workspace_ref, session_id, archive_root_id, + effective_archived_at, depth +) AS ( + SELECT source_kind, project_path, public_workspace_ref, session_id, session_id, + archived_at, 0 + FROM sessions + WHERE archived_at IS NOT NULL + UNION ALL + SELECT child.source_kind, child.project_path, child.public_workspace_ref, + child.session_id, parent.archive_root_id, parent.effective_archived_at, + parent.depth + 1 + FROM sessions child + JOIN archive_members parent + ON child.source_kind = parent.source_kind + AND child.project_path = parent.project_path + AND child.public_workspace_ref IS parent.public_workspace_ref + AND child.parent_id = parent.session_id + WHERE parent.depth < 128 +), +ranked_archive AS ( + SELECT source_kind, project_path, public_workspace_ref, session_id, + archive_root_id, effective_archived_at, + ROW_NUMBER() OVER ( + PARTITION BY source_kind, project_path, public_workspace_ref, session_id + ORDER BY depth ASC, archive_root_id ASC + ) AS rank + FROM archive_members +) \ No newline at end of file diff --git a/packages/cli/src/context/storage/sqlite/driver.ts b/packages/cli/src/context/storage/sqlite/driver.ts index 6357d0cf8..0996946ff 100644 --- a/packages/cli/src/context/storage/sqlite/driver.ts +++ b/packages/cli/src/context/storage/sqlite/driver.ts @@ -93,9 +93,7 @@ function initializationPragmas(busyTimeoutMs: number): string { ].join('\n'); } -/** - * 打开(或创建)一个 SQLite 数据库。失败返回 null(调用方回退 JSONL)。 - */ +/** 打开(或创建)一个 SQLite 数据库。失败返回 null(调用方回退 JSONL)。 */ export async function openDb( dbPath: string, options: OpenDbOptions = {} diff --git a/packages/cli/src/context/storage/sqlite/drop-all.sql b/packages/cli/src/context/storage/sqlite/drop-all.sql new file mode 100644 index 000000000..ae9766101 --- /dev/null +++ b/packages/cli/src/context/storage/sqlite/drop-all.sql @@ -0,0 +1,7 @@ + +DROP TABLE IF EXISTS parts_fts; +DROP TABLE IF EXISTS surface_messages; +DROP TABLE IF EXISTS parts; +DROP TABLE IF EXISTS sessions; +DROP TABLE IF EXISTS projection_state; +DROP TABLE IF EXISTS surface_projection_meta; diff --git a/packages/cli/src/context/storage/sqlite/projection.ts b/packages/cli/src/context/storage/sqlite/projection.ts index 62520fc07..6f781799f 100644 --- a/packages/cli/src/context/storage/sqlite/projection.ts +++ b/packages/cli/src/context/storage/sqlite/projection.ts @@ -59,35 +59,8 @@ const DEFAULT_SURFACE_HISTORY_BYTE_LIMIT = 512 * 1024; const MAX_PROJECTION_SNAPSHOT_ATTEMPTS = 3; const NEVER_ABORTED_SIGNAL = new AbortController().signal; const SURFACE_WORKSPACE_REFERENCE_PATTERN = /^acp-remote-workspace:[A-Za-z0-9_-]{43}$/; -const SURFACE_ARCHIVE_CTE = `WITH RECURSIVE archive_members( - source_kind, project_path, public_workspace_ref, session_id, archive_root_id, - effective_archived_at, depth -) AS ( - SELECT source_kind, project_path, public_workspace_ref, session_id, session_id, - archived_at, 0 - FROM sessions - WHERE archived_at IS NOT NULL - UNION ALL - SELECT child.source_kind, child.project_path, child.public_workspace_ref, - child.session_id, parent.archive_root_id, parent.effective_archived_at, - parent.depth + 1 - FROM sessions child - JOIN archive_members parent - ON child.source_kind = parent.source_kind - AND child.project_path = parent.project_path - AND child.public_workspace_ref IS parent.public_workspace_ref - AND child.parent_id = parent.session_id - WHERE parent.depth < 128 -), -ranked_archive AS ( - SELECT source_kind, project_path, public_workspace_ref, session_id, - archive_root_id, effective_archived_at, - ROW_NUMBER() OVER ( - PARTITION BY source_kind, project_path, public_workspace_ref, session_id - ORDER BY depth ASC, archive_root_id ASC - ) AS rank - FROM archive_members -)`; + +import SURFACE_ARCHIVE_CTE from './archive-cte.sql?raw'; interface ProjectionIO { readSession( @@ -258,9 +231,7 @@ function extractSearchText(payload: unknown): string | null { return null; } -/** - * 用一条会话的规范化事件重建 parts / parts_fts 行并写入。调用方保证在事务内。 - */ +/** 用一条会话的规范化事件重建 parts / parts_fts 行并写入。调用方保证在事务内。 */ function writeParts( db: SqliteDb, sourceKind: ProjectionSourceKind, @@ -1084,9 +1055,7 @@ async function syncSessionValidated( return true; } -/** - * 全量同步:枚举所有项目/会话文件逐个 syncSession,并 GC 掉 JSONL 已不存在的行。 - */ +/** 全量同步:枚举所有项目/会话文件逐个 syncSession,并 GC 掉 JSONL 已不存在的行。 */ const syncAllState = new WeakMap< SqliteDb, { inFlight?: Promise; lastCompletedAt?: number } diff --git a/packages/cli/src/context/storage/sqlite/schema.sql b/packages/cli/src/context/storage/sqlite/schema.sql new file mode 100644 index 000000000..c84ce1746 --- /dev/null +++ b/packages/cli/src/context/storage/sqlite/schema.sql @@ -0,0 +1,135 @@ + +CREATE TABLE IF NOT EXISTS surface_projection_meta ( + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + catalog_revision INTEGER NOT NULL DEFAULT 0 CHECK(catalog_revision >= 0) +); +INSERT OR IGNORE INTO surface_projection_meta (singleton, catalog_revision) +VALUES (1, 0); + +CREATE TABLE IF NOT EXISTS projection_state ( + source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), + source_file_path TEXT NOT NULL, + project_path TEXT, + session_id TEXT NOT NULL, + last_seq INTEGER NOT NULL DEFAULT 0, + file_size INTEGER NOT NULL DEFAULT 0, + mtime_ms INTEGER NOT NULL DEFAULT 0, + stat_fingerprint TEXT NOT NULL DEFAULT '', + PRIMARY KEY (source_kind, source_file_path) +); +CREATE INDEX IF NOT EXISTS idx_projection_state_session ON projection_state( + source_kind, project_path, session_id +); + +CREATE TABLE IF NOT EXISTS sessions ( + source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), + source_file_path TEXT NOT NULL, + project_path TEXT NOT NULL, + session_id TEXT NOT NULL, + root_id TEXT, + parent_id TEXT, + relation_type TEXT, + title TEXT, + agent_type TEXT, + model TEXT, + task_status TEXT, + task_priority TEXT, + task_kind TEXT, + task_due_at TEXT, + archived_at TEXT, + last_message_time TEXT, + project_sort_key TEXT NOT NULL, + public_workspace_ref TEXT, + public_workspace_sort_key TEXT NOT NULL, + session_sort_key TEXT NOT NULL, + first_message_time TEXT, + message_count INTEGER NOT NULL DEFAULT 0, + has_errors INTEGER NOT NULL DEFAULT 0, + is_subagent INTEGER NOT NULL DEFAULT 0, + -- 完整 StoredSessionMetadata JSON,读回时直接反序列化,保证与 JSONL 路径一致。 + metadata_json TEXT NOT NULL, + surface_digest TEXT NOT NULL, + PRIMARY KEY (source_kind, project_path, session_id) +); +CREATE INDEX IF NOT EXISTS idx_sessions_source_file ON sessions( + source_kind, source_file_path +); +CREATE INDEX IF NOT EXISTS idx_sessions_catalog ON sessions( + last_message_time DESC, + source_kind DESC, + public_workspace_sort_key ASC, + session_sort_key ASC +); +CREATE INDEX IF NOT EXISTS idx_sessions_archived ON sessions( + source_kind, + archived_at, + project_path, + parent_id +); +-- 任务看板过滤/排序:按项目锁定后再按状态、优先级、截止时间收敛。 +CREATE INDEX IF NOT EXISTS idx_sessions_task_board ON sessions( + project_path, + task_status, + task_priority, + task_due_at, + source_kind +); +CREATE INDEX IF NOT EXISTS idx_sessions_task_status ON sessions( + task_status, + project_path, + source_kind +); +CREATE INDEX IF NOT EXISTS idx_sessions_task_priority ON sessions( + task_priority, + task_due_at, + source_kind +) WHERE task_priority IS NOT NULL; +CREATE INDEX IF NOT EXISTS idx_sessions_task_due_at ON sessions( + task_due_at, + source_kind +) WHERE task_due_at IS NOT NULL; + +CREATE TABLE IF NOT EXISTS parts ( + source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), + project_path TEXT NOT NULL, + session_id TEXT NOT NULL, + part_id TEXT NOT NULL, + message_id TEXT, + part_type TEXT, + role TEXT, + seq INTEGER, + timestamp TEXT, + text TEXT, + PRIMARY KEY (source_kind, project_path, session_id, part_id) +); +CREATE INDEX IF NOT EXISTS idx_parts_session ON parts( + source_kind, project_path, session_id +); + +CREATE TABLE IF NOT EXISTS surface_messages ( + source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), + project_path TEXT NOT NULL, + session_id TEXT NOT NULL, + message_seq INTEGER NOT NULL CHECK(message_seq >= 0), + message_id TEXT NOT NULL, + message_json TEXT NOT NULL, + byte_count INTEGER NOT NULL CHECK(byte_count >= 0), + PRIMARY KEY ( + source_kind, project_path, session_id, message_seq, message_id + ) +); +CREATE INDEX IF NOT EXISTS idx_surface_messages_history ON surface_messages( + source_kind, project_path, session_id, message_seq DESC, message_id DESC +); + +CREATE VIRTUAL TABLE IF NOT EXISTS parts_fts USING fts5( + text, + source_kind UNINDEXED, + project_path UNINDEXED, + session_id UNINDEXED, + part_id UNINDEXED, + role UNINDEXED, + timestamp UNINDEXED, + seq UNINDEXED, + tokenize = 'unicode61' +); diff --git a/packages/cli/src/context/storage/sqlite/schema.ts b/packages/cli/src/context/storage/sqlite/schema.ts index a8bc71d30..9a61fbf3c 100644 --- a/packages/cli/src/context/storage/sqlite/schema.ts +++ b/packages/cli/src/context/storage/sqlite/schema.ts @@ -10,158 +10,11 @@ /** schema 版本;不兼容变更时递增,落后版本直接 drop 重建(缓存可弃)。 */ export const SCHEMA_VERSION = 8; -const DDL = ` -CREATE TABLE IF NOT EXISTS surface_projection_meta ( - singleton INTEGER PRIMARY KEY CHECK(singleton = 1), - catalog_revision INTEGER NOT NULL DEFAULT 0 CHECK(catalog_revision >= 0) -); -INSERT OR IGNORE INTO surface_projection_meta (singleton, catalog_revision) -VALUES (1, 0); - -CREATE TABLE IF NOT EXISTS projection_state ( - source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), - source_file_path TEXT NOT NULL, - project_path TEXT, - session_id TEXT NOT NULL, - last_seq INTEGER NOT NULL DEFAULT 0, - file_size INTEGER NOT NULL DEFAULT 0, - mtime_ms INTEGER NOT NULL DEFAULT 0, - stat_fingerprint TEXT NOT NULL DEFAULT '', - PRIMARY KEY (source_kind, source_file_path) -); -CREATE INDEX IF NOT EXISTS idx_projection_state_session ON projection_state( - source_kind, project_path, session_id -); - -CREATE TABLE IF NOT EXISTS sessions ( - source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), - source_file_path TEXT NOT NULL, - project_path TEXT NOT NULL, - session_id TEXT NOT NULL, - root_id TEXT, - parent_id TEXT, - relation_type TEXT, - title TEXT, - agent_type TEXT, - model TEXT, - task_status TEXT, - task_priority TEXT, - task_kind TEXT, - task_due_at TEXT, - archived_at TEXT, - last_message_time TEXT, - project_sort_key TEXT NOT NULL, - public_workspace_ref TEXT, - public_workspace_sort_key TEXT NOT NULL, - session_sort_key TEXT NOT NULL, - first_message_time TEXT, - message_count INTEGER NOT NULL DEFAULT 0, - has_errors INTEGER NOT NULL DEFAULT 0, - is_subagent INTEGER NOT NULL DEFAULT 0, - -- 完整 StoredSessionMetadata JSON,读回时直接反序列化,保证与 JSONL 路径一致。 - metadata_json TEXT NOT NULL, - surface_digest TEXT NOT NULL, - PRIMARY KEY (source_kind, project_path, session_id) -); -CREATE INDEX IF NOT EXISTS idx_sessions_source_file ON sessions( - source_kind, source_file_path -); -CREATE INDEX IF NOT EXISTS idx_sessions_catalog ON sessions( - last_message_time DESC, - source_kind DESC, - public_workspace_sort_key ASC, - session_sort_key ASC -); -CREATE INDEX IF NOT EXISTS idx_sessions_archived ON sessions( - source_kind, - archived_at, - project_path, - parent_id -); --- 任务看板过滤/排序:按项目锁定后再按状态、优先级、截止时间收敛。 -CREATE INDEX IF NOT EXISTS idx_sessions_task_board ON sessions( - project_path, - task_status, - task_priority, - task_due_at, - source_kind -); -CREATE INDEX IF NOT EXISTS idx_sessions_task_status ON sessions( - task_status, - project_path, - source_kind -); -CREATE INDEX IF NOT EXISTS idx_sessions_task_priority ON sessions( - task_priority, - task_due_at, - source_kind -) WHERE task_priority IS NOT NULL; -CREATE INDEX IF NOT EXISTS idx_sessions_task_due_at ON sessions( - task_due_at, - source_kind -) WHERE task_due_at IS NOT NULL; - -CREATE TABLE IF NOT EXISTS parts ( - source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), - project_path TEXT NOT NULL, - session_id TEXT NOT NULL, - part_id TEXT NOT NULL, - message_id TEXT, - part_type TEXT, - role TEXT, - seq INTEGER, - timestamp TEXT, - text TEXT, - PRIMARY KEY (source_kind, project_path, session_id, part_id) -); -CREATE INDEX IF NOT EXISTS idx_parts_session ON parts( - source_kind, project_path, session_id -); - -CREATE TABLE IF NOT EXISTS surface_messages ( - source_kind TEXT NOT NULL CHECK(source_kind IN ('local', 'acp-remote')), - project_path TEXT NOT NULL, - session_id TEXT NOT NULL, - message_seq INTEGER NOT NULL CHECK(message_seq >= 0), - message_id TEXT NOT NULL, - message_json TEXT NOT NULL, - byte_count INTEGER NOT NULL CHECK(byte_count >= 0), - PRIMARY KEY ( - source_kind, project_path, session_id, message_seq, message_id - ) -); -CREATE INDEX IF NOT EXISTS idx_surface_messages_history ON surface_messages( - source_kind, project_path, session_id, message_seq DESC, message_id DESC -); - -CREATE VIRTUAL TABLE IF NOT EXISTS parts_fts USING fts5( - text, - source_kind UNINDEXED, - project_path UNINDEXED, - session_id UNINDEXED, - part_id UNINDEXED, - role UNINDEXED, - timestamp UNINDEXED, - seq UNINDEXED, - tokenize = 'unicode61' -); -`; - -const DROP_ALL = ` -DROP TABLE IF EXISTS parts_fts; -DROP TABLE IF EXISTS surface_messages; -DROP TABLE IF EXISTS parts; -DROP TABLE IF EXISTS sessions; -DROP TABLE IF EXISTS projection_state; -DROP TABLE IF EXISTS surface_projection_meta; -`; - import type { SqliteDb } from './driver.js'; +import DROP_ALL from './drop-all.sql?raw'; +import DDL from './schema.sql?raw'; -/** - * 确保 schema 存在且为当前版本。版本落后(且无迁移路径)时 drop 重建 —— 投影是 - * 纯派生缓存,重建成本可接受,换取零迁移负担。 - */ +/** 确保 schema 存在且为当前版本。版本落后(且无迁移路径)时 drop 重建 —— 投影是 纯派生缓存,重建成本可接受,换取零迁移负担。 */ export function migrate(db: SqliteDb): void { const version = Number(db.pragma('user_version') ?? 0); if (version === SCHEMA_VERSION) return; diff --git a/packages/cli/src/context/types.ts b/packages/cli/src/context/types.ts index 85e2b8f9c..4f6027841 100644 --- a/packages/cli/src/context/types.ts +++ b/packages/cli/src/context/types.ts @@ -1,6 +1,4 @@ -/** - * 上下文管理模块的核心类型定义 - */ +/** 上下文管理模块的核心类型定义 */ import type { AcpRemotePathStyle } from '../acp/AcpRemotePath.js'; import type { @@ -14,7 +12,6 @@ import type { SessionStateStorage } from './storage/SessionStateStorage.js'; export const MAX_TURN_INPUT_MESSAGE_IDS = 120; export const MAX_TURN_INPUT_MESSAGE_ID_CHARS = 128; -export const MAX_TURN_ID_CHARS = 128; export function parseTurnInputMessageIds(value: unknown): string[] | undefined { if ( @@ -75,118 +72,12 @@ export function parseTurnAbortAcknowledgedInputMessageIds(data: unknown): string ); } -export interface ContextMessage { - id: string; - role: MessageRole; - content: string; - timestamp: number; - metadata?: JsonObject; -} - -export interface ToolCall { - id: string; - name: string; - input: JsonValue; - output?: JsonValue; - timestamp: number; - status: 'pending' | 'success' | 'error'; - error?: string; -} - -export interface SystemContext { - role: string; - capabilities: string[]; - tools: string[]; - version: string; -} - -export interface SessionContext { - sessionId: string; - userId?: string; - preferences: JsonObject; - configuration: JsonObject; - startTime: number; -} - -export interface ConversationContext { - messages: ContextMessage[]; - summary?: string; - topics: string[]; - lastActivity: number; -} - -interface ToolContext { - recentCalls: ToolCall[]; - toolStates: JsonObject; - dependencies: Record; -} - -export interface WorkspaceContext { - projectPath?: string; - currentFiles: string[]; - recentFiles: string[]; - gitInfo?: { - branch: string; - status: string; - lastCommit?: string; - }; - environment: JsonObject; -} - -export interface ContextLayer { - system: SystemContext; - session: SessionContext; - conversation: ConversationContext; - tool: ToolContext; - workspace: WorkspaceContext; -} - -export interface ContextData { - layers: ContextLayer; - metadata: { - totalTokens: number; - priority: number; - relevanceScore?: number; - lastUpdated: number; - }; -} - -export interface ContextFilter { - maxTokens?: number; - maxMessages?: number; - timeWindow?: number; // 毫秒 - priority?: number; - includeTools?: boolean; - includeWorkspace?: boolean; -} - -export interface CompressedContext { - summary: string; - keyPoints: string[]; - recentMessages: ContextMessage[]; - toolSummary?: string; - tokenCount: number; -} - -export interface ContextStorageOptions { - maxMemorySize: number; - persistentPath?: string; - cacheSize: number; - compressionEnabled: boolean; -} - export interface ContextManagerOptions { projectPath: string; stateStorage?: SessionStateStorage; - storage: ContextStorageOptions; - defaultFilter: ContextFilter; - compressionThreshold: number; - enableVectorSearch?: boolean; } -/** - * JSONL 消息类型 - */ +/** JSONL 消息类型 */ export type JSONLEventType = | 'session_created' | 'session_updated' diff --git a/packages/cli/src/goals/index.ts b/packages/cli/src/goals/index.ts deleted file mode 100644 index ed9a28955..000000000 --- a/packages/cli/src/goals/index.ts +++ /dev/null @@ -1,48 +0,0 @@ -export type { GoalExecutionFrontierPreparation } from './executionFrontier.js'; -export { - formatGoalExecutionFrontier, - getGoalTaskListId, - readGoalExecutionFrontier, -} from './executionFrontier.js'; -export type { GoalExecutionHostFailureAccumulator } from './executionHostFailure.js'; -export { - buildGoalExecutionHostFailurePrompt, - createGoalExecutionHostFailureAccumulator, - executionHostFailureForTerminalFailure, - isGoalExecutionHostFailureCategory, - observeGoalExecutionHostToolResult, - resolveGoalExecutionHostFailure, -} from './executionHostFailure.js'; -export { - classifyGoalFrontierStall, - formatGoalFrontierStall, -} from './frontierStall.js'; -export { GoalStore } from './GoalStore.js'; -export { detectGoalPrematureStop } from './prematureStop.js'; -export { buildGoalContinuationPrompt, formatGoalSummary } from './prompts.js'; -export type { - GoalChangeEvent, - GoalCreateInput, - GoalExecutionFrontier, - GoalExecutionHostFailureCategory, - GoalExecutionHostFailureState, - GoalFrontierStallCategory, - GoalFrontierStallInput, - GoalFrontierStallState, - GoalPrematureStopPattern, - GoalPrematureStopState, - GoalProgress, - GoalSnapshot, - GoalStatus, - GoalVerificationStallState, -} from './types.js'; -export { - GOAL_EXECUTION_HOST_FAILURE_CATEGORIES, - GOAL_FRONTIER_STALL_CATEGORIES, - GOAL_PREMATURE_STOP_PATTERNS, - MAX_CONSECUTIVE_GOAL_EXECUTION_HOST_FAILURES, - MAX_CONSECUTIVE_GOAL_FRONTIER_STALLS, - MAX_CONSECUTIVE_GOAL_PREMATURE_STOPS, - MAX_CONSECUTIVE_GOAL_VERIFICATION_STALLS, - MAX_GOAL_VERIFICATION_FEEDBACK_CHARS, -} from './types.js'; diff --git a/packages/cli/src/hooks/HookConfig.ts b/packages/cli/src/hooks/HookConfig.ts index 1c984f99e..0cf450de3 100644 --- a/packages/cli/src/hooks/HookConfig.ts +++ b/packages/cli/src/hooks/HookConfig.ts @@ -1,15 +1,8 @@ -/** - * Hook Configuration - * - * 默认配置和配置加载逻辑 - */ +/** Hook Configuration 默认配置和配置加载逻辑 */ import { type HookConfig, HookType } from './types/HookTypes.js'; -/** - * 默认 Hook 配置 - * 与 Claude Code 对齐的完整配置 - */ +/** 默认 Hook 配置 与 Claude Code 对齐的完整配置 */ export const DEFAULT_HOOK_CONFIG: Required = { enabled: false, // 默认禁用,需要显式启用 defaultTimeout: 60, // 60 秒 @@ -55,9 +48,7 @@ export const DEFAULT_HOOK_CONFIG: Required = { Compaction: [], }; -/** - * 合并配置 - */ +/** 合并配置 */ export function mergeHookConfig( base: HookConfig, override: Partial @@ -85,9 +76,7 @@ export function mergeHookConfig( }; } -/** - * 从环境变量解析配置 - */ +/** 从环境变量解析配置 */ export function parseEnvConfig(): Partial { const config: Partial = {}; diff --git a/packages/cli/src/hooks/HookExecutionGuard.ts b/packages/cli/src/hooks/HookExecutionGuard.ts index d34324910..b095f30c2 100644 --- a/packages/cli/src/hooks/HookExecutionGuard.ts +++ b/packages/cli/src/hooks/HookExecutionGuard.ts @@ -1,16 +1,10 @@ -/** - * Hook Execution Guard - * - * 保证单次工具调用只触发一次 Hook - */ +/** Hook Execution Guard 保证单次工具调用只触发一次 Hook */ export class HookExecutionGuard { // toolUseId -> Set private executedHooks = new Map>(); - /** - * 检查是否可以执行 - */ + /** 检查是否可以执行 */ canExecute(toolUseId: string, eventName: string): boolean { if (!this.executedHooks.has(toolUseId)) { this.executedHooks.set(toolUseId, new Set()); @@ -27,9 +21,7 @@ export class HookExecutionGuard { return true; } - /** - * 标记已执行 - */ + /** 标记已执行 */ markExecuted(toolUseId: string, eventName: string): void { const executed = this.executedHooks.get(toolUseId); if (executed) { @@ -37,16 +29,12 @@ export class HookExecutionGuard { } } - /** - * 清理已完成的工具 - */ + /** 清理已完成的工具 */ cleanup(toolUseId: string): void { this.executedHooks.delete(toolUseId); } - /** - * 清理所有 - */ + /** 清理所有 */ cleanupAll(): void { this.executedHooks.clear(); } diff --git a/packages/cli/src/hooks/HookExecutor.ts b/packages/cli/src/hooks/HookExecutor.ts index 6aae3734e..e3b351078 100644 --- a/packages/cli/src/hooks/HookExecutor.ts +++ b/packages/cli/src/hooks/HookExecutor.ts @@ -1,8 +1,4 @@ -/** - * Hook Executor - * - * 负责执行单个或多个 Hooks - */ +/** Hook Executor 负责执行单个或多个 Hooks */ import path from 'node:path'; import type { SessionModelResources } from '../agent/resources/WorkspaceModelResources.js'; @@ -23,7 +19,6 @@ import { type HookInput, HookType, type HttpHook, - type NotificationHookResult, type PermissionRequestHookResult, type PostToolHookResult, type PostToolUseFailureHookResult, @@ -39,9 +34,7 @@ import { const promptHookLogger = createLogger(LogCategory.EXECUTION); -/** - * 事件类型对应的 hookSpecificOutput 字段说明 - */ +/** 事件类型对应的 hookSpecificOutput 字段说明 */ const EVENT_SCHEMA_HINTS: Record = { PreToolUse: '{ "permissionDecision": "approve" | "deny" | "ask", "permissionDecisionReason": "...", "updatedInput": { ... } }', @@ -61,9 +54,7 @@ const EVENT_SCHEMA_HINTS: Record = { Compaction: '{ "blockCompaction": true, "blockReason": "阻止压缩的原因" }', }; -/** - * Hook 执行器 - */ +/** Hook 执行器 */ export class HookExecutor { private processExecutor = new SecureProcessExecutor(); private outputParser = new OutputParser(); @@ -121,13 +112,7 @@ export class HookExecutor { } } - /** - * 执行 PreToolUse Hooks (串行) - * - * 串行执行的原因: - * 1. 第一个 deny 需要立即中断 - * 2. updatedInput 需要累积应用 - */ + /** 执行 PreToolUse Hooks (串行) 串行执行的原因: 1. 第一个 deny 需要立即中断 2. updatedInput 需要累积应用 */ async executePreToolHooks( hooks: Hook[], input: HookInput, @@ -233,13 +218,7 @@ export class HookExecutor { }; } - /** - * 执行 PostToolUse Hooks (并行) - * - * 并行执行的原因: - * 1. 提高性能 - * 2. 结果互不影响,可以合并 - */ + /** 执行 PostToolUse Hooks (并行) 并行执行的原因: 1. 提高性能 2. 结果互不影响,可以合并 */ async executePostToolHooks( hooks: Hook[], input: HookInput, @@ -290,11 +269,7 @@ export class HookExecutor { }; } - /** - * 执行 Stop Hooks (串行) - * - * 任何一个 hook 返回 continue: false 就阻止停止 - */ + /** 执行 Stop Hooks (串行) 任何一个 hook 返回 continue: false 就阻止停止 */ async executeStopHooks( hooks: Hook[], input: HookInput, @@ -340,9 +315,7 @@ export class HookExecutor { }; } - /** - * 执行 SubagentStop Hooks (串行) - */ + /** 执行 SubagentStop Hooks (串行) */ async executeSubagentStopHooks( hooks: Hook[], input: HookInput, @@ -393,11 +366,7 @@ export class HookExecutor { }; } - /** - * 执行 PermissionRequest Hooks (串行) - * - * 第一个 approve 或 deny 决策立即返回 - */ + /** 执行 PermissionRequest Hooks (串行) 第一个 approve 或 deny 决策立即返回 */ async executePermissionRequestHooks( hooks: Hook[], input: HookInput, @@ -495,11 +464,7 @@ export class HookExecutor { }; } - /** - * 执行 UserPromptSubmit Hooks (串行) - * - * 收集 contextInjection (stdout) 和 updatedPrompt - */ + /** 执行 UserPromptSubmit Hooks (串行) 收集 contextInjection (stdout) 和 updatedPrompt */ async executeUserPromptSubmitHooks( hooks: Hook[], input: HookInput, @@ -560,11 +525,7 @@ export class HookExecutor { }; } - /** - * 执行 SessionStart Hooks (串行) - * - * 收集环境变量 - */ + /** 执行 SessionStart Hooks (串行) 收集环境变量 */ async executeSessionStartHooks( hooks: Hook[], input: HookInput, @@ -611,9 +572,7 @@ export class HookExecutor { }; } - /** - * 执行 SessionEnd Hooks (并行,不阻塞) - */ + /** 执行 SessionEnd Hooks (并行,不阻塞) */ async executeSessionEndHooks( hooks: Hook[], input: HookInput, @@ -645,9 +604,7 @@ export class HookExecutor { }; } - /** - * 执行 PostToolUseFailure Hooks (并行) - */ + /** 执行 PostToolUseFailure Hooks (并行) */ async executePostToolUseFailureHooks( hooks: Hook[], input: HookInput, @@ -687,61 +644,7 @@ export class HookExecutor { }; } - /** - * 执行 Notification Hooks (串行) - */ - async executeNotificationHooks( - hooks: Hook[], - input: HookInput, - context: HookExecutionContext - ): Promise { - const originalMessage = 'message' in input ? (input.message as string) : ''; - - if (hooks.length === 0) { - return { suppress: false, message: originalMessage }; - } - - const warnings: string[] = []; - let suppress = false; - let message = originalMessage; - - for (const hook of hooks) { - try { - const result = await this.executeHook(hook, input, context); - - if (!result.success) { - if (result.warning) { - warnings.push(result.warning); - } - continue; - } - - // 检查是否抑制通知 - if (result.output?.suppressOutput) { - suppress = true; - break; - } - - // 修改消息内容(来自 stdout) - if (result.stdout && result.stdout.trim()) { - message = result.stdout.trim(); - } - } catch (err) { - const errorMsg = err instanceof Error ? err.message : String(err); - warnings.push(`Hook failed: ${errorMsg}`); - } - } - - return { - suppress, - message, - warning: warnings.length > 0 ? warnings.join('\n') : undefined, - }; - } - - /** - * 执行 Compaction Hooks (串行) - */ + /** 执行 Compaction Hooks (串行) */ async executeCompactionHooks( hooks: Hook[], input: HookInput, @@ -784,9 +687,7 @@ export class HookExecutor { }; } - /** - * 执行单个 Hook - */ + /** 执行单个 Hook */ private async executeHook( hook: Hook, input: HookInput, @@ -834,9 +735,7 @@ export class HookExecutor { throw new Error(`Hook type ${(hook as Hook).type} not supported`); } - /** - * 执行命令 Hook - */ + /** 执行命令 Hook */ private async executeCommandHook( hook: CommandHook, input: HookInput, @@ -1025,10 +924,7 @@ export class HookExecutor { } } - /** - * 带超时地执行 function handler。 - * handler 本身不可被强制中止,超时后调用方立即拿到错误,handler 自行清理。 - */ + /** 带超时地执行 function handler。 handler 本身不可被强制中止,超时后调用方立即拿到错误,handler 自行清理。 */ private runFunctionWithTimeout( handler: FunctionHook['handler'], input: HookInput, @@ -1204,9 +1100,7 @@ export class HookExecutor { }; } - /** - * 获取或创建 ChatService 实例(按 modelId 缓存) - */ + /** 获取或创建 ChatService 实例(按 modelId 缓存) */ private async getOrCreateChatService( modelId: string | undefined, context: HookExecutionContext @@ -1278,9 +1172,7 @@ export class HookExecutor { }; } - /** - * 构建 PromptHook 的系统提示 - */ + /** 构建 PromptHook 的系统提示 */ private buildPromptHookSystemMessage(hook: PromptHook, eventType: string): string { const schemaHint = EVENT_SCHEMA_HINTS[eventType] || '{ ... 根据事件类型返回相应字段 }'; @@ -1302,9 +1194,7 @@ export class HookExecutor { ); } - /** - * 从 LLM 响应中提取 JSON(去除 markdown code block 包装) - */ + /** 从 LLM 响应中提取 JSON(去除 markdown code block 包装) */ private extractJsonFromLLMResponse(text: string): string { const trimmed = text.trim(); @@ -1317,9 +1207,7 @@ export class HookExecutor { return trimmed; } - /** - * 并发执行多个 Hooks (带并发限制) - */ + /** 并发执行多个 Hooks (带并发限制) */ private async executeHooksConcurrently( hooks: Hook[], input: HookInput, @@ -1362,9 +1250,7 @@ export class HookExecutor { } } -/** - * 读取 fetch Response body, 超过 maxBytes 时截断并附加警告注释 - */ +/** 读取 fetch Response body, 超过 maxBytes 时截断并附加警告注释 */ async function readBodyWithLimit( response: Response, maxBytes: number diff --git a/packages/cli/src/hooks/HookManager.ts b/packages/cli/src/hooks/HookManager.ts index e56d633df..b94612b48 100644 --- a/packages/cli/src/hooks/HookManager.ts +++ b/packages/cli/src/hooks/HookManager.ts @@ -1,8 +1,4 @@ -/** - * Hook Manager - * - * 管理 Hook 配置和执行 - */ +/** Hook Manager 管理 Hook 配置和执行 */ import path from 'node:path'; import { LRUCache } from 'lru-cache'; @@ -34,8 +30,6 @@ import { HookType, type MatchContext, type MatcherConfig, - type NotificationHookResult, - type NotificationInput, type PermissionRequestHookResult, type PermissionRequestInput, type PostToolHookResult, @@ -58,11 +52,7 @@ import { export const MAX_RESIDENT_HOOK_PROJECT_CONFIGS = 64; -/** - * Hook Manager - * - * 单例模式,管理整个应用的 Hook 系统 - */ +/** Hook Manager 单例模式,管理整个应用的 Hook 系统 */ export class HookManager { private static instance: HookManager | null = null; @@ -81,9 +71,7 @@ export class HookManager { private constructor() {} - /** - * 获取单例实例 - */ + /** 获取单例实例 */ static getInstance(): HookManager { if (!HookManager.instance) { HookManager.instance = new HookManager(); @@ -149,9 +137,7 @@ export class HookManager { HookManager.instance = null; } - /** - * 加载配置 - */ + /** 加载配置 */ loadConfig(config: Partial, projectDir: string = getCwd()): void { // 合并配置: 默认 -> 用户配置 -> 环境变量 let merged = mergeHookConfig(DEFAULT_HOOK_CONFIG, config); @@ -161,9 +147,7 @@ export class HookManager { this.storeProjectConfig(projectDir, merged); } - /** - * 检查是否启用 - */ + /** 检查是否启用 */ isEnabled(projectDir: string = getCwd(), sessionId?: string): boolean { const config = sessionId ? this.getExecutionConfig(sessionId, projectDir) @@ -187,17 +171,13 @@ export class HookManager { return true; } - /** - * Disable all hooks process-wide. Kept for host policy and test compatibility. - */ + /** Disable all hooks process-wide. Kept for host policy and test compatibility. */ disable(): void { this.processDisabled = true; console.log('[HookManager] Hooks disabled for this process'); } - /** - * Re-enable process-wide hook execution. - */ + /** Re-enable process-wide hook execution. */ enable(): void { this.processDisabled = false; console.log('[HookManager] Hooks enabled for this process'); @@ -219,9 +199,7 @@ export class HookManager { return this.disabledSessions.has(this.sessionStateKey(sessionId, projectDir)); } - /** - * 获取当前配置(只读) - */ + /** 获取当前配置(只读) */ getConfig(projectDir: string = getCwd()): Readonly { const projectKey = path.resolve(projectDir); const projectConfig = this.projectConfigs.get(projectKey); @@ -410,9 +388,7 @@ export class HookManager { }; } - /** - * 重新加载配置(直接从配置文件读取) - */ + /** 重新加载配置(直接从配置文件读取) */ async reloadConfig(projectDir: string = getCwd()): Promise { const fs = await import('node:fs/promises'); const path = await import('node:path'); @@ -431,9 +407,7 @@ export class HookManager { } } - /** - * 执行 PreToolUse Hooks - */ + /** 执行 PreToolUse Hooks */ async executePreToolHooks( toolName: string, toolUseId: string, @@ -534,9 +508,7 @@ export class HookManager { } } - /** - * 执行 PostToolUse Hooks - */ + /** 执行 PostToolUse Hooks */ async executePostToolHooks( toolName: string, toolUseId: string, @@ -626,9 +598,7 @@ export class HookManager { } } - /** - * 执行 Stop Hooks - */ + /** 执行 Stop Hooks */ async executeStopHooks(context: { projectDir: string; sessionId: string; @@ -684,9 +654,7 @@ export class HookManager { } } - /** - * 执行 SubagentStop Hooks - */ + /** 执行 SubagentStop Hooks */ async executeSubagentStopHooks( agentType: string, context: { @@ -752,9 +720,7 @@ export class HookManager { } } - /** - * 执行 PermissionRequest Hooks - */ + /** 执行 PermissionRequest Hooks */ async executePermissionRequestHooks( toolName: string, toolUseId: string, @@ -825,9 +791,7 @@ export class HookManager { } } - /** - * 执行 UserPromptSubmit Hooks - */ + /** 执行 UserPromptSubmit Hooks */ async executeUserPromptSubmitHooks( userPrompt: string, context: { @@ -889,9 +853,7 @@ export class HookManager { } } - /** - * 执行 SessionStart Hooks - */ + /** 执行 SessionStart Hooks */ async executeSessionStartHooks(context: { projectDir: string; sessionId: string; @@ -949,9 +911,7 @@ export class HookManager { } } - /** - * 执行 SessionEnd Hooks - */ + /** 执行 SessionEnd Hooks */ async executeSessionEndHooks( reason: SessionEndInput['reason'], context: { @@ -1005,9 +965,7 @@ export class HookManager { } } - /** - * 执行 PostToolUseFailure Hooks - */ + /** 执行 PostToolUseFailure Hooks */ async executePostToolUseFailureHooks( toolName: string, toolUseId: string, @@ -1085,71 +1043,6 @@ export class HookManager { } } - /** - * 执行 Notification Hooks - */ - async executeNotificationHooks( - notificationType: NotificationInput['notification_type'], - message: string, - context: { - projectDir: string; - sessionId: string; - permissionMode: PermissionMode; - title?: string; - abortSignal?: AbortSignal; - } - ): Promise { - const config = this.getExecutionConfig(context.sessionId, context.projectDir); - if (!this.isExecutionEnabled(config, context.sessionId, context.projectDir)) { - return { suppress: false, message }; - } - - // 构建 Hook 输入 - const hookInput: NotificationInput = { - hook_event_name: HookEvent.Notification, - hook_execution_id: nanoid(), - timestamp: new Date().toISOString(), - project_dir: context.projectDir, - session_id: context.sessionId, - permission_mode: context.permissionMode, - notification_type: notificationType, - title: context.title, - message, - }; - - // 获取 hooks - const hooks = this.getMatchingHooks(HookEvent.Notification, {}, config); - - if (hooks.length === 0) { - return { suppress: false, message }; - } - - // 构建执行上下文 - const execContext: HookExecutionContext = { - projectDir: context.projectDir, - sessionId: context.sessionId, - permissionMode: context.permissionMode, - config, - abortSignal: context.abortSignal, - }; - - try { - const results = await this.executor.executeNotificationHooks( - hooks, - hookInput, - execContext - ); - return results; - } catch (err) { - console.error('[HookManager] Error executing Notification hooks:', err); - return { - suppress: false, - message, - warning: `Hook execution failed: ${err instanceof Error ? err.message : String(err)}`, - }; - } - } - async executeElicitationHooks( details: McpElicitationDetails, context: { @@ -1253,9 +1146,7 @@ export class HookManager { } } - /** - * 执行 Compaction Hooks - */ + /** 执行 Compaction Hooks */ async executeCompactionHooks( trigger: 'manual' | 'auto', context: { @@ -1317,9 +1208,7 @@ export class HookManager { } } - /** - * 获取匹配的 Hooks - */ + /** 获取匹配的 Hooks */ private getMatchingHooks( event: HookEvent, context: MatchContext, @@ -1341,9 +1230,7 @@ export class HookManager { return matchedHooks; } - /** - * 从工具输入提取文件路径 - */ + /** 从工具输入提取文件路径 */ private extractFilePaths(toolInput: Record): string[] { const paths: string[] = []; // 常见的文件路径字段 @@ -1369,9 +1256,7 @@ export class HookManager { return [...new Set(paths)]; } - /** - * 从工具输入提取命令 - */ + /** 从工具输入提取命令 */ private extractCommand( toolName: string, toolInput: Record @@ -1387,9 +1272,7 @@ export class HookManager { return undefined; } - /** - * 清理所有状态 - */ + /** 清理所有状态 */ cleanup(): void { this.guard.cleanupAll(); this.config = DEFAULT_HOOK_CONFIG; diff --git a/packages/cli/src/hooks/HttpHookSecurity.ts b/packages/cli/src/hooks/HttpHookSecurity.ts index 468266c65..d96f82c15 100644 --- a/packages/cli/src/hooks/HttpHookSecurity.ts +++ b/packages/cli/src/hooks/HttpHookSecurity.ts @@ -1,13 +1,8 @@ /** - * HTTP Hook 安全检查 - * - * 防御要点: - * 1. SSRF: 默认拒绝 loopback (127.0.0.1/::1/localhost) 和 RFC1918 私有 IP 段 - * (10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16) 以及 link-local (169.254.0.0/16) - * 2. TLS: 默认要求 https:// - * 3. allowedHosts: 显式允许的 hostname (精确 or *.domain.com 通配) 可以绕过上述限制 - * - * 所有检查在请求发起前进行。 + * HTTP Hook 安全检查

防御要点: 1. SSRF: 默认拒绝 loopback (127.0.0.1/::1/localhost) 和 RFC1918 + * 私有 IP 段 (10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16) 以及 link-local (169.254.0.0/16) 2. + * TLS: 默认要求 https:// 3. allowedHosts: 显式允许的 hostname (精确 or *.domain.com 通配) 可以绕过上述限制 + *

所有检查在请求发起前进行。 */ import type { HttpHookPolicy } from './types/HookTypes.js'; @@ -19,10 +14,7 @@ export class HttpHookSecurityError extends Error { } } -/** - * 验证 HTTP Hook URL 是否可达 - * 不通过时抛 HttpHookSecurityError - */ +/** 验证 HTTP Hook URL 是否可达 不通过时抛 HttpHookSecurityError */ export function validateHookUrl(url: string, policy: HttpHookPolicy = {}): void { let parsed: URL; try { diff --git a/packages/cli/src/hooks/Matcher.ts b/packages/cli/src/hooks/Matcher.ts index f3bca1516..d737cf83c 100644 --- a/packages/cli/src/hooks/Matcher.ts +++ b/packages/cli/src/hooks/Matcher.ts @@ -20,13 +20,9 @@ import type { MatchContext, MatcherConfig } from './types/HookTypes.js'; */ const PARAM_PATTERN_REGEX = /^([A-Za-z0-9_|]+)\((.+)\)$/; -/** - * Hook Matcher - */ +/** Hook Matcher */ export class Matcher { - /** - * 检查是否匹配 - */ + /** 检查是否匹配 */ matches(config: MatcherConfig | undefined, context: MatchContext): boolean { // 没有 matcher 配置,匹配所有 if (!config) { @@ -62,9 +58,7 @@ export class Matcher { return true; } - /** - * 匹配工具名(支持字符串和数组) - */ + /** 匹配工具名(支持字符串和数组) */ private matchTools(tools: string | string[], context: MatchContext): boolean { // 数组格式:任一匹配即可 if (Array.isArray(tools)) { @@ -74,9 +68,7 @@ export class Matcher { return this.matchToolWithParams(tools, context); } - /** - * 匹配文件路径(支持字符串和数组) - */ + /** 匹配文件路径(支持字符串和数组) */ private matchPaths(paths: string | string[], filePath: string): boolean { // 数组格式:任一匹配即可 if (Array.isArray(paths)) { @@ -90,9 +82,7 @@ export class Matcher { return isMatch(filePath); } - /** - * 匹配命令(支持字符串和数组) - */ + /** 匹配命令(支持字符串和数组) */ private matchCommands(commands: string | string[], command: string): boolean { // 数组格式:任一匹配即可 if (Array.isArray(commands)) { @@ -130,9 +120,7 @@ export class Matcher { return false; } - // 然后匹配参数 - // 对于 Bash 工具,参数是 command - // 对于 Read/Edit/Write 等工具,参数是 filePath + // 然后匹配参数 对于 Bash 工具,参数是 command 对于 Read/Edit/Write 等工具,参数是 filePath const argValues = this.getArgValues( toolName!, command, @@ -148,9 +136,7 @@ export class Matcher { return argValues.some((argValue) => this.matchGlobOrPattern(argValue, argPattern)); } - /** - * 获取工具的参数值 - */ + /** 获取工具的参数值 */ private getArgValues( toolName: string, command?: string, @@ -171,9 +157,7 @@ export class Matcher { return command ? [command] : filePaths; } - /** - * 使用 glob 或简单模式匹配 - */ + /** 使用 glob 或简单模式匹配 */ private matchGlobOrPattern(value: string, pattern: string): boolean { // 如果包含 glob 特殊字符,使用 picomatch if (/[*?[\]{}!]/.test(pattern)) { diff --git a/packages/cli/src/hooks/OutputParser.ts b/packages/cli/src/hooks/OutputParser.ts index 3c97e4941..8ab503577 100644 --- a/packages/cli/src/hooks/OutputParser.ts +++ b/packages/cli/src/hooks/OutputParser.ts @@ -1,8 +1,4 @@ -/** - * Hook Output Parser - * - * 解析 Hook 命令的输出 - */ +/** Hook Output Parser 解析 Hook 命令的输出 */ import { safeParseHookOutput } from './schemas/HookSchemas.js'; import type { @@ -14,13 +10,9 @@ import type { ProcessResult, } from './types/HookTypes.js'; -/** - * 输出解析器 - */ +/** 输出解析器 */ export class OutputParser { - /** - * 解析进程结果 - */ + /** 解析进程结果 */ parse( result: ProcessResult, hook: Hook, @@ -158,9 +150,7 @@ export class OutputParser { return this.parseByExitCode(result, hook, config); } - /** - * 根据退出码解析 - */ + /** 根据退出码解析 */ private parseByExitCode( result: ProcessResult, hook: Hook, @@ -272,9 +262,7 @@ export class OutputParser { } } - /** - * 尝试解析 JSON - */ + /** 尝试解析 JSON */ private tryParseJSON(text: string): unknown | null { try { const trimmed = text.trim(); diff --git a/packages/cli/src/hooks/SecureProcessExecutor.ts b/packages/cli/src/hooks/SecureProcessExecutor.ts index aebd6a21d..7bf3eb1c6 100644 --- a/packages/cli/src/hooks/SecureProcessExecutor.ts +++ b/packages/cli/src/hooks/SecureProcessExecutor.ts @@ -1,8 +1,4 @@ -/** - * Secure Process Executor - * - * 安全地执行 Hook 子进程 - */ +/** Secure Process Executor 安全地执行 Hook 子进程 */ import { spawnOwnedProcess } from '../utils/process/OwnedProcessTree.js'; import type { @@ -12,9 +8,7 @@ import type { ProcessResult, } from './types/HookTypes.js'; -/** - * 流量限制器 - */ +/** 流量限制器 */ class StreamLimiter { private content = ''; private maxSize: number; @@ -39,17 +33,13 @@ class StreamLimiter { } } -/** - * 安全进程执行器 - */ +/** 安全进程执行器 */ export class SecureProcessExecutor { private readonly MAX_STDOUT_SIZE = 1 * 1024 * 1024; // 1MB private readonly MAX_STDERR_SIZE = 1 * 1024 * 1024; // 1MB private readonly MAX_INPUT_SIZE = 100 * 1024; // 100KB - /** - * 执行命令 - */ + /** 执行命令 */ async execute( command: string, input: HookInput, @@ -170,9 +160,7 @@ export class SecureProcessExecutor { }); } - /** - * 创建安全的环境变量 - */ + /** 创建安全的环境变量 */ private createSafeEnv( input: HookInput, environment: Readonly> = {} diff --git a/packages/cli/src/hooks/types/HookTypes.ts b/packages/cli/src/hooks/types/HookTypes.ts index bcb52e808..78e309619 100644 --- a/packages/cli/src/hooks/types/HookTypes.ts +++ b/packages/cli/src/hooks/types/HookTypes.ts @@ -1,187 +1,73 @@ -/** - * Hook System Types - * - * 定义 Blade Hooks System 的核心类型 - */ - import type { PermissionMode } from '../../config/types.js'; import type { McpElicitationAction, McpElicitationContent, } from '../../mcp/McpElicitation.js'; -// ============================================================================ -// Hook Events -// ============================================================================ - -/** - * Hook 事件类型 - * 与 Claude Code 对齐的完整事件列表 - */ export enum HookEvent { - // ========== 工具执行类 ========== - /** 工具执行前 (可阻止或修改输入) */ PreToolUse = 'PreToolUse', - - /** 工具执行后 (可添加上下文或修改输出) */ PostToolUse = 'PostToolUse', - - /** 工具执行失败后 */ PostToolUseFailure = 'PostToolUseFailure', - - /** 权限请求时 (可自动批准/拒绝) */ PermissionRequest = 'PermissionRequest', - - /** MCP 服务器请求用户输入时 */ Elicitation = 'Elicitation', - - /** MCP 用户输入即将返回服务器时 */ ElicitationResult = 'ElicitationResult', - - // ========== 会话生命周期类 ========== - /** 用户提交提示词时 (可注入上下文) */ UserPromptSubmit = 'UserPromptSubmit', - - /** 会话启动时 */ SessionStart = 'SessionStart', - - /** 会话结束时 */ SessionEnd = 'SessionEnd', - - // ========== 控制流类 ========== - /** Agent 停止响应时 (可阻止停止) */ Stop = 'Stop', - - /** 子 Agent (Task) 停止响应时 */ SubagentStop = 'SubagentStop', - - // ========== 其他 ========== - /** 通知事件 */ Notification = 'Notification', - - /** 上下文压缩时 */ Compaction = 'Compaction', } -// ============================================================================ -// Hook Input -// ============================================================================ - -/** - * Hook 输入基础字段 - */ interface HookInputBase { - /** Hook 事件名称 */ hook_event_name: HookEvent; - - /** Hook 执行唯一 ID */ hook_execution_id: string; - - /** 时间戳 (ISO 8601) */ timestamp: string; - - /** 项目目录 */ project_dir: string; - - /** 会话 ID */ session_id: string; - - /** 当前权限模式 */ permission_mode: PermissionMode; - - /** 元数据 */ _metadata?: { blade_version: string; hook_timeout_ms: number; }; } -/** - * PreToolUse 输入 - */ export interface PreToolUseInput extends HookInputBase { hook_event_name: HookEvent.PreToolUse; - - /** 工具名称 */ tool_name: string; - - /** 工具使用 ID */ tool_use_id: string; - - /** 工具输入参数 */ tool_input: Record; } -/** - * PostToolUse 输入 - */ export interface PostToolUseInput extends HookInputBase { hook_event_name: HookEvent.PostToolUse; - - /** 工具名称 */ tool_name: string; - - /** 工具使用 ID */ tool_use_id: string; - - /** 工具输入参数 */ tool_input: Record; - - /** 工具响应 */ tool_response: unknown; } -/** - * Stop 输入 - */ export interface StopInput extends HookInputBase { hook_event_name: HookEvent.Stop; - - /** 停止原因 */ reason?: string; } -/** - * PostToolUseFailure 输入 - */ export interface PostToolUseFailureInput extends HookInputBase { hook_event_name: HookEvent.PostToolUseFailure; - - /** 工具名称 */ tool_name: string; - - /** 工具使用 ID */ tool_use_id: string; - - /** 工具输入参数 */ tool_input: Record; - - /** 错误信息 */ error: string; - - /** 错误类型 */ error_type?: string; - - /** 是否被中断 */ is_interrupt: boolean; - - /** 是否超时 */ is_timeout: boolean; } -/** - * PermissionRequest 输入 - */ export interface PermissionRequestInput extends HookInputBase { hook_event_name: HookEvent.PermissionRequest; - - /** 工具名称 */ tool_name: string; - - /** 工具使用 ID */ tool_use_id: string; - - /** 工具输入参数 */ tool_input: Record; } @@ -204,42 +90,21 @@ export interface ElicitationResultInput extends HookInputBase { content?: McpElicitationContent; } -/** - * UserPromptSubmit 输入 - */ export interface UserPromptSubmitInput extends HookInputBase { hook_event_name: HookEvent.UserPromptSubmit; - - /** 用户原始提示词 */ user_prompt: string; - - /** 是否包含图片 */ has_images: boolean; - - /** 图片数量 */ image_count: number; } -/** - * SessionStart 输入 - */ export interface SessionStartInput extends HookInputBase { hook_event_name: HookEvent.SessionStart; - - /** 是否恢复会话 */ is_resume: boolean; - - /** 恢复的会话 ID */ resume_session_id?: string; } -/** - * SessionEnd 输入 - */ export interface SessionEndInput extends HookInputBase { hook_event_name: HookEvent.SessionEnd; - - /** 结束原因 */ reason: | 'user_exit' | 'error' @@ -252,35 +117,17 @@ export interface SessionEndInput extends HookInputBase { | 'other'; } -/** - * SubagentStop 输入 - */ export interface SubagentStopInput extends HookInputBase { hook_event_name: HookEvent.SubagentStop; - - /** 子 Agent 类型 */ agent_type: string; - - /** 任务描述 */ task_description?: string; - - /** 是否成功 */ success: boolean; - - /** 结果摘要 */ result_summary?: string; - - /** 错误信息 */ error?: string; } -/** - * Notification 输入 - */ export interface NotificationInput extends HookInputBase { hook_event_name: HookEvent.Notification; - - /** 通知类型 */ notification_type: | 'permission_prompt' | 'idle_prompt' @@ -289,33 +136,17 @@ export interface NotificationInput extends HookInputBase { | 'info' | 'warning' | 'error'; - - /** 通知标题 */ title?: string; - - /** 通知内容 */ message: string; } -/** - * Compaction 输入 - */ export interface CompactionInput extends HookInputBase { hook_event_name: HookEvent.Compaction; - - /** 触发方式 */ trigger: 'manual' | 'auto'; - - /** 压缩前消息数 */ messages_before: number; - - /** 压缩前 token 数 */ tokens_before: number; } -/** - * Hook 输入联合类型 - */ export type HookInput = | PreToolUseInput | PostToolUseInput @@ -331,101 +162,47 @@ export type HookInput = | NotificationInput | CompactionInput; -// ============================================================================ -// Hook Output -// ============================================================================ - -/** - * 决策行为 - */ export enum DecisionBehavior { - /** 批准,继续执行 */ Approve = 'approve', - - /** 阻止,停止执行 */ Block = 'block', - - /** 异步执行,不等待结果 */ Async = 'async', } -/** - * 权限决策 (与 Blade 权限体系对齐) - */ export enum PermissionDecision { Allow = 'allow', Deny = 'deny', Ask = 'ask', } -/** - * PreToolUse 特定输出 - */ interface PreToolUseOutput { hookEventName?: 'PreToolUse'; - - /** 权限决策 */ permissionDecision?: PermissionDecision; - - /** 权限决策原因 */ permissionDecisionReason?: string; - - /** 修改后的工具输入 */ updatedInput?: Record; } -/** - * PostToolUse 特定输出 - */ interface PostToolUseOutput { hookEventName?: 'PostToolUse'; - - /** 添加给 LLM 的额外上下文 */ additionalContext?: string; - - /** 修改后的工具输出 */ updatedOutput?: unknown; } -/** - * Stop 特定输出 - */ interface StopOutput { hookEventName?: 'Stop'; - - /** 阻止停止,继续执行 */ continue?: boolean; - - /** 继续执行的原因 (发送给 LLM) */ continueReason?: string; } -/** - * SubagentStop 特定输出 - */ interface SubagentStopOutput { hookEventName?: 'SubagentStop'; - - /** 阻止停止,继续执行 */ continue?: boolean; - - /** 继续执行的原因 */ continueReason?: string; - - /** 额外上下文 */ additionalContext?: string; } -/** - * PermissionRequest 特定输出 - */ interface PermissionRequestOutput { hookEventName?: 'PermissionRequest'; - - /** 权限决策: approve (直接批准), deny (拒绝), ask (显示确认对话框) */ permissionDecision?: 'approve' | 'deny' | 'ask'; - - /** 决策原因 */ permissionDecisionReason?: string; } @@ -441,45 +218,23 @@ interface ElicitationResultOutput { content?: McpElicitationContent; } -/** - * UserPromptSubmit 特定输出 - */ interface UserPromptSubmitOutput { hookEventName?: 'UserPromptSubmit'; - - /** 修改后的用户提示词 */ updatedPrompt?: string; - - /** 注入到上下文的内容 (来自 stdout) */ contextInjection?: string; } -/** - * SessionStart 特定输出 - */ interface SessionStartOutput { hookEventName?: 'SessionStart'; - - /** 环境变量 (持久化到整个会话) */ env?: Record; } -/** - * Compaction 特定输出 - */ interface CompactionOutput { hookEventName?: 'Compaction'; - - /** 阻止压缩 */ blockCompaction?: boolean; - - /** 阻止原因 */ blockReason?: string; } -/** - * Hook 特定输出联合类型 - */ export type HookSpecificOutput = | PreToolUseOutput | PostToolUseOutput @@ -491,33 +246,15 @@ export type HookSpecificOutput = | UserPromptSubmitOutput | SessionStartOutput | CompactionOutput; - -/** - * Hook 输出结构 - */ export interface HookOutput { - /** 通用决策 */ decision?: { behavior?: DecisionBehavior; }; - - /** 系统消息 (显示给用户,不发送给 LLM) */ systemMessage?: string; - - /** 事件特定输出 */ hookSpecificOutput?: HookSpecificOutput; - - /** 抑制输出 (不显示成功消息) */ suppressOutput?: boolean; } -// ============================================================================ -// Hook Configuration -// ============================================================================ - -/** - * Hook 类型 - */ export enum HookType { Command = 'command', Prompt = 'prompt', @@ -525,37 +262,17 @@ export enum HookType { Http = 'http', } -/** - * 命令 Hook - */ export interface CommandHook { type: HookType.Command; - - /** Shell 命令 */ command: string; - - /** 超时时间 (秒) */ timeout?: number; - - /** 状态消息 (显示在 UI) */ statusMessage?: string; } -/** - * 提示词 Hook — 推理型传感器 - * - * 在 Hook 事件触发时发起 LLM 调用,用于代码质量评估、安全审查等需要推理的场景。 - */ export interface PromptHook { type: HookType.Prompt; - - /** 评估指令(发送给 LLM 的 prompt) */ prompt: string; - - /** 可选模型 ID(如 'haiku')。默认使用当前配置的模型。 */ model?: string; - - /** 超时时间 (秒) */ timeout?: number; } @@ -584,8 +301,6 @@ export interface FunctionHook { input: HookInput, ctx: HookExecutionContext ) => Promise | HookOutput | undefined; - - /** 超时时间 (秒), 默认 10s */ timeout?: number; } @@ -596,33 +311,14 @@ export interface PluginHookSource { pluginRoot: string; } -/** - * Hook 联合类型 - */ export type Hook = (CommandHook | PromptHook | FunctionHook | HttpHook) & { source?: PluginHookSource; }; -/** - * HTTP Hook — 远程 Webhook - * - * 把 Hook 事件转发到外部 HTTP 服务,用于企业级审批、集中化审计、 - * 第三方安全扫描等场景。 - * - * 固定语义: - * - 方法: POST - * - 请求体: JSON (HookInput 结构, 与 shell hook stdin 相同) - * - 响应: JSON (HookOutput 结构),非 JSON 按错误处理 - * - * 安全默认 (见 HttpHookPolicy): - * - 默认拒绝 loopback (127.0.0.1) 和私有 IP 段 (10/172.16-31/192.168/169.254) - * - 默认要求 HTTPS (本地例外需显式 allowedHosts) - * - 不跟随 redirect (防 allowlist 绕过) - */ +/** POSTs HookInput and accepts HookOutput JSON. Redirects, HTTP, loopback, and + * private networks are denied unless HttpHookPolicy explicitly allows them. */ export interface HttpHook { type: HookType.Http; - - /** 完整 URL (必须是 http:// 或 https://) */ url: string; /** @@ -630,281 +326,112 @@ export interface HttpHook { * 例: { Authorization: 'Bearer ${SECURITY_HOOK_TOKEN}' } */ headers?: Record; - - /** 超时时间 (秒), 默认 10s */ timeout?: number; - - /** 重试次数, 默认 0 (不重试); 指数退避 */ retries?: number; - - /** 跳过 TLS 证书校验 (仅 dev/自签名), 默认 false */ allowInsecureTLS?: boolean; - - /** 响应体最大字节数, 默认 256 KB */ maxResponseBytes?: number; } -/** - * HTTP Hook 全局安全策略 (进程级) - */ +/** HTTP Hook 全局安全策略 (进程级) */ export interface HttpHookPolicy { - /** - * 白名单 hostname (精确匹配或 *.example.com 通配). - * 匹配成功时跳过 loopback/private 检查以及 HTTPS 强制。 - */ allowedHosts?: string[]; - - /** 允许访问 loopback (127.x / ::1 / localhost), 默认 false */ allowLoopback?: boolean; - - /** 允许访问 RFC1918 私有 IP 段, 默认 false */ allowPrivateRanges?: boolean; - - /** 允许 http:// (非 HTTPS), 默认 false */ allowHttp?: boolean; } -/** - * Matcher 配置 - */ export interface MatcherConfig { - /** 工具名匹配 (支持精确、管道分隔、正则、或数组) */ tools?: string | string[]; - - /** 文件路径匹配 (glob 模式) */ paths?: string | string[]; - - /** 命令匹配 (正则) */ commands?: string | string[]; } -/** - * Hook Matcher - */ export interface HookMatcher { - /** 可选的名称 (用于日志和 UI) */ name?: string; - - /** 匹配器配置 */ matcher?: MatcherConfig; - - /** Hook 列表 */ hooks: Hook[]; } -/** - * Hook 配置 - * 与 Claude Code 对齐的完整配置 - */ export interface HookConfig { - /** 是否启用 hooks */ enabled?: boolean; - - /** 默认超时 (秒) */ defaultTimeout?: number; - - /** 超时行为 */ timeoutBehavior?: 'ignore' | 'deny' | 'ask'; - - /** 失败行为 */ failureBehavior?: 'ignore' | 'deny' | 'ask'; - - /** 最大并发 Hook 数 */ maxConcurrentHooks?: number; /** HTTP Hook 全局安全策略 */ httpPolicy?: HttpHookPolicy; - - // ========== 工具执行类 ========== - /** PreToolUse Hooks */ PreToolUse?: HookMatcher[]; - - /** PostToolUse Hooks */ PostToolUse?: HookMatcher[]; - - /** PostToolUseFailure Hooks */ PostToolUseFailure?: HookMatcher[]; - - /** PermissionRequest Hooks */ PermissionRequest?: HookMatcher[]; - - /** MCP Elicitation Hooks */ Elicitation?: HookMatcher[]; - - /** MCP ElicitationResult Hooks */ ElicitationResult?: HookMatcher[]; - - // ========== 会话生命周期类 ========== - /** UserPromptSubmit Hooks */ UserPromptSubmit?: HookMatcher[]; - - /** SessionStart Hooks */ SessionStart?: HookMatcher[]; - - /** SessionEnd Hooks */ SessionEnd?: HookMatcher[]; - - // ========== 控制流类 ========== - /** Stop Hooks */ Stop?: HookMatcher[]; - - /** SubagentStop Hooks */ SubagentStop?: HookMatcher[]; - - // ========== 其他 ========== - /** Notification Hooks */ Notification?: HookMatcher[]; - - /** Compaction Hooks */ Compaction?: HookMatcher[]; } -// ============================================================================ -// Hook Execution Results -// ============================================================================ - -/** - * Hook 退出码 - */ export enum HookExitCode { - /** 成功,继续 */ SUCCESS = 0, - - /** 非阻塞错误,记录但继续 */ NON_BLOCKING_ERROR = 1, - - /** 阻塞错误,停止执行 */ BLOCKING_ERROR = 2, - - /** 超时 */ TIMEOUT = 124, } -/** - * 进程执行结果 - */ export interface ProcessResult { - /** 标准输出 */ stdout: string; - - /** 标准错误 */ stderr: string; - - /** 退出码 */ exitCode: number; - - /** 是否超时 */ timedOut: boolean; } -/** - * Hook 执行结果 - */ export interface HookExecutionResult { - /** 是否成功 */ success: boolean; - - /** 是否阻塞 */ blocking?: boolean; - - /** 是否需要用户确认 (ask 行为) */ needsConfirmation?: boolean; - - /** 错误信息 */ error?: string; - - /** 警告信息 */ warning?: string; - - /** 解析后的输出 */ output?: HookOutput; - - /** 原始标准输出 */ stdout?: string; - - /** 原始标准错误 */ stderr?: string; - - /** 退出码 */ exitCode?: number; - - /** Hook 配置 */ hook?: Hook; } -/** - * PreToolUse Hook 执行结果 - */ export interface PreToolHookResult { - /** 决策 */ decision: 'allow' | 'deny' | 'ask'; - - /** 原因 */ reason?: string; - - /** 修改后的输入 */ modifiedInput?: Record; - - /** 警告信息 */ warning?: string; } -/** - * PostToolUse Hook 执行结果 - */ export interface PostToolHookResult { - /** 额外上下文 */ additionalContext?: string; - - /** 修改后的输出 */ modifiedOutput?: unknown; - - /** 警告信息 */ warning?: string; } -/** - * Stop Hook 执行结果 - */ export interface StopHookResult { - /** 是否应该停止 (false 表示继续执行) */ shouldStop: boolean; - - /** 继续执行的原因 */ continueReason?: string; - - /** 警告信息 */ warning?: string; } -/** - * SubagentStop Hook 执行结果 - */ export interface SubagentStopHookResult { - /** 是否应该停止 */ shouldStop: boolean; - - /** 继续执行的原因 */ continueReason?: string; - - /** 额外上下文 */ additionalContext?: string; - - /** 警告信息 */ warning?: string; } -/** - * PermissionRequest Hook 执行结果 - */ export interface PermissionRequestHookResult { - /** 权限决策 */ decision: 'approve' | 'deny' | 'ask'; - - /** 决策原因 */ reason?: string; - - /** 警告信息 */ warning?: string; } @@ -917,125 +444,48 @@ export interface ElicitationHookResult { warning?: string; } -/** - * UserPromptSubmit Hook 执行结果 - */ export interface UserPromptSubmitHookResult { - /** 是否继续处理 */ proceed: boolean; - - /** 修改后的提示词 */ updatedPrompt?: string; - - /** 注入到上下文的内容 */ contextInjection?: string; - - /** 警告信息 */ warning?: string; } -/** - * SessionStart Hook 执行结果 - */ export interface SessionStartHookResult { - /** 是否继续启动 */ proceed: boolean; - - /** 环境变量 */ env?: Record; - - /** 警告信息 */ warning?: string; } -/** - * SessionEnd Hook 执行结果 - * (通常不需要特殊处理) - */ export interface SessionEndHookResult { - /** 警告信息 */ warning?: string; } -/** - * PostToolUseFailure Hook 执行结果 - */ export interface PostToolUseFailureHookResult { - /** 额外上下文 */ additionalContext?: string; - - /** 警告信息 */ - warning?: string; -} - -/** - * Notification Hook 执行结果 - */ -export interface NotificationHookResult { - /** 是否抑制通知 */ - suppress: boolean; - - /** 修改后的消息 */ - message: string; - - /** 警告信息 */ warning?: string; } -/** - * Compaction Hook 执行结果 - */ export interface CompactionHookResult { - /** 是否阻止压缩 */ blockCompaction: boolean; - - /** 阻止原因 */ blockReason?: string; - - /** 警告信息 */ warning?: string; } -// ============================================================================ -// Hook Execution Context -// ============================================================================ - -/** - * Hook 执行上下文 - */ export interface HookExecutionContext { - /** 项目目录 */ projectDir: string; - - /** 会话 ID */ sessionId: string; - - /** 权限模式 */ permissionMode: PermissionMode; - - /** Hook 配置 */ config: HookConfig; /** Session-scoped explicit environment; never the mutable process environment. */ environment?: Readonly>; - - /** 中止信号 */ abortSignal?: AbortSignal; } -/** - * Matcher 匹配上下文 - */ export interface MatchContext { - /** 工具名称 */ toolName?: string; - - /** 文件路径 */ filePath?: string; - - /** 多文件工具涉及的全部路径 */ filePaths?: string[]; - - /** 命令 */ command?: string; } diff --git a/packages/cli/src/ide/detectIde.ts b/packages/cli/src/ide/detectIde.ts index b15f6e718..16631ca90 100644 --- a/packages/cli/src/ide/detectIde.ts +++ b/packages/cli/src/ide/detectIde.ts @@ -1,8 +1,4 @@ -/** - * IDE 检测模块 - * - * 检测当前运行环境中的 IDE 信息 - */ +/** IDE 检测模块 检测当前运行环境中的 IDE 信息 */ import { exec } from 'child_process'; import { promisify } from 'util'; @@ -17,9 +13,7 @@ export interface IdeInfo { } export class IdeDetector { - /** - * 检测当前 IDE 环境 - */ + /** 检测当前 IDE 环境 */ static async detectIde(): Promise { // 检查环境变量判断是否在 VS Code 终端中 const termProgram = process.env.TERM_PROGRAM; @@ -37,9 +31,7 @@ export class IdeDetector { return null; } - /** - * 检测 VS Code - */ + /** 检测 VS Code */ private static async detectVsCode(): Promise { try { const { stdout } = await execAsync('code --version'); @@ -64,15 +56,4 @@ export class IdeDetector { return null; } } - - /** - * 检测是否运行在 IDE 终端中 - */ - static isRunningInIdeTerminal(): boolean { - const termProgram = process.env.TERM_PROGRAM; - const vscodeTerminal = process.env.VSCODE_INJECTION; - const vscodeIpc = process.env.VSCODE_IPC_HOOK; - - return termProgram === 'vscode' || !!vscodeTerminal || !!vscodeIpc; - } } diff --git a/packages/cli/src/ide/ideInstaller.ts b/packages/cli/src/ide/ideInstaller.ts index f14927ef7..bf20b0bba 100644 --- a/packages/cli/src/ide/ideInstaller.ts +++ b/packages/cli/src/ide/ideInstaller.ts @@ -1,8 +1,4 @@ -/** - * IDE 安装器模块 - * - * 检测和安装 IDE 扩展 - */ +/** IDE 安装器模块 检测和安装 IDE 扩展 */ import { exec } from 'child_process'; import { promisify } from 'util'; @@ -17,9 +13,7 @@ export interface InstalledIde { } export class IdeInstaller { - /** - * 获取已安装的 IDE 列表 - */ + /** 获取已安装的 IDE 列表 */ static async getInstalledIdes(): Promise { const ides: InstalledIde[] = []; @@ -38,9 +32,7 @@ export class IdeInstaller { return ides; } - /** - * 检查指定 IDE 是否已安装 - */ + /** 检查指定 IDE 是否已安装 */ static async isIdeInstalled(ideId: string): Promise { switch (ideId) { case 'vscode': @@ -54,42 +46,7 @@ export class IdeInstaller { } } - /** - * 安装 Blade Code 扩展到 VS Code - */ - static async installExtension( - ideId: string - ): Promise<{ success: boolean; message: string }> { - let command: string; - - switch (ideId) { - case 'vscode': - command = 'code --install-extension blade-code.blade-code'; - break; - case 'vscode-insiders': - command = 'code-insiders --install-extension blade-code.blade-code'; - break; - case 'cursor': - command = 'cursor --install-extension blade-code.blade-code'; - break; - default: - return { success: false, message: '不支持的 IDE: ' + ideId }; - } - - try { - await execAsync(command); - return { success: true, message: '扩展安装成功' }; - } catch (error) { - return { - success: false, - message: '安装失败: ' + (error instanceof Error ? error.message : '未知错误'), - }; - } - } - - /** - * 检测 VS Code - */ + /** 检测 VS Code */ private static async checkVsCode(): Promise { try { const { stdout } = await execAsync('code --version'); @@ -104,9 +61,7 @@ export class IdeInstaller { } } - /** - * 检测 VS Code Insiders - */ + /** 检测 VS Code Insiders */ private static async checkVsCodeInsiders(): Promise { try { const { stdout } = await execAsync('code-insiders --version'); @@ -121,9 +76,7 @@ export class IdeInstaller { } } - /** - * 检测 Cursor - */ + /** 检测 Cursor */ private static async checkCursor(): Promise { try { const { stdout } = await execAsync('cursor --version'); diff --git a/packages/cli/src/logging/Logger.ts b/packages/cli/src/logging/Logger.ts index d7798334b..8479acce7 100644 --- a/packages/cli/src/logging/Logger.ts +++ b/packages/cli/src/logging/Logger.ts @@ -186,10 +186,6 @@ export class Logger { Logger.globalDebugConfig = null; } - public setEnabled(enabled: boolean): void { - this.enabled = enabled; - } - private parseDebugFilter(debugValue: string | boolean): { enabled: boolean; filter?: { mode: 'include' | 'exclude'; categories: string[] }; diff --git a/packages/cli/src/logging/StreamDebugLogger.ts b/packages/cli/src/logging/StreamDebugLogger.ts index 8f633b551..9c197e9bf 100644 --- a/packages/cli/src/logging/StreamDebugLogger.ts +++ b/packages/cli/src/logging/StreamDebugLogger.ts @@ -1,9 +1,4 @@ -/** - * 临时流式调试日志 - * - * 专门用于调试流式响应问题,写入独立文件便于分析 - * 调试完成后删除此文件 - */ +/** 临时流式调试日志 专门用于调试流式响应问题,写入独立文件便于分析 调试完成后删除此文件 */ import { appendFileSync, mkdirSync, writeFileSync } from 'node:fs'; import os from 'node:os'; diff --git a/packages/cli/src/lsp/LspClient.ts b/packages/cli/src/lsp/LspClient.ts index 762cdf04b..0895bae18 100644 --- a/packages/cli/src/lsp/LspClient.ts +++ b/packages/cli/src/lsp/LspClient.ts @@ -44,10 +44,6 @@ export class LspClient { private readonly onCrash: (error: Error) => void ) {} - get serverCapabilities(): ServerCapabilities | undefined { - return this.capabilities; - } - get isInitialized(): boolean { return this.initialized; } diff --git a/packages/cli/src/mcp/HealthMonitor.ts b/packages/cli/src/mcp/HealthMonitor.ts index 8bf850b49..772b909df 100644 --- a/packages/cli/src/mcp/HealthMonitor.ts +++ b/packages/cli/src/mcp/HealthMonitor.ts @@ -1,15 +1,10 @@ -/** - * MCP 健康监控 - * 周期性检查连接状态并自动触发重连 - */ +/** MCP 健康监控 周期性检查连接状态并自动触发重连 */ import { EventEmitter } from 'events'; import type { McpClient } from './McpClient.js'; import { McpConnectionStatus } from './types.js'; -/** - * 健康检查配置 - */ +/** 健康检查配置 */ export interface HealthCheckConfig { /** 检查间隔(毫秒),默认 30 秒 */ interval?: number; @@ -21,9 +16,7 @@ export interface HealthCheckConfig { failureThreshold?: number; } -/** - * 健康状态 - */ +/** 健康状态 */ export enum HealthStatus { HEALTHY = 'healthy', DEGRADED = 'degraded', // 有失败但未达到阈值 @@ -31,9 +24,7 @@ export enum HealthStatus { CHECKING = 'checking', } -/** - * 健康检查结果 - */ +/** 健康检查结果 */ export interface HealthCheckResult { status: HealthStatus; timestamp: number; @@ -61,9 +52,7 @@ function boundedInteger( return value; } -/** - * MCP 健康监控器 - */ +/** MCP 健康监控器 */ export class HealthMonitor extends EventEmitter { private client: McpClient; private config: Required; @@ -106,18 +95,14 @@ export class HealthMonitor extends EventEmitter { }; } - /** - * 启动健康监控 - */ + /** 启动健康监控 */ start(): void { if (this.running || !this.config.enabled) return; this.running = true; this.scheduleNextCheck(); } - /** - * 停止健康监控 - */ + /** 停止健康监控 */ stop(): void { this.running = false; if (this.checkTimer) { @@ -126,9 +111,7 @@ export class HealthMonitor extends EventEmitter { } } - /** - * 调度下一次检查 - */ + /** 调度下一次检查 */ private scheduleNextCheck(): void { if (!this.running) return; this.checkTimer = setTimeout(async () => { @@ -139,9 +122,7 @@ export class HealthMonitor extends EventEmitter { this.checkTimer.unref(); } - /** - * 执行健康检查 - */ + /** 执行健康检查 */ async performHealthCheck(): Promise { if (this.isChecking) { return this.getLastResult(); @@ -204,9 +185,7 @@ export class HealthMonitor extends EventEmitter { } } - /** - * 设置状态 - */ + /** 设置状态 */ private setStatus(status: HealthStatus): void { if (this.currentStatus !== status) { const oldStatus = this.currentStatus; @@ -215,16 +194,7 @@ export class HealthMonitor extends EventEmitter { } } - /** - * 获取当前状态 - */ - getStatus(): HealthStatus { - return this.currentStatus; - } - - /** - * 获取最后检查结果 - */ + /** 获取最后检查结果 */ getLastResult(): HealthCheckResult { return { status: this.currentStatus, @@ -233,29 +203,12 @@ export class HealthMonitor extends EventEmitter { }; } - /** - * 获取统计信息 - */ - getStatistics() { - return { - status: this.currentStatus, - consecutiveFailures: this.consecutiveFailures, - lastCheckTime: this.lastCheckTime, - isChecking: this.isChecking, - config: this.config, - }; - } - - /** - * 立即执行健康检查 - */ + /** 立即执行健康检查 */ async checkNow(): Promise { return this.performHealthCheck(); } - /** - * 重置失败计数 - */ + /** 重置失败计数 */ resetFailureCount(): void { this.consecutiveFailures = 0; if (this.currentStatus !== HealthStatus.CHECKING) { diff --git a/packages/cli/src/mcp/McpClient.ts b/packages/cli/src/mcp/McpClient.ts index 90d2eaf78..80043a62f 100644 --- a/packages/cli/src/mcp/McpClient.ts +++ b/packages/cli/src/mcp/McpClient.ts @@ -142,9 +142,7 @@ import { type McpToolDefinition, } from './types.js'; -/** - * 错误类型枚举 - */ +/** 错误类型枚举 */ export enum ErrorType { NETWORK_TEMPORARY = 'network_temporary', // 临时网络错误(可重试) NETWORK_PERMANENT = 'network_permanent', // 永久网络错误 @@ -154,9 +152,7 @@ export enum ErrorType { UNKNOWN = 'unknown', // 未知错误 } -/** - * 分类后的错误 - */ +/** 分类后的错误 */ interface ClassifiedError { type: ErrorType; isRetryable: boolean; @@ -202,9 +198,7 @@ export interface McpClientContentCatalogChange extends McpContentCatalogDelta { reason: 'initial' | 'notification' | 'manual'; } -/** - * 错误分类函数 - */ +/** 错误分类函数 */ function classifyError(error: unknown): ClassifiedError { if ( error instanceof McpOAuthAuthorizationRequiredError || @@ -292,9 +286,7 @@ function classifyError(error: unknown): ClassifiedError { }; } -/** - * MCP客户端 - */ +/** MCP客户端 */ export class McpClient extends EventEmitter { private status: McpConnectionStatus = McpConnectionStatus.DISCONNECTED; private sdkClient: Client | null = null; @@ -418,24 +410,12 @@ export class McpClient extends EventEmitter { return { ...this.taskPolicy }; } - get server(): { name: string; version: string } | null { - return this.serverInfo; - } - get instructions(): McpServerInstruction | undefined { return this.serverInstructions ? structuredClone(this.serverInstructions) : undefined; } - get healthCheck(): HealthMonitor | null { - return this.healthMonitor; - } - - get recovery(): McpClientConnectionLifecycleChange | undefined { - return this.recoveryState ? structuredClone(this.recoveryState) : undefined; - } - get logging(): McpLoggingPolicy { return { ...this.loggingPolicy }; } @@ -477,9 +457,7 @@ export class McpClient extends EventEmitter { await this.oauthProvider.logout(); } - /** - * 连接到MCP服务器(带重试) - */ + /** 连接到MCP服务器(带重试) */ async connect(): Promise { return this.connectWithRetry(3, 1000); } @@ -932,9 +910,7 @@ export class McpClient extends EventEmitter { } } - /** - * 断开连接 - */ + /** 断开连接 */ async disconnect(): Promise { this.desiredConnected = false; this.connectionGeneration++; @@ -962,9 +938,7 @@ export class McpClient extends EventEmitter { this.emit('disconnected'); } - /** - * 调用MCP工具 - */ + /** 调用MCP工具 */ async callTool( name: string, arguments_: Record = {}, @@ -1530,9 +1504,7 @@ export class McpClient extends EventEmitter { return response.approved ? 'accept' : 'decline'; } - /** - * 创建传输层(支持 OAuth) - */ + /** 创建传输层(支持 OAuth) */ private async createTransport(): Promise { const { type, command, args, env, cwd, url, headers, oauth } = this.config; let authProvider: OAuthProvider | undefined; @@ -1595,9 +1567,7 @@ export class McpClient extends EventEmitter { throw new Error(`不支持的传输类型: ${type}`); } - /** - * 加载工具列表 - */ + /** 加载工具列表 */ async refreshTools( reason: McpClientToolCatalogChange['reason'] = 'manual' ): Promise { @@ -1632,12 +1602,6 @@ export class McpClient extends EventEmitter { return promise; } - async waitForToolRefresh(): Promise { - while (this.toolRefreshPromise) { - await this.toolRefreshPromise; - } - } - private async runToolRefresh( initialReason: McpClientToolCatalogChange['reason'] ): Promise { @@ -2177,49 +2141,11 @@ export class McpClient extends EventEmitter { } } - /** - * 设置连接状态 - */ + /** 设置连接状态 */ private setStatus(status: McpConnectionStatus): void { const oldStatus = this.status; if (oldStatus === status) return; this.status = status; this.emit('statusChanged', status, oldStatus); } - - // ======================================== - // 兼容性方法(保持与 Registry 的接口一致) - // ======================================== - - async initialize(): Promise { - return this.connect(); - } - - async destroy(): Promise { - return this.disconnect(); - } - - async connectToServer(serverId?: string): Promise { - return this.connect(); - } - - async disconnectFromServer(serverId?: string): Promise { - return this.disconnect(); - } - - async listResources(_serverId?: string): Promise { - return structuredClone(this.resources); - } - - async listResourceTemplates(): Promise { - return structuredClone(this.resourceTemplates); - } - - async listPrompts(): Promise { - return structuredClone(this.prompts); - } - - async listTools(serverId?: string): Promise { - return this.availableTools; - } } diff --git a/packages/cli/src/mcp/McpRegistry.ts b/packages/cli/src/mcp/McpRegistry.ts index 4157692c8..aa069d535 100644 --- a/packages/cli/src/mcp/McpRegistry.ts +++ b/packages/cli/src/mcp/McpRegistry.ts @@ -1,7 +1,7 @@ import { EventEmitter } from 'events'; import type { McpServerConfig } from '../config/types.js'; import type { Tool } from '../tools/types/index.js'; -import type { McpOAuthLoginHandle, McpOAuthStatus } from './auth/index.js'; +import type { McpOAuthLoginHandle } from './auth/index.js'; import { createMcpTool } from './createMcpTool.js'; import { McpClient, @@ -49,9 +49,7 @@ import { } from './McpToolCatalog.js'; import { McpConnectionStatus, type McpToolDefinition } from './types.js'; -/** - * MCP服务器信息 - */ +/** MCP服务器信息 */ export interface McpServerInfo { config: McpServerConfig; client: McpClient; @@ -132,14 +130,10 @@ export type McpRegisteredResourceTemplate = McpResourceTemplateDefinition & { }; export type McpRegisteredPrompt = McpPromptDefinition & { server: string }; -/** - * MCP注册表 - * 管理MCP服务器连接和工具发现 - */ +/** MCP注册表 管理MCP服务器连接和工具发现 */ export class McpRegistry extends EventEmitter { private static instance: McpRegistry | null = null; private servers: Map = new Map(); - private isDiscovering = false; private catalogRevision = 0; private contentCatalogRevision = 0; private connectionRevision = 0; @@ -152,9 +146,7 @@ export class McpRegistry extends EventEmitter { super(); } - /** - * 获取单例实例 - */ + /** 获取单例实例 */ static getInstance(): McpRegistry { if (!McpRegistry.instance) { McpRegistry.instance = new McpRegistry(); @@ -162,16 +154,12 @@ export class McpRegistry extends EventEmitter { return McpRegistry.instance; } - /** - * 创建由单个 runtime 独占的注册表,避免会话级 MCP 配置污染全局实例。 - */ + /** 创建由单个 runtime 独占的注册表,避免会话级 MCP 配置污染全局实例。 */ static createIsolated(runtimeOptions: McpClientRuntimeOptions = {}): McpRegistry { return new McpRegistry(runtimeOptions); } - /** - * 注册MCP服务器 - */ + /** 注册MCP服务器 */ async registerServer( name: string, config: McpServerConfig, @@ -209,9 +197,7 @@ export class McpRegistry extends EventEmitter { } } - /** - * 注销MCP服务器 - */ + /** 注销MCP服务器 */ async unregisterServer(name: string): Promise { const serverInfo = this.servers.get(name); if (!serverInfo) { @@ -230,9 +216,7 @@ export class McpRegistry extends EventEmitter { this.emit('serverUnregistered', name); } - /** - * 连接到指定服务器 - */ + /** 连接到指定服务器 */ async connectServer(name: string): Promise { const serverInfo = this.servers.get(name); if (!serverInfo) { @@ -256,9 +240,7 @@ export class McpRegistry extends EventEmitter { } } - /** - * 断开指定服务器 - */ + /** 断开指定服务器 */ async disconnectServer(name: string): Promise { const serverInfo = this.servers.get(name); if (!serverInfo) { @@ -270,10 +252,7 @@ export class McpRegistry extends EventEmitter { serverInfo.connectedAt = undefined; } - /** - * 重连指定服务器(用于从 ERROR 状态恢复) - * 这个方法可以在首次连接失败或意外断开后使用 - */ + /** 重连指定服务器(用于从 ERROR 状态恢复) 这个方法可以在首次连接失败或意外断开后使用 */ async reconnectServer(name: string): Promise { const serverInfo = this.servers.get(name); if (!serverInfo) { @@ -297,15 +276,6 @@ export class McpRegistry extends EventEmitter { } } - /** - * 获取所有可用工具。 - * - * 工具始终使用 mcp____ provider 名称。 - */ - async getAvailableTools(): Promise { - return [...this.projectedTools.values()]; - } - getCatalogSnapshot(): { revision: number; tools: Tool[]; @@ -473,19 +443,6 @@ export class McpRegistry extends EventEmitter { return McpTaskManager.getInstance().list(owner, serverName); } - getTask(taskId: string, owner: McpTaskOwner): McpTaskSnapshot | undefined { - return McpTaskManager.getInstance().get(taskId, owner); - } - - waitForTask( - taskId: string, - owner: McpTaskOwner, - timeoutMs: number, - signal?: AbortSignal - ): Promise { - return McpTaskManager.getInstance().wait(taskId, owner, timeoutMs, signal); - } - cancelTask( taskId: string, owner: McpTaskOwner, @@ -511,55 +468,21 @@ export class McpRegistry extends EventEmitter { server.logging = server.client.logging; } - /** - * 根据名称查找工具 - */ + /** 根据名称查找工具 */ async findTool(toolName: string): Promise { return this.projectedTools.get(toolName) ?? null; } - /** - * 按服务器获取工具 - */ - getToolsByServer(serverName: string): Tool[] { - const serverInfo = this.servers.get(serverName); - if (!serverInfo || serverInfo.status !== McpConnectionStatus.CONNECTED) { - return []; - } - - return serverInfo.tools.map((mcpTool) => - createMcpTool( - serverInfo.client, - serverName, - mcpTool, - createMcpProviderToolName(serverName, mcpTool.name), - this.runtimeOptions.artifactWriter - ) - ); - } - - /** - * 获取服务器状态 - */ + /** 获取服务器状态 */ getServerStatus(name: string): McpServerInfo | null { return this.servers.get(name) || null; } - /** - * 获取所有服务器信息 - */ + /** 获取所有服务器信息 */ getAllServers(): Map { return new Map(this.servers); } - async getServerOAuthStatus(name: string): Promise { - const serverInfo = this.servers.get(name); - if (!serverInfo) { - throw new Error(`MCP服务器 "${name}" 未注册`); - } - return serverInfo.client.getOAuthStatus(); - } - async beginOAuthLogin( name: string, options: { openBrowser?: boolean } = {} @@ -608,41 +531,7 @@ export class McpRegistry extends EventEmitter { this.emit('serverOAuthStatusChanged', name, 'unauthenticated'); } - /** - * 刷新所有服务器工具列表 - */ - async refreshAllTools(): Promise { - const refreshPromises: Promise[] = []; - - for (const [serverName, serverInfo] of this.servers) { - if (serverInfo.status === McpConnectionStatus.CONNECTED) { - refreshPromises.push(this.refreshServerTools(serverName)); - } - } - - await Promise.allSettled(refreshPromises); - } - - /** - * 刷新指定服务器工具列表 - */ - async refreshServerTools(name: string): Promise { - const serverInfo = this.servers.get(name); - if (!serverInfo || serverInfo.status !== McpConnectionStatus.CONNECTED) { - return; - } - - try { - await serverInfo.client.refreshTools('manual'); - } catch (error) { - console.warn(`刷新服务器 "${name}" 工具列表失败:`, error); - throw error; - } - } - - /** - * 设置客户端事件处理器 - */ + /** 设置客户端事件处理器 */ private setupClientEventHandlers( client: McpClient, serverInfo: McpServerInfo, @@ -992,74 +881,7 @@ export class McpRegistry extends EventEmitter { return server; } - /** - * 自动发现MCP服务器 (基础实现,可扩展) - */ - async discoverServers(): Promise { - if (this.isDiscovering) { - return Array.from(this.servers.values()); - } - - this.isDiscovering = true; - this.emit('discoveryStarted'); - - try { - // 这里可以实现自动发现逻辑 - // 例如扫描常见的MCP服务器安装位置 - // 或者读取配置文件中的服务器列表 - - // 目前返回已注册的服务器 - return Array.from(this.servers.values()); - } finally { - this.isDiscovering = false; - this.emit('discoveryCompleted'); - } - } - - /** - * 批量注册服务器 - */ - async registerServers(servers: Record): Promise { - const registrationPromises = Object.entries(servers).map(([name, config]) => - this.registerServer(name, config).catch((error) => { - console.warn(`注册MCP服务器 "${name}" 失败:`, error); - return error; - }) - ); - - await Promise.allSettled(registrationPromises); - } - - /** - * 获取统计信息 - */ - getStatistics() { - let connectedCount = 0; - let totalTools = 0; - let errorCount = 0; - - for (const serverInfo of this.servers.values()) { - if (serverInfo.status === McpConnectionStatus.CONNECTED) { - connectedCount++; - totalTools += serverInfo.tools.length; - } else if (serverInfo.status === McpConnectionStatus.ERROR) { - errorCount++; - } - } - - return { - totalServers: this.servers.size, - connectedServers: connectedCount, - errorServers: errorCount, - totalTools, - isDiscovering: this.isDiscovering, - }; - } - - /** - * 断开所有 MCP 服务器连接 - * 在应用退出时调用 - */ + /** 断开所有 MCP 服务器连接 在应用退出时调用 */ async disconnectAll(): Promise { const disconnectPromises: Promise[] = []; diff --git a/packages/cli/src/mcp/createMcpTool.ts b/packages/cli/src/mcp/createMcpTool.ts index 5a643fe5a..ab1581c28 100644 --- a/packages/cli/src/mcp/createMcpTool.ts +++ b/packages/cli/src/mcp/createMcpTool.ts @@ -12,9 +12,7 @@ import { } from './McpToolResult.js'; import type { McpToolDefinition } from './types.js'; -/** - * 将 MCP 工具定义转换为 Blade Tool 实例 - */ +/** 将 MCP 工具定义转换为 Blade Tool 实例 */ export function createMcpTool( mcpClient: McpClient, serverName: string, diff --git a/packages/cli/src/mcp/loadMcpConfig.ts b/packages/cli/src/mcp/loadMcpConfig.ts index 384b04361..a0763e423 100644 --- a/packages/cli/src/mcp/loadMcpConfig.ts +++ b/packages/cli/src/mcp/loadMcpConfig.ts @@ -1,11 +1,6 @@ /** - * MCP 配置加载器 - * - * 职责: - * - 从 CLI --mcp-config 参数加载 MCP 配置 - * - 支持 JSON 文件路径或 JSON 字符串 - * - 提供无副作用解析器,供 SessionRuntime 构造会话级 MCP 配置 - * - 保留 Store 注入兼容入口 + * MCP 配置加载器

职责: - 从 CLI --mcp-config 参数加载 MCP 配置 - 支持 JSON 文件路径或 JSON 字符串 - + * 提供无副作用解析器,供 SessionRuntime 构造会话级 MCP 配置 - 保留 Store 注入兼容入口 */ import fs from 'fs/promises'; @@ -13,7 +8,6 @@ import path from 'path'; import { getOriginalCwd } from '../bootstrap/state.js'; import type { McpServerConfig } from '../config/types.js'; import { createLogger, LogCategory } from '../logging/Logger.js'; -import { getMcpServers, getState } from '../store/vanilla.js'; const logger = createLogger(LogCategory.GENERAL); @@ -76,17 +70,3 @@ export async function resolveMcpConfigFromCli( } return servers; } - -/** - * 从 CLI --mcp-config 参数加载 MCP 配置 - * 支持多种格式: - * - JSON 文件路径: "./mcp-config.json" - * - JSON 字符串 (单个服务器): '{"name": "xxx", "type": "stdio", "command": "xxx"}' - * - JSON 字符串 (多个服务器): '{"server1": {...}, "server2": {...}}' - * - * @param mcpConfigs - CLI 参数数组 - */ -export async function loadMcpConfigFromCli(mcpConfigs: string[]): Promise { - const updatedServers = await resolveMcpConfigFromCli(mcpConfigs, getMcpServers()); - getState().config.actions.updateConfig({ mcpServers: updatedServers }); -} diff --git a/packages/cli/src/mcp/types.ts b/packages/cli/src/mcp/types.ts index 59b5606fd..229d3d32c 100644 --- a/packages/cli/src/mcp/types.ts +++ b/packages/cli/src/mcp/types.ts @@ -1,13 +1,8 @@ -/** - * MCP (Model Context Protocol) 类型定义 - * 基于MCP协议规范的TypeScript接口 - */ +/** MCP (Model Context Protocol) 类型定义 基于MCP协议规范的TypeScript接口 */ import type { JSONSchema7 } from 'json-schema'; -/** - * MCP连接状态 - */ +/** MCP连接状态 */ export enum McpConnectionStatus { DISCONNECTED = 'disconnected', CONNECTING = 'connecting', @@ -16,9 +11,7 @@ export enum McpConnectionStatus { ERROR = 'error', } -/** - * MCP工具定义 - */ +/** MCP工具定义 */ export interface McpToolDefinition { name: string; description: string; @@ -26,9 +19,7 @@ export interface McpToolDefinition { taskSupport?: 'required' | 'optional' | 'forbidden'; } -/** - * MCP工具调用响应 - */ +/** MCP工具调用响应 */ export interface McpToolCallResponse { content: Array<{ type: 'text' | 'image' | 'audio' | 'resource' | 'resource_link'; diff --git a/packages/cli/src/memory/AutoMemoryManager.ts b/packages/cli/src/memory/AutoMemoryManager.ts index 8e9ed4f25..1c262d38c 100644 --- a/packages/cli/src/memory/AutoMemoryManager.ts +++ b/packages/cli/src/memory/AutoMemoryManager.ts @@ -127,18 +127,14 @@ export class AutoMemoryManager { this.config = { ...DEFAULT_AUTO_MEMORY_CONFIG, ...config }; } - /** - * 确保 memory 目录存在 - */ + /** 确保 memory 目录存在 */ async initialize(): Promise { if (this.initialized) return; await fs.mkdir(this.memoryDir, { recursive: true }); this.initialized = true; } - /** - * 加载 MEMORY.md 前 N 行,用于注入 system prompt - */ + /** 加载 MEMORY.md 前 N 行,用于注入 system prompt */ async loadIndex(): Promise { if (!this.config.enabled) return null; @@ -172,9 +168,7 @@ export class AutoMemoryManager { } } - /** - * 读取主题文件 - */ + /** 读取主题文件 */ async readTopic(topic: string): Promise { await this.initialize(); const filePath = this.resolveTopicPath(topic); @@ -193,9 +187,7 @@ export class AutoMemoryManager { } } - /** - * 写入主题文件 - */ + /** 写入主题文件 */ async writeTopic( topic: string, content: string, @@ -218,9 +210,7 @@ export class AutoMemoryManager { } } - /** - * 更新 MEMORY.md 索引 - */ + /** 更新 MEMORY.md 索引 */ async updateIndex( content: string, mode: 'overwrite' | 'append' = 'overwrite' @@ -307,9 +297,7 @@ export class AutoMemoryManager { }); } - /** - * 列出所有主题文件 - */ + /** 列出所有主题文件 */ async listTopics(): Promise { await this.initialize(); @@ -336,9 +324,7 @@ export class AutoMemoryManager { } } - /** - * 删除主题文件 - */ + /** 删除主题文件 */ async deleteTopic(topic: string): Promise { const filePath = this.resolveTopicPath(topic); try { @@ -349,9 +335,7 @@ export class AutoMemoryManager { } } - /** - * 清空所有记忆 - */ + /** 清空所有记忆 */ async clearAll(): Promise { const topics = await this.listTopics(); let count = 0; @@ -367,16 +351,12 @@ export class AutoMemoryManager { return count; } - /** - * 获取 memory 目录路径 - */ + /** 获取 memory 目录路径 */ getMemoryDir(): string { return this.memoryDir; } - /** - * 解析主题文件路径,防止路径穿越 - */ + /** 解析主题文件路径,防止路径穿越 */ private resolveTopicPath(topic: string): string { // 安全:只允许简单文件名,不允许路径分隔符 const safeName = topic.replace(/[/\\:*?"<>|]/g, '-'); diff --git a/packages/cli/src/memory/MemoryConsolidation.ts b/packages/cli/src/memory/MemoryConsolidation.ts index 106ab1d9d..eae981905 100644 --- a/packages/cli/src/memory/MemoryConsolidation.ts +++ b/packages/cli/src/memory/MemoryConsolidation.ts @@ -1,9 +1,7 @@ /** - * Memory consolidation for reusable project knowledge. - * - * Planning is pure and bounded. Persistence is explicit, workspace-scoped, and - * best-effort so a memory failure never turns a completed compaction into a task - * failure. + * Memory consolidation for reusable project knowledge.

Planning is pure and + * bounded. Persistence is explicit, workspace-scoped, and best-effort so a memory + * failure never turns a completed compaction into a task failure. */ import type { @@ -189,21 +187,3 @@ export async function commitMemoryConsolidation( return { outcome: 'failed', entries: 0, topics: [] }; } } - -/** Compatibility helper for internal callers migrating to the plan API. */ -export function extractLearnings(messages: Message[]): Map { - const result = new Map(); - for (const entry of planMemoryConsolidation(messages).entries) { - const values = result.get(entry.topic) ?? []; - values.push(entry.content); - result.set(entry.topic, values); - } - return result; -} - -export async function consolidateAfterCompaction( - discardedMessages: Message[], - options: MemoryConsolidationCommitOptions -): Promise { - return commitMemoryConsolidation(planMemoryConsolidation(discardedMessages), options); -} diff --git a/packages/cli/src/memory/index.ts b/packages/cli/src/memory/index.ts deleted file mode 100644 index 26cc5ddd0..000000000 --- a/packages/cli/src/memory/index.ts +++ /dev/null @@ -1,18 +0,0 @@ -export { AutoMemoryManager } from './AutoMemoryManager.js'; -export type { - MemoryConsolidationEntry, - MemoryConsolidationPlan, -} from './MemoryConsolidation.js'; -export type { - MemoryConsolidationProjection, - MemoryConsolidationTopic, -} from '../api/memoryConsolidationSchemas.js'; -export { - commitMemoryConsolidation, - EMPTY_MEMORY_CONSOLIDATION_PLAN, - planMemoryConsolidation, -} from './MemoryConsolidation.js'; -export type { MemorySafetyResult } from './MemorySafety.js'; -export { classifyMemoryContent } from './MemorySafety.js'; -export type { AutoMemoryConfig, MemoryTopicInfo } from './types.js'; -export { DEFAULT_AUTO_MEMORY_CONFIG } from './types.js'; diff --git a/packages/cli/src/memory/types.ts b/packages/cli/src/memory/types.ts index 19f62b50c..c45379f98 100644 --- a/packages/cli/src/memory/types.ts +++ b/packages/cli/src/memory/types.ts @@ -1,6 +1,4 @@ -/** - * Auto Memory 类型定义 - */ +/** Auto Memory 类型定义 */ export interface AutoMemoryConfig { /** 是否启用 Auto Memory */ diff --git a/packages/cli/src/plugins/PluginIntegrator.ts b/packages/cli/src/plugins/PluginIntegrator.ts index e4898cfe2..716a7a162 100644 --- a/packages/cli/src/plugins/PluginIntegrator.ts +++ b/packages/cli/src/plugins/PluginIntegrator.ts @@ -1,8 +1,7 @@ /** - * Blade Code Plugins System - Plugin Integrator - * - * This module is responsible for integrating loaded plugins into - * the existing subsystems (commands, skills, agents, hooks, MCP). + * Blade Code Plugins System - Plugin Integrator

This module is responsible for + * integrating loaded plugins into the existing subsystems (commands, skills, agents, + * hooks, MCP). */ import path from 'node:path'; @@ -57,9 +56,7 @@ function cloneHookConfig(config: Readonly): HookConfig { return cloned; } -/** - * Integration result for a single plugin - */ +/** Integration result for a single plugin */ interface PluginIntegrationResult { pluginName: string; commandsRegistered: number; @@ -71,9 +68,7 @@ interface PluginIntegrationResult { errors: string[]; } -/** - * Overall integration result - */ +/** Overall integration result */ interface IntegrationResult { plugins: PluginIntegrationResult[]; totalCommands: number; @@ -267,9 +262,7 @@ class PluginIntegrator { return count; } - /** - * Integrate skills from a plugin - */ + /** Integrate skills from a plugin */ private integrateSkills(plugin: LoadedPlugin): number { let count = 0; @@ -281,9 +274,7 @@ class PluginIntegrator { return count; } - /** - * Integrate agents from a plugin - */ + /** Integrate agents from a plugin */ private integrateAgents(plugin: LoadedPlugin): number { let count = 0; @@ -354,9 +345,7 @@ class PluginIntegrator { this.hookManager.loadConfig(effective, this.workspaceRoot); } - /** - * Integrate MCP servers from a plugin - */ + /** Integrate MCP servers from a plugin */ private integrateMcp(plugin: LoadedPlugin): number { // SessionRuntime resolves these definitions for its exact workspace and // connects them through an isolated registry. @@ -364,9 +353,7 @@ class PluginIntegrator { } } -/** - * Convenience function to integrate all plugins - */ +/** Convenience function to integrate all plugins */ export async function integrateAllPlugins( workspaceRoot: string = getCwd() ): Promise { @@ -374,9 +361,7 @@ export async function integrateAllPlugins( return integrator.integrateAll(); } -/** - * Convenience function to clear all plugin resources - */ +/** Convenience function to clear all plugin resources */ export function clearAllPluginResources(workspaceRoot: string = getCwd()): void { const integrator = new PluginIntegrator(workspaceRoot); integrator.clearAllPluginResources(); diff --git a/packages/cli/src/plugins/PluginLoader.ts b/packages/cli/src/plugins/PluginLoader.ts index 2c84262f5..0565e448d 100644 --- a/packages/cli/src/plugins/PluginLoader.ts +++ b/packages/cli/src/plugins/PluginLoader.ts @@ -1,8 +1,7 @@ /** - * Blade Code Plugins System - Plugin Loader - * - * This module is responsible for loading plugins from directories, - * including their commands, agents, skills, hooks, and MCP configurations. + * Blade Code Plugins System - Plugin Loader

This module is responsible for loading + * plugins from directories, including their commands, agents, skills, hooks, and MCP + * configurations. */ import * as fs from 'node:fs/promises'; @@ -109,9 +108,7 @@ export class PluginLoader { }; } - /** - * Load commands from the commands/ directory - */ + /** Load commands from the commands/ directory */ private async loadCommands( pluginDir: string, pluginName: string @@ -143,9 +140,7 @@ export class PluginLoader { return commands; } - /** - * Parse a command file - */ + /** Parse a command file */ private async parseCommandFile( filePath: string, basePath: string, @@ -184,9 +179,7 @@ export class PluginLoader { }; } - /** - * Normalize command frontmatter to CustomCommandConfig - */ + /** Normalize command frontmatter to CustomCommandConfig */ private normalizeCommandConfig(data: Record): CustomCommandConfig { return { description: this.asString(data.description), @@ -197,9 +190,7 @@ export class PluginLoader { }; } - /** - * Load agents from the agents/ directory - */ + /** Load agents from the agents/ directory */ private async loadAgents( pluginDir: string, pluginName: string @@ -231,9 +222,7 @@ export class PluginLoader { return agents; } - /** - * Parse an agent file - */ + /** Parse an agent file */ private async parseAgentFile( filePath: string, basePath: string, @@ -273,9 +262,7 @@ export class PluginLoader { }; } - /** - * Load skills from the skills/ directory - */ + /** Load skills from the skills/ directory */ private async loadSkills( pluginDir: string, pluginName: string @@ -325,9 +312,7 @@ export class PluginLoader { return skills; } - /** - * Load hooks configuration from hooks/hooks.json - */ + /** Load hooks configuration from hooks/hooks.json */ private async loadHooks(pluginDir: string): Promise { const hooksPath = path.join(pluginDir, 'hooks', 'hooks.json'); @@ -344,9 +329,7 @@ export class PluginLoader { } } - /** - * Load MCP configuration from .mcp.json - */ + /** Load MCP configuration from .mcp.json */ private async loadMcpConfig( pluginDir: string, pluginName: string @@ -425,9 +408,7 @@ export class PluginLoader { } } - /** - * Recursively scan for .md files in a directory - */ + /** Recursively scan for .md files in a directory */ private async scanMarkdownFiles(dir: string): Promise { const files: string[] = []; @@ -449,9 +430,7 @@ export class PluginLoader { return files; } - /** - * Check if a directory exists - */ + /** Check if a directory exists */ private async dirExists(dirPath: string): Promise { try { const stat = await fs.stat(dirPath); @@ -461,9 +440,7 @@ export class PluginLoader { } } - /** - * Safely convert value to string - */ + /** Safely convert value to string */ private asString(value: unknown): string | undefined { if (typeof value === 'string' && value.trim()) { return value.trim(); @@ -471,9 +448,7 @@ export class PluginLoader { return undefined; } - /** - * Parse a string array from various formats - */ + /** Parse a string array from various formats */ private parseStringArray(value: unknown): string[] | undefined { if (!value) return undefined; diff --git a/packages/cli/src/plugins/PluginManifest.ts b/packages/cli/src/plugins/PluginManifest.ts index a01f84ce2..9b68192c3 100644 --- a/packages/cli/src/plugins/PluginManifest.ts +++ b/packages/cli/src/plugins/PluginManifest.ts @@ -1,8 +1,7 @@ /** - * Blade Code Plugins System - Plugin Manifest Parser - * - * This module handles parsing and validation of plugin.json manifest files. - * It supports both .blade-plugin/ and .claude-plugin/ directories. + * Blade Code Plugins System - Plugin Manifest Parser

This module handles parsing + * and validation of plugin.json manifest files. It supports both .blade-plugin/ and + * .claude-plugin/ directories. */ import * as fs from 'node:fs/promises'; @@ -13,9 +12,7 @@ import { validatePluginManifestConstraints } from './PluginCompatibility.js'; import { pluginManifestSchema } from './schemas.js'; import type { ManifestSource, PluginManifest } from './types.js'; -/** - * Result of parsing a plugin manifest - */ +/** Result of parsing a plugin manifest */ export interface ParseManifestResult { /** The parsed manifest */ manifest: PluginManifest; @@ -99,21 +96,3 @@ export async function parsePluginManifest( // No manifest found in any directory return null; } - -/** - * Check if a directory is a valid plugin directory - * - * A valid plugin directory must contain a plugin.json in either - * .blade-plugin/ or .claude-plugin/ subdirectory. - * - * @param dirPath - Path to check - * @returns True if the directory is a valid plugin - */ -export async function isValidPluginDir(dirPath: string): Promise { - try { - const result = await parsePluginManifest(dirPath); - return result !== null; - } catch { - return false; - } -} diff --git a/packages/cli/src/plugins/PluginRegistry.ts b/packages/cli/src/plugins/PluginRegistry.ts index c42c35dc9..752c6f5eb 100644 --- a/packages/cli/src/plugins/PluginRegistry.ts +++ b/packages/cli/src/plugins/PluginRegistry.ts @@ -1,8 +1,7 @@ /** - * Blade Code Plugins System - Plugin Registry - * - * This module provides a singleton registry for managing loaded plugins. - * It handles plugin discovery, loading, and provides lookup methods. + * Blade Code Plugins System - Plugin Registry

This module provides a singleton + * registry for managing loaded plugins. It handles plugin discovery, loading, and + * provides lookup methods. */ import path from 'node:path'; @@ -21,11 +20,9 @@ import { } from './PluginSourcePolicy.js'; import type { LoadedPlugin, - PluginAgent, PluginCommand, PluginDiscoveryResult, PluginMarketplaceRecord, - PluginSkill, PluginSource, } from './types.js'; @@ -56,9 +53,7 @@ export class PluginRegistry { this.workspaceRoot = workspaceRoot; } - /** - * Get the singleton instance - */ + /** Get the singleton instance */ static getInstance(workspaceRoot: string = getCwd()): PluginRegistry { const key = path.resolve(workspaceRoot); let registry = PluginRegistry.instances.get(key); @@ -69,9 +64,7 @@ export class PluginRegistry { return registry; } - /** - * Reset the singleton instance (mainly for testing) - */ + /** Reset the singleton instance (mainly for testing) */ static resetInstance(): void { PluginRegistry.instances.clear(); } @@ -217,9 +210,7 @@ export class PluginRegistry { }; } - /** - * Check if the registry has been initialized - */ + /** Check if the registry has been initialized */ isInitialized(): boolean { return this.initialized; } @@ -237,34 +228,21 @@ export class PluginRegistry { }; } - /** - * Get all loaded plugins - */ + /** Get all loaded plugins */ getAll(): LoadedPlugin[] { return Array.from(this.plugins.values()); } - /** - * Get all active plugins - */ + /** Get all active plugins */ getActive(): LoadedPlugin[] { return Array.from(this.plugins.values()).filter((p) => p.status === 'active'); } - /** - * Get a plugin by name - */ + /** Get a plugin by name */ get(name: string): LoadedPlugin | undefined { return this.plugins.get(name); } - /** - * Check if a plugin exists - */ - has(name: string): boolean { - return this.plugins.has(name); - } - async reapplyEnabledSettings(): Promise { this.enabledSettings = await ConfigManager.getInstance().loadWorkspacePluginSettings(this.workspaceRoot); @@ -319,9 +297,7 @@ export class PluginRegistry { ); } - /** - * Get plugins grouped by source - */ + /** Get plugins grouped by source */ getBySource(): Record { const result: Record = { cli: [], @@ -336,9 +312,7 @@ export class PluginRegistry { return result; } - /** - * Get all namespaced commands from all active plugins - */ + /** Get all namespaced commands from all active plugins */ getAllCommands(): PluginCommand[] { const commands: PluginCommand[] = []; @@ -351,198 +325,6 @@ export class PluginRegistry { return commands; } - /** - * Get all namespaced skills from all active plugins - */ - getAllSkills(): PluginSkill[] { - const skills: PluginSkill[] = []; - - for (const plugin of this.plugins.values()) { - if (plugin.status === 'active') { - skills.push(...plugin.skills); - } - } - - return skills; - } - - /** - * Get all namespaced agents from all active plugins - */ - getAllAgents(): PluginAgent[] { - const agents: PluginAgent[] = []; - - for (const plugin of this.plugins.values()) { - if (plugin.status === 'active') { - agents.push(...plugin.agents); - } - } - - return agents; - } - - /** - * Find a command by name - * - * Supports: - * - Full namespaced name: "plugin:command" - * - Short name if unique: "command" - * - * @param name - Command name to find - * @returns Plugin command or undefined - */ - findCommand(name: string): PluginCommand | undefined { - // Try exact namespaced match first - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const cmd of plugin.commands) { - if (cmd.namespacedName === name) { - return cmd; - } - } - } - - // Try short name match (if unique) - const matches: PluginCommand[] = []; - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const cmd of plugin.commands) { - if (cmd.originalName === name) { - matches.push(cmd); - } - } - } - - // Only return if exactly one match - if (matches.length === 1) { - return matches[0]; - } - - return undefined; - } - - /** - * Find a skill by name - * - * @param name - Skill name (namespaced or short) - * @returns Plugin skill or undefined - */ - findSkill(name: string): PluginSkill | undefined { - // Try exact namespaced match first - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const skill of plugin.skills) { - if (skill.namespacedName === name) { - return skill; - } - } - } - - // Try short name match - const matches: PluginSkill[] = []; - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const skill of plugin.skills) { - if (skill.originalName === name) { - matches.push(skill); - } - } - } - - if (matches.length === 1) { - return matches[0]; - } - - return undefined; - } - - /** - * Find an agent by name - * - * @param name - Agent name (namespaced or short) - * @returns Plugin agent or undefined - */ - findAgent(name: string): PluginAgent | undefined { - // Try exact namespaced match first - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const agent of plugin.agents) { - if (agent.namespacedName === name) { - return agent; - } - } - } - - // Try short name match - const matches: PluginAgent[] = []; - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const agent of plugin.agents) { - if (agent.originalName === name) { - matches.push(agent); - } - } - } - - if (matches.length === 1) { - return matches[0]; - } - - return undefined; - } - - /** - * Check if a command name has multiple matches (conflict) - * - * @param shortName - Short command name - * @returns True if multiple plugins provide this command - */ - hasCommandConflict(shortName: string): boolean { - let count = 0; - - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const cmd of plugin.commands) { - if (cmd.originalName === shortName) { - count++; - if (count > 1) return true; - } - } - } - - return false; - } - - /** - * Get all plugins that provide a command with the given short name - * - * @param shortName - Short command name - * @returns Array of plugin names - */ - getCommandProviders(shortName: string): string[] { - const providers: string[] = []; - - for (const plugin of this.plugins.values()) { - if (plugin.status !== 'active') continue; - - for (const cmd of plugin.commands) { - if (cmd.originalName === shortName) { - providers.push(plugin.manifest.name); - break; - } - } - } - - return providers; - } - /** * Disable a plugin * @@ -559,22 +341,6 @@ export class PluginRegistry { return false; } - /** - * Enable a plugin - * - * @param name - Plugin name - * @returns True if the plugin was enabled - */ - enable(name: string): boolean { - const plugin = this.plugins.get(name); - if (plugin && plugin.status === 'inactive') { - plugin.status = 'active'; - logger.info(`Plugin "${name}" enabled`); - return true; - } - return false; - } - /** * Refresh the plugin list * @@ -589,9 +355,7 @@ export class PluginRegistry { return this.initialize(this.workspaceRoot, this.cliPluginDirs); } - /** - * Get plugin statistics - */ + /** Get plugin statistics */ getStats(): { total: number; active: number; @@ -626,56 +390,9 @@ export class PluginRegistry { agents, }; } - - /** - * Generate a formatted list of plugins for display - */ - formatPluginList(): string { - const plugins = this.getAll(); - - if (plugins.length === 0) { - return '没有已加载的插件。'; - } - - const lines: string[] = []; - const bySource = this.getBySource(); - - if (bySource.cli.length > 0) { - lines.push('## CLI 指定的插件'); - for (const p of bySource.cli) { - const status = p.status === 'inactive' ? ' (禁用)' : ''; - lines.push(`- **${p.manifest.name}** v${p.manifest.version}${status}`); - lines.push(` ${p.manifest.description}`); - } - lines.push(''); - } - - if (bySource.project.length > 0) { - lines.push('## 项目级插件'); - for (const p of bySource.project) { - const status = p.status === 'inactive' ? ' (禁用)' : ''; - lines.push(`- **${p.manifest.name}** v${p.manifest.version}${status}`); - lines.push(` ${p.manifest.description}`); - } - lines.push(''); - } - - if (bySource.user.length > 0) { - lines.push('## 用户级插件'); - for (const p of bySource.user) { - const status = p.status === 'inactive' ? ' (禁用)' : ''; - lines.push(`- **${p.manifest.name}** v${p.manifest.version}${status}`); - lines.push(` ${p.manifest.description}`); - } - } - - return lines.join('\n'); - } } -/** - * Convenience function to get the plugin registry instance - */ +/** Convenience function to get the plugin registry instance */ export function getPluginRegistry(workspaceRoot: string = getCwd()): PluginRegistry { return PluginRegistry.getInstance(workspaceRoot); } diff --git a/packages/cli/src/plugins/PluginSourcePolicy.ts b/packages/cli/src/plugins/PluginSourcePolicy.ts index cc153c3be..b05fca7b4 100644 --- a/packages/cli/src/plugins/PluginSourcePolicy.ts +++ b/packages/cli/src/plugins/PluginSourcePolicy.ts @@ -133,7 +133,3 @@ export function assertPluginSourceAllowed( `Marketplace "${source.marketplace}"` ); } - -export function isFullPluginGitSha(value: string | undefined): boolean { - return value !== undefined && FULL_GIT_SHA_PATTERN.test(value); -} diff --git a/packages/cli/src/plugins/namespacing.ts b/packages/cli/src/plugins/namespacing.ts index 16b430695..5ba3c0c99 100644 --- a/packages/cli/src/plugins/namespacing.ts +++ b/packages/cli/src/plugins/namespacing.ts @@ -1,16 +1,11 @@ /** - * Blade Code Plugins System - Namespacing Utilities - * - * This module provides utilities for handling plugin namespacing. - * Plugin resources (commands, skills, agents) are namespaced to prevent conflicts. - * - * Format: plugin-name:resource-name - * Example: my-plugin:commit + * Blade Code Plugins System - Namespacing Utilities

This module provides utilities + * for handling plugin namespacing. Plugin resources (commands, skills, agents) are + * namespaced to prevent conflicts.

Format: plugin-name:resource-name Example: + * my-plugin:commit */ -/** - * Namespace separator used between plugin name and resource name - */ +/** Namespace separator used between plugin name and resource name */ const NAMESPACE_SEPARATOR = ':'; /** diff --git a/packages/cli/src/plugins/types.ts b/packages/cli/src/plugins/types.ts index ff708ac3c..136e22a87 100644 --- a/packages/cli/src/plugins/types.ts +++ b/packages/cli/src/plugins/types.ts @@ -1,19 +1,8 @@ -/** - * Blade Code Plugins System - Type Definitions - * - * This module defines the core types for the plugin system that allows - * extending Blade Code with custom commands, agents, skills, hooks, and MCP servers. - */ - import type { SubagentConfig } from '../agent/subagents/types.js'; import type { LspServerConfig, McpServerConfig } from '../config/types.js'; import type { HookConfig } from '../hooks/types/HookTypes.js'; import type { SkillMetadata } from '../skills/types.js'; import type { CustomCommandConfig } from '../slash-commands/custom/types.js'; - -/** - * Plugin author information - */ export interface PluginAuthor { name: string; email?: string; @@ -29,32 +18,14 @@ export interface PluginAuthor { export interface PluginManifest { /** Unique plugin identifier (used for namespacing), kebab-case, 2-64 chars */ name: string; - - /** Short description of the plugin */ description: string; - - /** Semantic version (e.g., "1.0.0") */ version: string; - - /** Author information */ author?: PluginAuthor; - - /** License identifier (e.g., "MIT", "Apache-2.0") */ license?: string; - - /** Repository URL */ repository?: string; - - /** Homepage URL */ homepage?: string; - - /** Keywords for search/discovery */ keywords?: string[]; - - /** Dependencies on other plugins (name -> version range) */ dependencies?: Record; - - /** Minimum required Blade version */ bladeVersion?: string; } @@ -72,7 +43,6 @@ export type PluginInstallSource = type: 'marketplace'; marketplace: string; }; - export interface InstalledPluginRecord { name: string; source: PluginInstallSource; @@ -94,7 +64,6 @@ export type PluginMarketplaceSource = type: 'local'; path: string; }; - export interface PluginMarketplaceRecord { name: string; source: PluginMarketplaceSource; @@ -140,31 +109,19 @@ export interface PluginPackageState { marketplaces: Record; } -/** - * Plugin source type indicating where the plugin was loaded from - */ export type PluginSource = | 'cli' // --plugin-dir argument (highest priority) | 'project' // .blade/plugins/ or .claude/plugins/ | 'user'; // ~/.blade/plugins/ or ~/.claude/plugins/ -/** - * The directory type where the manifest was found - */ export type ManifestSource = 'blade' | 'claude'; - -/** - * Plugin status - */ export type PluginStatus = 'active' | 'inactive' | 'error'; - export type PluginCompatibilityIssueCode = | 'blade-version' | 'dependency-missing' | 'dependency-version' | 'dependency-inactive' | 'source-policy'; - export interface PluginCompatibilityIssue { code: PluginCompatibilityIssueCode; message: string; @@ -173,107 +130,44 @@ export interface PluginCompatibilityIssue { actual?: string; } -/** - * A namespaced command from a plugin - */ export interface PluginCommand { - /** Original command name (e.g., "commit") */ originalName: string; - - /** Namespaced name (e.g., "my-plugin:commit") */ namespacedName: string; - - /** Plugin that provides this command */ pluginName: string; - - /** Command configuration from frontmatter */ config: CustomCommandConfig; - - /** Command content (markdown body) */ content: string; - - /** File path to the command .md file */ path: string; } -/** - * A namespaced skill from a plugin - */ export interface PluginSkill { - /** Original skill name */ originalName: string; - - /** Namespaced name (e.g., "my-plugin:my-skill") */ namespacedName: string; - - /** Plugin that provides this skill */ pluginName: string; - - /** Skill metadata */ metadata: SkillMetadata; - - /** Path to the skill directory */ path: string; } -/** - * A namespaced agent from a plugin - */ export interface PluginAgent { - /** Original agent name */ originalName: string; - - /** Namespaced name (e.g., "my-plugin:my-agent") */ namespacedName: string; - - /** Plugin that provides this agent */ pluginName: string; - - /** Agent configuration */ config: SubagentConfig; - - /** Path to the agent .md file */ path: string; } -/** - * A fully loaded plugin with all its resources - */ export interface LoadedPlugin { /** Plugin manifest from plugin.json */ manifest: PluginManifest; - - /** Absolute path to the plugin root directory */ basePath: string; - - /** Where the plugin was loaded from */ source: PluginSource; - - /** Which manifest directory was used (.blade-plugin or .claude-plugin) */ manifestSource: ManifestSource; - - /** Loaded commands */ commands: PluginCommand[]; - - /** Loaded agents */ agents: PluginAgent[]; - - /** Loaded skills */ skills: PluginSkill[]; - - /** Hooks configuration from hooks/hooks.json */ hooks?: HookConfig; - - /** MCP server configurations from .mcp.json */ mcpServers?: Record; - - /** LSP server configurations from .lsp.json */ lspServers?: Record; - - /** Plugin status */ status: PluginStatus; - - /** Error message if status is 'error' */ error?: string; /** Compatibility or source-policy reasons that prevent activation */ @@ -281,39 +175,20 @@ export interface LoadedPlugin { /** Immutable package-manager installation metadata */ installation?: InstalledPluginRecord; - - /** When the plugin was loaded */ loadedAt: Date; } -/** - * Result of plugin discovery - */ export interface PluginDiscoveryResult { - /** Successfully loaded plugins */ plugins: LoadedPlugin[]; - - /** Errors encountered during discovery */ errors: PluginDiscoveryError[]; } -/** - * Error during plugin discovery - */ export interface PluginDiscoveryError { - /** Path to the plugin directory */ path: string; - - /** Error message */ error: string; - - /** Error code for programmatic handling */ code?: PluginErrorCode; } -/** - * Plugin error codes - */ type PluginErrorCode = | 'INVALID_MANIFEST' // plugin.json is invalid | 'MANIFEST_NOT_FOUND' // No plugin.json found @@ -328,39 +203,17 @@ type PluginErrorCode = | 'DEPENDENCY_MISSING' // Required dependency not found | 'IO_ERROR'; // File system error -/** - * Options for plugin loading - */ export interface PluginLoadOptions { - /** Skip loading commands */ skipCommands?: boolean; - - /** Skip loading agents */ skipAgents?: boolean; - - /** Skip loading skills */ skipSkills?: boolean; - - /** Skip loading hooks */ skipHooks?: boolean; - - /** Skip loading MCP config */ skipMcp?: boolean; - - /** Skip loading LSP config */ skipLsp?: boolean; } -/** - * Directory info for plugin scanning - */ export interface PluginSearchDir { - /** Absolute path to the directory */ path: string; - - /** Source type */ source: PluginSource; - - /** Whether this is a Blade or Claude Code directory */ type: 'blade' | 'claude'; } diff --git a/packages/cli/src/prompts/builder.ts b/packages/cli/src/prompts/builder.ts index 181e04157..b91d29706 100644 --- a/packages/cli/src/prompts/builder.ts +++ b/packages/cli/src/prompts/builder.ts @@ -38,45 +38,29 @@ import { loadProjectInstructions } from './projectInstructions.js'; /** available_skills 占位符的正则表达式 */ const AVAILABLE_SKILLS_REGEX = /\s*<\/available_skills>/; -/** - * 提示词构建选项 - */ +/** 提示词构建选项 */ export interface BuildSystemPromptOptions { /** Host workspace resources available to this prompt build. */ workspaceAccess?: 'full' | 'none'; - /** - * 项目路径,用于查找分层项目指令 - */ + /** 项目路径,用于查找分层项目指令 */ projectPath?: string; - /** - * 替换默认提示(仅替换 DEFAULT_SYSTEM_PROMPT,不影响项目指令) - */ + /** 替换默认提示(仅替换 DEFAULT_SYSTEM_PROMPT,不影响项目指令) */ replaceDefault?: string; - /** - * 追加到提示词末尾 - */ + /** 追加到提示词末尾 */ append?: string; - /** - * 权限模式(Plan 模式会使用独立的 system prompt) - */ + /** 权限模式(Plan 模式会使用独立的 system prompt) */ mode?: PermissionMode; - /** - * 是否包含环境上下文(默认 true) - */ + /** 是否包含环境上下文(默认 true) */ includeEnvironment?: boolean; - /** - * 环境上下文选项 - */ + /** 环境上下文选项 */ environmentOptions?: EnvironmentContextOptions; - /** - * AI 回复语言(如 'zh-CN', 'en-US') - */ + /** AI 回复语言(如 'zh-CN', 'en-US') */ language?: string; /** @@ -85,14 +69,10 @@ export interface BuildSystemPromptOptions { */ projectTrusted?: boolean; - /** - * Immutable project-rule catalog owned by the current Session. - */ + /** Immutable project-rule catalog owned by the current Session. */ projectRuleCatalog?: ProjectRuleCatalog; - /** - * Source checkout path represented by the execution workspace. - */ + /** Source checkout path represented by the execution workspace. */ projectInstructionSourcePath?: string; /** @@ -107,24 +87,16 @@ export interface BuildSystemPromptOptions { */ communicationStyle?: CommunicationStyleSelection; - /** - * Immutable style catalog owned by the current Session. - */ + /** Immutable style catalog owned by the current Session. */ communicationStyleCatalog?: CommunicationStyleCatalog; } -/** - * 提示词构建结果 - */ +/** 提示词构建结果 */ export interface BuildSystemPromptResult { - /** - * 最终的系统提示词 - */ + /** 最终的系统提示词 */ prompt: string; - /** - * 各部分来源(用于调试) - */ + /** 各部分来源(用于调试) */ sources: Array<{ name: string; loaded: boolean; @@ -174,8 +146,7 @@ export async function buildSystemPrompt( const parts: string[] = []; const sources: BuildSystemPromptResult['sources'] = []; - // 1. 默认提示或替换内容 - // Plan 模式使用独立的 system prompt + // 1. 默认提示或替换内容 Plan 模式使用独立的 system prompt const isPlanMode = mode === PermissionMode.PLAN; let basePrompt: string; @@ -294,9 +265,7 @@ export async function buildSystemPrompt( return { prompt, sources }; } -/** - * 注入 Skills 列表到系统提示的 占位符 - */ +/** 注入 Skills 列表到系统提示的 占位符 */ function injectSkillsToPrompt( prompt: string, projectPath?: string, diff --git a/packages/cli/src/prompts/default.ts b/packages/cli/src/prompts/default.ts index a0c015f83..1ba8fdd66 100644 --- a/packages/cli/src/prompts/default.ts +++ b/packages/cli/src/prompts/default.ts @@ -1,9 +1,5 @@ /** - * 默认系统提示内容 - * - * 模块化 section 设计: - * - 各段由 sections.ts 的独立函数生成 - * - buildDefaultPrompt() 按固定顺序组装 + * 默认系统提示内容

模块化 section 设计: - 各段由 sections.ts 的独立函数生成 - buildDefaultPrompt() 按固定顺序组装 * - Skills / Auto Memory / Language 作为尾部段保留在此文件 */ @@ -74,21 +70,17 @@ You have a persistent memory system that survives across sessions. Your memories // ============================================================ /** - * 组装默认系统提示 - * - * 段顺序: - * 1. Intro — 身份 + 网络安全 - * 2. System — 工具结果、权限、hooks、上下文压缩 - * 3. Doing tasks — 软件工程任务指导 - * 4. Actions — 可逆性、爆炸半径 - * 5. Using your tools — 工具使用偏好 - * 6. Browser and GUI work — 原生浏览器优先与视觉兜底 - * 7. Tone and style — 格式、引用 - * 8. Output efficiency — 简洁输出 - * 9. Session-specific guidance — Agent/Explore、搜索策略、Skill - * 10. Skills — 可用技能列表 - * 11. Auto Memory — 持久记忆 - * 12. Language — 语言指令 + + * 组装默认系统提示

段顺序: 1. Intro — 身份 + 网络安全 2. System — 工具结果、权限、hooks、上下文压缩 3. Doing + + * tasks — 软件工程任务指导 4. Actions — 可逆性、爆炸半径 5. Using your tools — 工具使用偏好 6. Browser and + + * GUI work — 原生浏览器优先与视觉兜底 7. Tone and style — 格式、引用 8. Output efficiency — 简洁输出 9. + + * Session-specific guidance — Agent/Explore、搜索策略、Skill 10. Skills — 可用技能列表 11. Auto + + * Memory — 持久记忆 12. Language — 语言指令 + */ export function buildDefaultPrompt(): string { const sections = [ @@ -109,61 +101,14 @@ export function buildDefaultPrompt(): string { return sections.join('\n\n'); } -/** - * 向后兼容导出:buildDefaultPrompt() 的默认结果 - */ +/** 向后兼容导出:buildDefaultPrompt() 的默认结果 */ export const DEFAULT_SYSTEM_PROMPT = buildDefaultPrompt(); -/** - * Plan Mode System Prompt (Compact Version) - * 精简版:核心目标 + 关键约束 + 检查点 - * 解耦工具名:使用"只读探索代理"/"只读检索工具"等描述性语言 - */ -export const PLAN_MODE_SYSTEM_PROMPT = `You are in **PLAN MODE** - a read-only research phase for designing implementation plans. - -## Core Objective +import PLAN_MODE_SYSTEM_PROMPT from './plan-mode.md?raw'; -Research the codebase thoroughly, then create a detailed implementation plan. No file modifications allowed until plan is approved. +export { PLAN_MODE_SYSTEM_PROMPT }; -## Key Constraints - -1. **Read-only tools only**: File readers, search tools, web fetchers, and exploration subagents -2. **Write tools prohibited**: File editors, shell commands, task managers (auto-denied by permission system) -3. **Text output required**: You MUST output text summaries between tool calls - never call 3+ tools without explaining findings - -## Phase Checkpoints - -Each phase requires text output before proceeding: - -| Phase | Goal | Required Output | -|-------|------|-----------------| -| **1. Explore** | Understand codebase | Launch exploration subagents -> Output findings summary (100+ words) | -| **2. Design** | Plan approach | (Optional: launch planning subagent) -> Output design decisions | -| **3. Review** | Verify details | Read critical files -> Output review summary with any questions | -| **4. Present Plan** | Show complete plan | Output your complete implementation plan to the user | -| **5. Exit** | Submit for approval | **MUST call ExitPlanMode tool** with your plan content | - -## Critical Rules - -- **Phase 1**: Use exploration subagents for initial research, not direct file searches -- **Loop prevention**: If calling 3+ tools without text output, STOP and summarize findings -- **Future tense**: Say "I will create X" not "I created X" (plan mode cannot modify files) -- **Research tasks**: Answer directly without ExitPlanMode (e.g., "Where is routing?") -- **Implementation tasks**: After presenting plan, MUST call ExitPlanMode to submit for approval - -## Plan Format - -Your plan should include: -1. **Summary** - What and why -2. **Current State** - Relevant existing code -3. **Steps** - Detailed implementation steps with file paths -4. **Testing** - How to verify changes -5. **Risks** - Potential issues and mitigations -`; - -/** - * 生成 Plan 模式的 system-reminder(每轮注入到用户消息中) - */ +/** 生成 Plan 模式的 system-reminder(每轮注入到用户消息中) */ export function createPlanModeReminder(userMessage: string): string { return ( `Plan mode is active. You MUST NOT make any file changes or run non-readonly tools. Research only, then call ExitPlanMode with your plan.\n\n` + diff --git a/packages/cli/src/prompts/index.ts b/packages/cli/src/prompts/index.ts index 12ed8dfd4..84e178109 100644 --- a/packages/cli/src/prompts/index.ts +++ b/packages/cli/src/prompts/index.ts @@ -1,7 +1,4 @@ -/** - * Prompts 模块入口 - * 导出系统提示相关的核心功能 - */ +/** Prompts 模块入口 导出系统提示相关的核心功能 */ export { buildSystemPrompt } from './builder.js'; diff --git a/packages/cli/src/prompts/plan-mode.md b/packages/cli/src/prompts/plan-mode.md new file mode 100644 index 000000000..f9ba490ec --- /dev/null +++ b/packages/cli/src/prompts/plan-mode.md @@ -0,0 +1,40 @@ +You are in **PLAN MODE** - a read-only research phase for designing implementation plans. + +## Core Objective + +Research the codebase thoroughly, then create a detailed implementation plan. No file modifications allowed until plan is approved. + +## Key Constraints + +1. **Read-only tools only**: File readers, search tools, web fetchers, and exploration subagents +2. **Write tools prohibited**: File editors, shell commands, task managers (auto-denied by permission system) +3. **Text output required**: You MUST output text summaries between tool calls - never call 3+ tools without explaining findings + +## Phase Checkpoints + +Each phase requires text output before proceeding: + +| Phase | Goal | Required Output | +|-------|------|-----------------| +| **1. Explore** | Understand codebase | Launch exploration subagents -> Output findings summary (100+ words) | +| **2. Design** | Plan approach | (Optional: launch planning subagent) -> Output design decisions | +| **3. Review** | Verify details | Read critical files -> Output review summary with any questions | +| **4. Present Plan** | Show complete plan | Output your complete implementation plan to the user | +| **5. Exit** | Submit for approval | **MUST call ExitPlanMode tool** with your plan content | + +## Critical Rules + +- **Phase 1**: Use exploration subagents for initial research, not direct file searches +- **Loop prevention**: If calling 3+ tools without text output, STOP and summarize findings +- **Future tense**: Say "I will create X" not "I created X" (plan mode cannot modify files) +- **Research tasks**: Answer directly without ExitPlanMode (e.g., "Where is routing?") +- **Implementation tasks**: After presenting plan, MUST call ExitPlanMode to submit for approval + +## Plan Format + +Your plan should include: +1. **Summary** - What and why +2. **Current State** - Relevant existing code +3. **Steps** - Detailed implementation steps with file paths +4. **Testing** - How to verify changes +5. **Risks** - Potential issues and mitigations diff --git a/packages/cli/src/prompts/processors/AtMentionParser.ts b/packages/cli/src/prompts/processors/AtMentionParser.ts index a24a0176b..b8b6728a7 100644 --- a/packages/cli/src/prompts/processors/AtMentionParser.ts +++ b/packages/cli/src/prompts/processors/AtMentionParser.ts @@ -20,14 +20,10 @@ export class AtMentionParser { */ private static readonly PATTERN = /@"([^"]+)"|@([^\s]+)/g; - /** - * 行号范围模式:#L10 或 #L10-20 - */ + /** 行号范围模式:#L10 或 #L10-20 */ private static readonly LINE_RANGE_PATTERN = /#L(\d+)(?:-(\d+))?$/; - /** - * Glob 通配符模式:检测 *, ?, [ 等字符 - */ + /** Glob 通配符模式:检测 *, ?, [ 等字符 */ private static readonly GLOB_PATTERN = /[*?[\]]/; /** diff --git a/packages/cli/src/prompts/processors/AttachmentCollector.ts b/packages/cli/src/prompts/processors/AttachmentCollector.ts index ae6ee6ff2..a7abe513a 100644 --- a/packages/cli/src/prompts/processors/AttachmentCollector.ts +++ b/packages/cli/src/prompts/processors/AttachmentCollector.ts @@ -22,9 +22,7 @@ function isFileTree(value: FileTreeEntry): value is FileTree { return value instanceof Map; } -/** - * 附件收集器 - */ +/** 附件收集器 */ export class AttachmentCollector { private fileCache = new Map(); private options: Required; @@ -137,9 +135,7 @@ export class AttachmentCollector { return await this.readFile(realPath, mention.path, mention.lineRange); } - /** - * 读取文件内容 - */ + /** 读取文件内容 */ private async readFile( absolutePath: string, relativePath: string, @@ -181,9 +177,7 @@ export class AttachmentCollector { return this.formatFileAttachment(relativePath, content, lineRange); } - /** - * 格式化文件附件 - */ + /** 格式化文件附件 */ private formatFileAttachment( relativePath: string, content: string, @@ -243,9 +237,7 @@ export class AttachmentCollector { }; } - /** - * 渲染目录树结构(不读取文件内容,仅展示结构) - */ + /** 渲染目录树结构(不读取文件内容,仅展示结构) */ private async renderDirectoryTree( absolutePath: string, relativePath: string @@ -300,9 +292,7 @@ export class AttachmentCollector { }; } - /** - * 构建文件树结构 - */ + /** 构建文件树结构 */ private buildFileTree(files: string[]): FileTree { const tree: FileTree = new Map(); @@ -330,9 +320,7 @@ export class AttachmentCollector { return tree; } - /** - * 打印树形结构为 ASCII 格式 - */ + /** 打印树形结构为 ASCII 格式 */ private printTree( tree: FileTree, rootPath: string, @@ -370,9 +358,7 @@ export class AttachmentCollector { return lines.filter((l) => l).join('\n'); } - /** - * 处理 Glob 模式 - */ + /** 处理 Glob 模式 */ private async processGlob(pattern: string): Promise { // 使用 fast-glob 展开模式 const files = (await fg(pattern, { @@ -470,41 +456,4 @@ export class AttachmentCollector { }, }; } - - /** - * 清理过期缓存 - */ - clearExpiredCache(): void { - const now = Date.now(); - let cleared = 0; - - for (const [key, value] of this.fileCache.entries()) { - if (now - value.timestamp > 60000) { - this.fileCache.delete(key); - cleared++; - } - } - - if (cleared > 0) { - logger.debug(`Cleared ${cleared} expired cache entries`); - } - } - - /** - * 清空所有缓存 - */ - clearCache(): void { - this.fileCache.clear(); - logger.debug('Cleared all cache'); - } - - /** - * 获取缓存统计 - */ - getCacheStats(): { size: number; keys: string[] } { - return { - size: this.fileCache.size, - keys: Array.from(this.fileCache.keys()), - }; - } } diff --git a/packages/cli/src/prompts/processors/types.ts b/packages/cli/src/prompts/processors/types.ts index 2244cae3d..3ab51f792 100644 --- a/packages/cli/src/prompts/processors/types.ts +++ b/packages/cli/src/prompts/processors/types.ts @@ -2,9 +2,7 @@ * @ 文件提及的类型定义 */ -/** - * 行号范围 - */ +/** 行号范围 */ export interface LineRange { start: number; end?: number; @@ -28,14 +26,10 @@ export interface AtMention { isGlob?: boolean; } -/** - * 附件类型 - */ +/** 附件类型 */ export type AttachmentType = 'file' | 'directory' | 'error'; -/** - * 附件元数据 - */ +/** 附件元数据 */ export interface AttachmentMetadata { /** 文件大小(字节) */ size?: number; @@ -47,9 +41,7 @@ export interface AttachmentMetadata { lineRange?: LineRange; } -/** - * 附件对象 - */ +/** 附件对象 */ export interface Attachment { /** 附件类型 */ type: AttachmentType; @@ -63,9 +55,7 @@ export interface Attachment { error?: string; } -/** - * 附件收集器选项 - */ +/** 附件收集器选项 */ export interface CollectorOptions { /** 工作目录 */ cwd: string; diff --git a/packages/cli/src/prompts/sections.ts b/packages/cli/src/prompts/sections.ts index 9f1e368e9..9cb75753c 100644 --- a/packages/cli/src/prompts/sections.ts +++ b/packages/cli/src/prompts/sections.ts @@ -1,9 +1,4 @@ -/** - * 模块化系统提示段 - * - * 每个函数返回一个独立的提示段,由 buildDefaultPrompt() 按序组装。 - * - */ +/** 模块化系统提示段 每个函数返回一个独立的提示段,由 buildDefaultPrompt() 按序组装。 */ // ============================================================ // 1. Intro — 身份 + 网络安全 diff --git a/packages/cli/src/server/routes/session.ts b/packages/cli/src/server/routes/session.ts index a1bd5f727..817e68aee 100644 --- a/packages/cli/src/server/routes/session.ts +++ b/packages/cli/src/server/routes/session.ts @@ -1,10 +1,7 @@ import { Mutex } from 'async-mutex'; import { Hono } from 'hono'; import { streamSSE } from 'hono/streaming'; -import { LRUCache } from 'lru-cache'; import { nanoid } from 'nanoid'; -import { Agent } from '../../agent/Agent.js'; -import { drainLoop } from '../../agent/loop/index.js'; import type { LoopEvent } from '../../agent/loop/types.js'; import { resolveWorkspaceAgentResources, @@ -36,7 +33,6 @@ import { } from '../../agent/runtime/SessionRuntimeResidency.js'; import { createLocalSessionWorkspace } from '../../agent/runtime/SessionWorkspace.js'; import { - type TaskAdmissionHandle, TaskAdmissionQueueFullError, taskRunScheduler, } from '../../agent/runtime/TaskRunScheduler.js'; @@ -46,14 +42,13 @@ import { toPublicAgentSession, } from '../../agent/subagents/AgentSessionStore.js'; import { isTeamMessageMetadata, TeamMailbox } from '../../agent/teams/TeamMailbox.js'; -import type { ChatContext, LoopResult, UserMessageContent } from '../../agent/types.js'; +import type { UserMessageContent } from '../../agent/types.js'; import { MAX_INLINE_ATTACHMENT_BYTES, MAX_USER_MESSAGE_TEXT_BYTES, } from '../../api/attachmentLimits.js'; import { CodeReviewRequestSchema, - FOLLOW_UP_QUEUE_MAX_ITEMS, type FollowUpQueueErrorCode, type FollowUpQueueMutationRequest, FollowUpQueueMutationSchema, @@ -67,16 +62,6 @@ import { SideConversationRequestSchema, UserShellCommandRequestSchema, } from '../../api/schemas.js'; -import { - MAX_BROWSER_DIAGNOSTIC_RESULT_ENTRIES, - MAX_BROWSER_ID_BYTES, - MAX_BROWSER_ORIGIN_BYTES, - MAX_BROWSER_PROJECTED_URL_BYTES, - MAX_BROWSER_REF_BYTES, - MAX_BROWSER_SCREENSHOT_BYTES, - MAX_BROWSER_TITLE_BYTES, -} from '../../browser/constants.js'; -import { isBrowserToolName } from '../../browser/types.js'; import { DEFAULT_MAX_RESIDENT_SESSION_PROJECTIONS, DEFAULT_SESSION_PROJECTION_IDLE_MS, @@ -153,11 +138,7 @@ import { projectDurableToolResult, SERVER_TOOL_DETAIL_MAX_CHARS, } from '../../tools/display/ToolResultProjector.js'; -import { - CONFIRMATION_ABORTED_REASON, - type ConfirmationDetails, - type ConfirmationResponse, -} from '../../tools/types/ExecutionTypes.js'; +import { type ConfirmationResponse } from '../../tools/types/ExecutionTypes.js'; import type { ToolResultMetadata } from '../../tools/types/ToolTypes.js'; import { formatToolDisplay, @@ -199,18 +180,31 @@ import { } from '../sessionRef.js'; import { WebBrowserSessionRegistry } from '../WebBrowserSessionRegistry.js'; import { BrowserRoutes } from './browser.js'; +import { projectSessionLoopEvent } from './sessionLoopEventProjection.js'; +import { executeRunAsync } from './sessionRunExecutor.js'; +import { + activeRunCount, + buildPendingInteractionEvent, + cancelRun, + findActivePermissionRun, + forgetRun, + getRun, + isActiveRun, + listActiveRuns, + type RunState, + refreshSessionTaskMetadata, + registerRun, + resetSessionRuns, + type SessionInfo, + sessionRefFromSession, + syncSessionTaskMetadata, + type WebPendingResumeAttempt, +} from './sessionRunState.js'; +import { sanitizeToolMetadata } from './sessionToolMetadata.js'; + +export { sanitizeToolMetadata } from './sessionToolMetadata.js'; const logger = createLogger(LogCategory.SERVICE); -const WEB_PENDING_RESUME_DEADLINE_ABORT = - 'web-pending-resume-recovery-budget-exhausted'; - -interface WebPendingResumeAttempt { - attempt: number; - deadlineAt: number; - generation: number; - projectedInputIds: Set; -} - interface WebPendingResumeState { attempt: number; generation: number; @@ -222,13 +216,6 @@ interface WebPendingResumeState { timer: ReturnType | undefined; } -class WebAgentRunFailure extends Error { - constructor(readonly evidence: PendingResumeFailureEvidence) { - super(evidence.taskFailure.message); - this.name = 'WebAgentRunFailure'; - } -} - const CreateSessionSchema = Type.Object({ title: Type.Optional(Type.String()), projectPath: Type.Optional(Type.String()), @@ -298,81 +285,6 @@ const UpdateGoalSchema = Type.Union([ }), ]); -export interface RunState { - id: string; - sessionId: string; - projectPath: string; - status: - | 'queued' - | 'running' - | 'waiting_permission' - | 'attention_required' - | 'completed' - | 'failed' - | 'cancelled'; - abortController: AbortController; - pendingPermission?: { - permissionId: string; - resolve: (response: ConfirmationResponse) => void; - details: ConfirmationDetails; - }; - pendingFollowUpRequested?: boolean; - taskAdmission?: TaskAdmissionHandle; - taskAdmissionUpdate?: Promise; - disposeRuntimeOnSettle?: boolean; - pendingResume?: WebPendingResumeAttempt; - projectionLease: SessionProjectionLease; - completion?: Promise; - createdAt: Date; -} - -interface SessionInfo { - id: string; - projectPath: string; - title: string; - createdAt: Date; - updatedAt: Date; - rootId: string; - parentId?: string; - messageCount: number; - currentRunId?: string; - relationType?: 'subagent' | 'fork'; - taskStatus: SessionMetadata['taskStatus']; - taskStatusReason?: string; - taskFailure?: SessionMetadata['taskFailure']; - taskStartedAt?: string; - taskCompletedAt?: string; - taskPromptSummary?: string; - taskPriority?: SessionTaskPriority; - taskKind?: SessionTaskKind; - taskDueAt?: string; - taskModelId?: string; - selectedModelId?: string; - permissionMode?: PermissionMode; - reasoningEffort?: ReasoningEffortSelection; - serviceTier?: ServiceTierSelection; - responseVerbosity?: ResponseVerbositySelection; - communicationStyle?: CommunicationStyleSelection; - communicationStyleDigest?: string; - projectInstructionsDigest?: string; - pendingInteraction?: SessionMetadata['pendingInteraction']; - taskRetryAvailable?: boolean; - taskRetriedFrom?: SessionTaskRetryRef; - taskDelivery?: SessionTaskDelivery; - taskIsolation?: SessionMetadata['taskIsolation']; - taskSourceProjectPath?: string; - taskWorktreePath?: string; - taskWorktreeBranch?: string; - taskBaseCommit?: string; - taskDiffStat?: SessionMetadata['taskDiffStat']; - taskQueuePosition?: number; - taskQueueDepth?: number; - taskConcurrencyLimit?: number; - archivedAt?: string; - archivedBySessionId?: string; - taskWorktree?: SessionTaskWorktree; -} - type SessionHydrationInvalidationReason = | 'archive' | 'delete' @@ -403,7 +315,6 @@ interface SessionHydrationOwner { let activeSessionHydrationOwner: SessionHydrationOwner | undefined; let resetPendingResumeRecoveries: (() => void) | undefined; -const activeRuns = new Map(); const activeUserShellRuns = new Map< string, { @@ -421,11 +332,6 @@ const activeReviewRuns = new Map< projectionLease?: SessionProjectionLease; } >(); -const recentRuns = new LRUCache({ - max: 100, - ttl: 30 * 60 * 1000, -}); - function cloneSessionInfo(session: SessionInfo): SessionInfo { const cloned = structuredClone({ ...session, @@ -439,103 +345,6 @@ function cloneSessionInfo(session: SessionInfo): SessionInfo { }; } -function runRef(run: RunState): SessionRef { - return { sessionId: run.sessionId, projectPath: run.projectPath }; -} - -function getRun(runId: string | undefined): RunState | undefined { - if (!runId) return undefined; - return activeRuns.get(runId) ?? recentRuns.get(runId); -} - -function settleRun(run: RunState): void { - if (activeRuns.get(run.id) !== run) return; - activeRuns.delete(run.id); - recentRuns.set(run.id, run); -} - -function forgetRun(runId: string): void { - activeRuns.delete(runId); - recentRuns.delete(runId); -} - -function isActiveRun(run: RunState | undefined): run is RunState { - return ( - run?.status === 'queued' || - run?.status === 'running' || - run?.status === 'waiting_permission' - ); -} - -function cancelRun(run: RunState, reason = 'user-cancel'): boolean { - if ( - run.status === 'cancelled' || - run.status === 'completed' || - run.status === 'failed' || - run.status === 'attention_required' - ) { - return false; - } - - const pendingPermission = run.pendingPermission; - run.pendingPermission = undefined; - pendingPermission?.resolve({ - approved: false, - reason: CONFIRMATION_ABORTED_REASON, - }); - if (pendingPermission) { - Bus.publish(runRef(run), 'interaction.resolved', { - requestId: pendingPermission.permissionId, - }); - } - run.abortController.abort(reason); - run.status = 'cancelled'; - Bus.publish(runRef(run), 'run.cancelled', { runId: run.id }); - return true; -} - -function buildPendingInteractionEvent( - pending: NonNullable, - replayed = false -): { type: string; properties: Record } { - const { permissionId, details } = pending; - if (details.type === 'askUserQuestion' && details.questions) { - return { - type: 'question.required', - properties: { - requestId: permissionId, - toolCallId: details.toolCallId ?? permissionId, - questions: details.questions, - details, - ...(replayed ? { replayed: true } : {}), - }, - }; - } - if (details.type === 'mcpElicitation' && details.mcpElicitation) { - return { - type: 'elicitation.required', - properties: { - requestId: permissionId, - toolCallId: details.toolCallId ?? permissionId, - elicitation: details.mcpElicitation, - ...(replayed ? { replayed: true } : {}), - }, - }; - } - - return { - type: 'permission.asked', - properties: { - requestId: permissionId, - toolName: details.toolName, - description: details.message, - args: details.args, - details, - ...(replayed ? { replayed: true } : {}), - }, - }; -} - function sessionBusEventSseMessage( event: import('../bus.js').BusEvent ): SerializedSseMessage | undefined { @@ -580,15 +389,11 @@ function resetSharedSessionRouteState(): void { previousOwner.invalidateAll('route-reset'); } activeSessionHydrationOwner = undefined; - for (const run of activeRuns.values()) { - cancelRun(run, 'route-reset'); - } - activeRuns.clear(); + resetSessionRuns('route-reset'); for (const run of activeUserShellRuns.values()) { run.controller.abort('route-reset'); } activeUserShellRuns.clear(); - recentRuns.clear(); resetPendingResumeRecoveries?.(); resetPendingResumeRecoveries = undefined; } @@ -598,399 +403,6 @@ type Variables = { requestSignal: AbortSignal; }; -function sanitizeToolAdmissionMetadata( - value: unknown -): Record | undefined { - if (!value || typeof value !== 'object' || Array.isArray(value)) return undefined; - const admission = value as Record; - const code = admission.code; - const reason = admission.reason; - const scope = admission.scope; - const kind = admission.kind; - const limit = admission.limit; - if ( - (code !== 'tool_busy' && code !== 'tool_batch_full') || - (reason !== 'queue_full' && reason !== 'wait_timeout' && reason !== 'turn_limit') || - (scope !== 'global' && scope !== 'session') || - typeof admission.retryable !== 'boolean' || - !Number.isSafeInteger(limit) || - (limit as number) <= 0 || - (kind !== undefined && - kind !== 'readonly' && - kind !== 'write' && - kind !== 'execute') - ) { - return undefined; - } - return { - code, - reason, - scope, - retryable: admission.retryable, - ...(kind === undefined ? {} : { kind }), - limit, - }; -} - -export const sanitizeToolMetadata = ( - toolName: string, - metadata: ToolResultMetadata | undefined -) => { - if (!metadata || typeof metadata !== 'object') return metadata; - const sanitized = { ...(metadata as Record) }; - const toolAdmission = sanitizeToolAdmissionMetadata(sanitized.tool_admission); - if (toolAdmission) sanitized.tool_admission = toolAdmission; - else delete sanitized.tool_admission; - if (isBrowserToolName(toolName)) { - const source = - sanitized.browser && - typeof sanitized.browser === 'object' && - !Array.isArray(sanitized.browser) - ? (sanitized.browser as Record) - : {}; - const projected: Record = {}; - const boundedString = (key: string, maximum: number, pattern?: RegExp): void => { - const value = source[key]; - if ( - typeof value === 'string' && - Buffer.byteLength(value) <= maximum && - (!pattern || pattern.test(value)) - ) { - projected[key] = value; - } - }; - boundedString('action', 64); - boundedString('status', 16, /^(?:ok|warning|error)$/); - boundedString('pageId', MAX_BROWSER_ID_BYTES, /^browser_page_[a-f0-9-]+$/); - boundedString('snapshotId', MAX_BROWSER_ID_BYTES, /^browser_snapshot_[a-f0-9-]+$/); - boundedString('origin', MAX_BROWSER_ORIGIN_BYTES); - boundedString('candidateOrigin', MAX_BROWSER_ORIGIN_BYTES); - boundedString('url', MAX_BROWSER_PROJECTED_URL_BYTES); - boundedString('title', MAX_BROWSER_TITLE_BYTES); - boundedString('errorCode', 64, /^browser_[a-z_]+$/); - if (typeof source.truncated === 'boolean') { - projected.truncated = source.truncated; - } - if ( - typeof source.actionApplied === 'boolean' || - source.actionApplied === 'unknown' - ) { - projected.actionApplied = source.actionApplied; - } - if (typeof source.sideEffectsUncertain === 'boolean') { - projected.sideEffectsUncertain = source.sideEffectsUncertain; - } - if ( - typeof source.diagnosticCount === 'number' && - Number.isSafeInteger(source.diagnosticCount) && - source.diagnosticCount >= 0 && - source.diagnosticCount <= MAX_BROWSER_DIAGNOSTIC_RESULT_ENTRIES - ) { - projected.diagnosticCount = source.diagnosticCount; - } - if ( - source.interaction && - typeof source.interaction === 'object' && - !Array.isArray(source.interaction) - ) { - const interaction = source.interaction as Record; - const allowedActions = new Set([ - 'click', - 'hover', - 'fill', - 'type', - 'press', - 'select', - 'check', - 'uncheck', - 'scroll', - ]); - if ( - typeof interaction.action === 'string' && - allowedActions.has(interaction.action) - ) { - const projectedInteraction: Record = { - action: interaction.action, - }; - if ( - typeof interaction.ref === 'string' && - Buffer.byteLength(interaction.ref) <= MAX_BROWSER_REF_BYTES && - /^[a-z][a-z0-9]*$/.test(interaction.ref) - ) { - projectedInteraction.ref = interaction.ref; - } - const boundedNumber = ( - value: unknown, - minimum: number, - maximum: number - ): value is number => - typeof value === 'number' && - Number.isFinite(value) && - value >= minimum && - value <= maximum; - if ( - interaction.viewport && - typeof interaction.viewport === 'object' && - !Array.isArray(interaction.viewport) - ) { - const viewport = interaction.viewport as Record; - if ( - boundedNumber(viewport.width, 1, 16_384) && - boundedNumber(viewport.height, 1, 16_384) - ) { - projectedInteraction.viewport = { - width: viewport.width, - height: viewport.height, - }; - } - } - if ( - interaction.targetBox && - typeof interaction.targetBox === 'object' && - !Array.isArray(interaction.targetBox) - ) { - const targetBox = interaction.targetBox as Record; - if ( - boundedNumber(targetBox.x, -16_384, 32_768) && - boundedNumber(targetBox.y, -16_384, 32_768) && - boundedNumber(targetBox.width, 0, 16_384) && - boundedNumber(targetBox.height, 0, 16_384) - ) { - projectedInteraction.targetBox = { - x: targetBox.x, - y: targetBox.y, - width: targetBox.width, - height: targetBox.height, - }; - } - } - projected.interaction = projectedInteraction; - } - } - if ( - source.artifact && - typeof source.artifact === 'object' && - !Array.isArray(source.artifact) - ) { - const artifact = source.artifact as Record; - if ( - typeof artifact.id === 'string' && - /^[a-f0-9]{64}$/.test(artifact.id) && - artifact.sha256 === artifact.id && - artifact.kind === 'image' && - artifact.mimeType === 'image/png' && - typeof artifact.size === 'number' && - Number.isSafeInteger(artifact.size) && - artifact.size >= 0 && - artifact.size <= MAX_BROWSER_SCREENSHOT_BYTES && - artifact.persisted === true - ) { - projected.artifact = { - id: artifact.id, - sha256: artifact.sha256, - kind: artifact.kind, - mimeType: artifact.mimeType, - size: artifact.size, - persisted: true, - ...(typeof artifact.path === 'string' && - Buffer.byteLength(artifact.path) <= 8_192 - ? { path: artifact.path } - : {}), - }; - } - } - return { - ...(typeof sanitized.summary === 'string' - ? { summary: sanitized.summary.slice(0, 512) } - : {}), - browser: projected, - ...(toolAdmission ? { tool_admission: toolAdmission } : {}), - } as ToolResultMetadata; - } - if (toolName === 'Bash') { - const projected: Record = {}; - const stringFields = ['message', 'signal', 'status', 'summary'] as const; - const booleanFields = [ - 'aborted', - 'acp_mode', - 'admission_failed', - 'auto_backgrounded', - 'background', - 'capture_truncated', - 'finalization_failed', - 'has_stderr', - 'output_accounting_complete', - 'output_truncated', - 'projection_truncated', - 'sandbox_required', - 'sandboxed', - 'stderr_projection_truncated', - 'stdout_projection_truncated', - 'terminal_output_merged', - 'timeout', - ] as const; - const numberFields = [ - 'execution_time', - 'foreground_budget_ms', - 'pid', - 'raw_output_bytes', - 'stderr_length', - 'stderr_omitted_bytes', - 'stderr_retained_bytes', - 'stderr_total_bytes', - 'stdout_length', - 'stdout_omitted_bytes', - 'stdout_retained_bytes', - 'stdout_total_bytes', - ] as const; - for (const field of stringFields) { - const value = sanitized[field]; - if (typeof value === 'string') projected[field] = value.slice(0, 8_192); - } - for (const field of booleanFields) { - const value = sanitized[field]; - if (typeof value === 'boolean') projected[field] = value; - } - for (const field of numberFields) { - const value = sanitized[field]; - if (typeof value === 'number' && Number.isSafeInteger(value) && value >= 0) { - projected[field] = value; - } - } - if ( - sanitized.exit_code === null || - (typeof sanitized.exit_code === 'number' && - Number.isSafeInteger(sanitized.exit_code)) - ) { - projected.exit_code = sanitized.exit_code; - } - if ( - sanitized.terminal_transport === 'local' || - sanitized.terminal_transport === 'acp' || - sanitized.terminal_transport === 'local_fallback' - ) { - projected.terminal_transport = sanitized.terminal_transport; - } - if ( - sanitized.background_reason === 'explicit' || - sanitized.background_reason === 'foreground_budget' - ) { - projected.background_reason = sanitized.background_reason; - } - for (const field of ['bash_id', 'shell_id'] as const) { - const value = sanitized[field]; - if ( - typeof value === 'string' && - value.length <= 128 && - /^bash_[A-Za-z0-9-]+$/.test(value) - ) { - projected[field] = value; - } - } - if (toolAdmission) projected.tool_admission = toolAdmission; - const backgroundAdmission = sanitized.background_shell_admission; - if ( - backgroundAdmission && - typeof backgroundAdmission === 'object' && - !Array.isArray(backgroundAdmission) - ) { - const value = backgroundAdmission as Record; - if ( - value.code === 'background_shell_busy' && - (value.scope === 'session' || value.scope === 'global') && - value.retryable === true && - Number.isSafeInteger(value.limit) && - (value.limit as number) > 0 - ) { - projected.background_shell_admission = { - code: value.code, - scope: value.scope, - retryable: value.retryable, - limit: value.limit, - }; - } - } - return projected as ToolResultMetadata; - } - const MAX_INLINE_CONTENT = 200000; - const safeInteger = (value: unknown, maximum: number): number => - typeof value === 'number' && - Number.isSafeInteger(value) && - value >= 0 && - value <= maximum - ? value - : 0; - if ( - typeof sanitized.oldContent === 'string' && - sanitized.oldContent.length > MAX_INLINE_CONTENT - ) { - delete sanitized.oldContent; - } - if ( - typeof sanitized.newContent === 'string' && - sanitized.newContent.length > MAX_INLINE_CONTENT - ) { - delete sanitized.newContent; - } - if ( - sanitized.mcpResult && - typeof sanitized.mcpResult === 'object' && - !Array.isArray(sanitized.mcpResult) - ) { - const result = sanitized.mcpResult as Record; - const artifacts = Array.isArray(result.artifacts) - ? result.artifacts.slice(0, 64).flatMap((value) => { - if (!value || typeof value !== 'object' || Array.isArray(value)) return []; - const artifact = value as Record; - const artifactKinds = new Set(['text', 'image', 'audio', 'resource']); - if ( - typeof artifact.id !== 'string' || - !/^[a-f0-9]{64}$/.test(artifact.id) || - typeof artifact.sha256 !== 'string' || - artifact.sha256 !== artifact.id || - typeof artifact.kind !== 'string' || - !artifactKinds.has(artifact.kind) || - safeInteger(artifact.size, 64 * 1024 * 1024) !== artifact.size || - typeof artifact.persisted !== 'boolean' - ) { - return []; - } - return [ - { - id: artifact.id.slice(0, 128), - sha256: artifact.sha256.slice(0, 128), - kind: artifact.kind, - size: artifact.size, - persisted: artifact.persisted, - ...(typeof artifact.mimeType === 'string' - ? { mimeType: artifact.mimeType.slice(0, 256) } - : {}), - ...(typeof artifact.sourceUri === 'string' - ? { sourceUri: artifact.sourceUri.slice(0, 8_192) } - : {}), - ...(typeof artifact.path === 'string' - ? { path: artifact.path.slice(0, 8_192) } - : {}), - }, - ]; - }) - : []; - sanitized.mcpResult = { - isError: result.isError === true, - contentCount: safeInteger(result.contentCount, 64), - textBytes: safeInteger(result.textBytes, 4 * 1024 * 1024), - structuredBytes: safeInteger(result.structuredBytes, 4 * 1024 * 1024), - artifactCount: safeInteger(result.artifactCount, 64), - truncated: result.truncated === true, - binaryOmitted: result.binaryOmitted === true, - artifacts, - }; - } else { - delete sanitized.mcpResult; - } - return sanitized as ToolResultMetadata; -}; - export function projectCommittedSessionEvent(event: SessionEvent): | { type: string; @@ -1073,53 +485,21 @@ function publishSubagentLoopEvent( subagentSessionId: string, event: LoopEvent ): void { + const projection = projectSessionLoopEvent(event); + if (projection) { + if (projection.type === 'tool.start') { + Bus.publish(ref, 'subagent.update', { + subagentSessionId, + toolName: projection.toolName, + }); + } + Bus.publish(ref, `subagent.${projection.type}`, { + subagentSessionId, + ...projection.properties, + }); + return; + } switch (event.kind) { - case 'tool_start': - if ('function' in event.toolCall) { - Bus.publish(ref, 'subagent.update', { - subagentSessionId, - toolName: event.toolCall.function.name, - }); - Bus.publish(ref, 'subagent.tool.start', { - subagentSessionId, - toolCallId: event.toolCall.id, - toolName: event.toolCall.function.name, - arguments: event.toolCall.function.arguments, - toolKind: event.toolKind, - }); - } - break; - case 'tool_result': - if ('function' in event.toolCall) { - Bus.publish(ref, 'subagent.tool.result', { - subagentSessionId, - toolCallId: event.toolCall.id, - toolName: event.toolCall.function.name, - success: event.result.success, - summary: event.result.metadata?.summary, - output: renderToolDisplayToString( - fitToolDisplayForSurface( - formatToolDisplay(event.toolCall.function.name, event.result), - SERVER_TOOL_DETAIL_MAX_CHARS - ) - ), - metadata: sanitizeToolMetadata( - event.toolCall.function.name, - event.result.metadata - ), - }); - } - break; - case 'tool_progress': - if ('function' in event.toolCall) { - Bus.publish(ref, 'subagent.tool.progress', { - subagentSessionId, - toolCallId: event.toolCall.id, - toolName: event.toolCall.function.name, - ...event.update, - }); - } - break; case 'content_delta': Bus.publish(ref, 'subagent.delta', { subagentSessionId, @@ -1135,173 +515,6 @@ function publishSubagentLoopEvent( case 'stream_end': Bus.publish(ref, 'subagent.stream.end', { subagentSessionId }); break; - case 'provider_admission': - Bus.publish(ref, 'subagent.provider.admission', { - subagentSessionId, - phase: event.phase, - requestClass: event.requestClass, - resource: event.resource, - scope: event.scope, - reason: event.reason, - queuePosition: event.queuePosition, - queueDepth: event.queueDepth, - inFlight: event.inFlight, - limit: event.limit, - waitMs: event.waitMs, - maxWaitMs: event.maxWaitMs, - recoveryRemainingMs: event.recoveryRemainingMs, - }); - break; - case 'provider_circuit': - Bus.publish(ref, 'subagent.provider.circuit', { - subagentSessionId, - phase: event.phase, - reason: event.reason, - statusCode: event.statusCode, - retryAfterMs: event.retryAfterMs, - nextProbeAt: event.nextProbeAt, - openDurationMs: event.openDurationMs, - sampleCount: event.sampleCount, - failureCount: event.failureCount, - recoveryRemainingMs: event.recoveryRemainingMs, - }); - break; - case 'provider_retry': - Bus.publish(ref, 'subagent.provider.retry', { - subagentSessionId, - phase: event.phase, - attempt: event.attempt, - maxRetries: event.maxRetries, - reason: event.reason, - statusCode: event.statusCode, - delayMs: event.delayMs, - nextRetryAt: event.nextRetryAt, - mode: event.mode, - recoveryBudgetMs: event.recoveryBudgetMs, - recoveryElapsedMs: event.recoveryElapsedMs, - recoveryRemainingMs: event.recoveryRemainingMs, - exhaustedBy: event.exhaustedBy, - }); - break; - case 'provider_stall': - Bus.publish(ref, 'subagent.provider.stall', { - subagentSessionId, - phase: event.phase, - stallCount: event.stallCount, - durationMs: event.durationMs, - warningAfterMs: event.warningAfterMs, - timeoutMs: event.timeoutMs, - outputStarted: event.outputStarted, - }); - break; - case 'action_stationarity': - Bus.publish(ref, 'subagent.action.stationarity', { - subagentSessionId, - phase: event.phase, - toolName: event.toolName, - runLength: event.runLength, - nudgeThreshold: event.nudgeThreshold, - haltThreshold: event.haltThreshold, - progressAware: event.progressAware, - }); - break; - case 'mcp_catalog_changed': - Bus.publish(ref, 'subagent.mcp.catalog.changed', { - subagentSessionId, - revision: event.revision, - serverName: event.serverName, - added: event.added, - removed: event.removed, - updated: event.updated, - }); - break; - case 'mcp_content_changed': - Bus.publish(ref, 'subagent.mcp.content.changed', { - subagentSessionId, - revision: event.revision, - serverName: event.serverName, - contentKind: event.contentKind, - added: event.added, - removed: event.removed, - updated: event.updated, - }); - break; - case 'mcp_resource_updated': - Bus.publish(ref, 'subagent.mcp.resource.updated', { - subagentSessionId, - revision: event.revision, - serverName: event.serverName, - uri: event.uri, - }); - break; - case 'mcp_connection_changed': - Bus.publish(ref, 'subagent.mcp.connection.changed', { - subagentSessionId, - revision: event.revision, - serverName: event.serverName, - phase: event.phase, - reason: event.reason, - attempt: event.attempt, - maxAttempts: event.maxAttempts, - nextRetryAt: event.nextRetryAt, - error: event.error, - }); - break; - case 'mcp_log': - Bus.publish(ref, 'subagent.mcp.log', { - subagentSessionId, - revision: event.revision, - serverName: event.serverName, - level: event.level, - logger: event.logger, - message: event.message, - projectedBytes: event.projectedBytes, - dataSha256: event.dataSha256, - truncated: event.truncated, - detailsOmitted: event.detailsOmitted, - timestamp: event.timestamp, - synthetic: event.synthetic, - }); - break; - case 'mcp_instructions_changed': - Bus.publish(ref, 'subagent.mcp.instructions.changed', { - subagentSessionId, - revision: event.revision, - serverName: event.serverName, - action: event.action, - reason: event.reason, - text: event.text, - sourceBytes: event.sourceBytes, - projectedBytes: event.projectedBytes, - sha256: event.sha256, - truncated: event.truncated, - detailsOmitted: event.detailsOmitted, - }); - break; - case 'mcp_task_changed': - Bus.publish(ref, 'subagent.mcp.task.changed', { - subagentSessionId, - revision: event.revision, - taskId: event.taskId, - serverName: event.serverName, - toolName: event.toolName, - status: event.status, - statusMessage: event.statusMessage, - createdAt: event.createdAt, - updatedAt: event.updatedAt, - completedAt: event.completedAt, - hasResult: event.hasResult, - error: event.error, - }); - break; - case 'project_rules_loaded': - Bus.publish(ref, 'subagent.project.rules.loaded', { - subagentSessionId, - files: event.files, - triggerPaths: event.triggerPaths, - blockedWrite: event.blockedWrite, - }); - break; default: break; } @@ -1330,13 +543,6 @@ function validateSessionIdOrThrow(sessionId: string): void { } } -function sessionRefFromSession(session: SessionInfo): SessionRef { - return { - sessionId: session.id, - projectPath: session.projectPath, - }; -} - function sessionInfoFromMetadata( metadata: SessionMetadata, taskWorktree?: SessionTaskWorktree @@ -1481,56 +687,6 @@ export function projectClientMessages(messages: readonly Message[]): Message[] { }); } -function syncSessionTaskMetadata( - session: SessionInfo, - metadata: SessionMetadata -): void { - session.title = metadata.title ?? session.title; - session.taskStatus = metadata.taskStatus; - session.taskStatusReason = metadata.taskStatusReason; - session.taskFailure = metadata.taskFailure; - session.taskStartedAt = metadata.taskStartedAt; - session.taskCompletedAt = metadata.taskCompletedAt; - session.taskPromptSummary = metadata.taskPromptSummary; - session.taskPriority = metadata.taskPriority; - session.taskKind = metadata.taskKind; - session.taskDueAt = metadata.taskDueAt; - session.taskModelId = metadata.taskModelId; - session.selectedModelId = metadata.selectedModelId; - session.permissionMode = metadata.permissionMode as PermissionMode | undefined; - session.reasoningEffort = metadata.reasoningEffort; - session.serviceTier = metadata.serviceTier; - session.responseVerbosity = metadata.responseVerbosity; - session.communicationStyle = metadata.communicationStyle; - session.communicationStyleDigest = metadata.communicationStyleDigest; - session.projectInstructionsDigest = metadata.projectInstructionsDigest; - session.pendingInteraction = metadata.pendingInteraction; - session.taskRetryAvailable = metadata.taskRetryAvailable; - session.taskRetriedFrom = metadata.taskRetriedFrom; - session.taskDelivery = metadata.taskDelivery; - session.taskIsolation = metadata.taskIsolation; - session.taskSourceProjectPath = metadata.taskSourceProjectPath; - session.taskWorktreePath = metadata.taskWorktreePath; - session.taskWorktreeBranch = metadata.taskWorktreeBranch; - session.taskBaseCommit = metadata.taskBaseCommit; - session.taskDiffStat = metadata.taskDiffStat; - session.taskQueuePosition = metadata.taskQueuePosition; - session.taskQueueDepth = metadata.taskQueueDepth; - session.taskConcurrencyLimit = metadata.taskConcurrencyLimit; - session.archivedAt = metadata.archivedAt; - session.archivedBySessionId = metadata.archivedBySessionId; - session.messageCount = metadata.messageCount; - session.updatedAt = new Date(metadata.lastMessageTime); -} - -async function refreshSessionTaskMetadata(session: SessionInfo): Promise { - const metadata = await SessionService.findSessionMetadata( - session.id, - session.projectPath - ); - if (metadata) syncSessionTaskMetadata(session, metadata); -} - function projectActiveSession(session: SessionInfo) { const run = getRun(session.currentRunId); const taskStatus = @@ -1662,14 +818,6 @@ export async function resolveSessionRef( return matches.values().next().value as SessionRef; } -function getDisplayContent(content: UserMessageContent): string { - if (typeof content === 'string') return content; - return content - .filter((part) => part.type === 'text') - .map((part) => part.text) - .join('\n'); -} - function buildUserMessageContent( content: string, attachments?: Array<{ type: 'file' | 'image' | 'url'; content?: string }> @@ -2148,7 +1296,7 @@ export const createSessionRouteController = (): SessionRouteController => { ): Promise => taskDeliveryLocks.runExclusive(sessionRefKey(ref), operation); const hasActiveRunForRef = (ref: SessionRef): boolean => - [...activeRuns.values()].some( + listActiveRuns().some( (run) => run.sessionId === ref.sessionId && run.projectPath === ref.projectPath && @@ -2866,7 +2014,7 @@ export const createSessionRouteController = (): SessionRouteController => { session.taskQueueDepth = snapshot.queueDepth; session.taskConcurrencyLimit = snapshot.maxConcurrent; } - activeRuns.set(runId, run); + registerRun(run); session.currentRunId = runId; run.completion = executeRunAsync( run, @@ -5829,7 +4977,7 @@ export const createSessionRouteController = (): SessionRouteController => { }; const observedCompletions = new Set>(); const signalActiveWork = (): void => { - for (const run of activeRuns.values()) { + for (const run of listActiveRuns()) { if (run.completion) observedCompletions.add(run.completion); if (isActiveRun(run)) cancelRun(run, reason); } @@ -5910,904 +5058,6 @@ export const createSessionRouteController = (): SessionRouteController => { export const SessionRoutes = () => createSessionRouteController().app; -async function executeRunAsync( - run: RunState, - session: SessionInfo, - content: UserMessageContent, - permissionMode: PermissionMode, - acquireRuntime: ( - session: SessionInfo - ) => Promise>, - options: { - pendingInputOnly?: boolean; - preparedInputTurn?: PreparedInputTurn; - goalContinuationOnly?: boolean; - outputSchema?: SessionTaskDispatch['outputSchema']; - taskAdmission?: TaskAdmissionHandle; - runtimeLease?: SessionRuntimeResidencyLease; - projectionLease?: SessionProjectionLease; - disposeRuntime?: (session: SessionInfo, runtime?: SessionRuntime) => Promise; - pendingResume?: WebPendingResumeAttempt; - onPendingResumeFailure?: ( - attempt: WebPendingResumeAttempt, - evidence: PendingResumeFailureEvidence, - workStillPending: boolean, - deadlineExceeded: boolean - ) => boolean; - onPendingResumeSuccess?: (attempt: WebPendingResumeAttempt) => boolean; - onPendingResumeCancelled?: (attempt: WebPendingResumeAttempt) => void; - } = {} -): Promise { - const { abortController, sessionId, id: runId } = run; - const userMessageId = options.preparedInputTurn?.messageId ?? nanoid(12); - const startsFromPending = - options.pendingInputOnly === true || - options.preparedInputTurn?.mode === 'pending' || - options.goalContinuationOnly === true; - let assistantMessageId: string | undefined = startsFromPending - ? undefined - : nanoid(12); - let runtimeLease = options.runtimeLease; - const projectionLease = options.projectionLease; - let runtime: SessionRuntime | undefined; - let agent: Agent | undefined; - let outputStarted = false; - let toolExecutionStarted = false; - let pendingResumeDeadlineTimer: ReturnType | undefined; - const sessionRef = sessionRefFromSession(session); - const projectedInboxMessageIds = - options.pendingResume?.projectedInputIds ?? new Set(); - const rememberProjectedInboxMessageId = (messageId: string): void => { - if (projectedInboxMessageIds.has(messageId)) return; - if (projectedInboxMessageIds.size >= FOLLOW_UP_QUEUE_MAX_ITEMS) { - const oldest = projectedInboxMessageIds.values().next().value; - if (oldest !== undefined) projectedInboxMessageIds.delete(oldest); - } - projectedInboxMessageIds.add(messageId); - }; - - const settleRecoveryAttention = async (result: LoopResult): Promise => { - const assessment = result.metadata?.recoveryAttention; - if (!assessment || !runtime) return false; - if (options.preparedInputTurn) { - await runtime - .finishTurn(options.preparedInputTurn.handle, { - preserveStartupRecovery: true, - outcome: { - status: 'aborted', - cause: 'failed', - turnsCount: 0, - toolCallsCount: 0, - durationMs: 0, - }, - }) - .catch(() => undefined); - } - const reason = `Turn recovery requires attention: ${assessment.reason}`; - const metadata = await runtime - .setTaskStatus('interrupted', reason) - .catch(() => undefined); - if (metadata) syncSessionTaskMetadata(session, metadata); - else { - session.taskStatus = 'interrupted'; - session.taskStatusReason = reason; - session.taskCompletedAt = undefined; - } - run.status = 'attention_required'; - emit('session.status', { status: 'idle' }); - return true; - }; - - const emit = (type: string, properties: Record) => { - Bus.publish(sessionRef, type, properties); - }; - - const finalizeCancellation = async (): Promise => { - const reason = String(abortController.signal.reason || 'Task run cancelled'); - if (session.taskIsolation) { - if (!runtimeLease) { - runtimeLease = await acquireRuntime(session).catch(() => undefined); - } - const taskRuntime = runtime ?? runtimeLease?.value; - if (reason === 'user-cancel') { - await taskRuntime?.discardPendingInput().catch((error) => { - logger.warn( - `[SessionRoutes] Failed to discard cancelled input for ${session.id}:`, - error - ); - }); - } - const metadata = await taskRuntime - ?.setTaskStatus('cancelled', reason) - .catch(() => undefined); - if (metadata) { - syncSessionTaskMetadata(session, metadata); - } else { - await refreshSessionTaskMetadata(session).catch(() => undefined); - } - } - session.taskStatus = 'cancelled'; - session.taskStatusReason = reason; - session.taskCompletedAt ??= new Date().toISOString(); - }; - - try { - if (options.pendingResume) { - const remainingMs = options.pendingResume.deadlineAt - Date.now(); - if (remainingMs <= 0) { - abortController.abort(WEB_PENDING_RESUME_DEADLINE_ABORT); - throw new WebAgentRunFailure({ - taskFailure: taskFailureForCode('timeout'), - outputStarted: false, - toolExecutionStarted: false, - toolCallsCount: 0, - }); - } else { - pendingResumeDeadlineTimer = setTimeout(() => { - abortController.abort(WEB_PENDING_RESUME_DEADLINE_ABORT); - const pendingPermission = run.pendingPermission; - run.pendingPermission = undefined; - pendingPermission?.resolve({ - approved: false, - reason: CONFIRMATION_ABORTED_REASON, - }); - if (pendingPermission) { - emit('interaction.resolved', { - requestId: pendingPermission.permissionId, - }); - } - }, remainingMs); - pendingResumeDeadlineTimer.unref?.(); - } - } - if (options.taskAdmission) { - await options.taskAdmission.ready; - await run.taskAdmissionUpdate; - if (abortController.signal.aborted) { - throw new Error(String(abortController.signal.reason || 'Task run cancelled')); - } - } - - session.taskStatus = 'running'; - session.taskStatusReason = undefined; - session.taskStartedAt = new Date().toISOString(); - session.taskCompletedAt = undefined; - if (!options.pendingInputOnly && !options.goalContinuationOnly) { - emit('message.created', { - messageId: userMessageId, - role: 'user', - content: getDisplayContent(content), - ...(options.preparedInputTurn?.metadata - ? { metadata: options.preparedInputTurn.metadata } - : {}), - }); - } - emit('session.status', { status: 'running' }); - if (assistantMessageId) { - emit('message.created', { - messageId: assistantMessageId, - role: 'assistant', - content: '', - }); - } - - runtimeLease ??= await acquireRuntime(session); - runtime = runtimeLease.value; - const runtimeOwner = runtime; - const structuredOutputExpected = Boolean( - options.outputSchema ?? - runtimeOwner - .getPendingSteeringMessages() - .find((pending) => pending.outputSchema)?.outputSchema - ); - agent = await Agent.createWithRuntime(runtimeOwner, { - sessionId, - ...(session.taskWorktree - ? { toolBlacklist: ['EnterWorktree', 'ExitWorktree'] } - : {}), - }); - - const requestConfirmation = async ( - details: ConfirmationDetails - ): Promise => { - const permissionId = details.interactionRequestId ?? nanoid(12); - const PERMISSION_TIMEOUT = 5 * 60 * 1000; - - run.status = 'waiting_permission'; - - const resultPromise = new Promise((resolve) => { - const timeout = setTimeout(() => { - logger.warn( - `[SessionRoutes] Permission ${permissionId} timed out after ${PERMISSION_TIMEOUT}ms` - ); - emit('permission.timeout', { requestId: permissionId }); - resolve({ approved: false, reason: 'timeout' }); - }, PERMISSION_TIMEOUT); - - run.pendingPermission = { - permissionId, - resolve: (response) => { - clearTimeout(timeout); - resolve(response); - }, - details, - }; - }); - - const pendingInteraction = run.pendingPermission; - if (!pendingInteraction) { - throw new Error('Permission request was not registered'); - } - const interaction = buildPendingInteractionEvent(pendingInteraction); - emit(interaction.type, interaction.properties); - - logger.info( - `[SessionRoutes] Permission request created: ${permissionId}, runId: ${runId}` - ); - - const response = await resultPromise; - logger.info( - `[SessionRoutes] Permission response received: ${permissionId}, approved: ${response.approved}` - ); - if (!abortController.signal.aborted) { - run.status = 'running'; - } - if (run.pendingPermission === pendingInteraction) { - run.pendingPermission = undefined; - } - emit('interaction.resolved', { requestId: permissionId }); - - return response; - }; - - const modelContext = await SessionService.loadSessionModelContext( - session.id, - session.projectPath - ); - const chatContext: ChatContext = { - messages: modelContext, - userId: 'web-user', - sessionId, - workspaceRoot: session.projectPath, - signal: abortController.signal, - permissionMode, - onPermissionModeChange: async (nextMode) => { - session.permissionMode = nextMode; - session.updatedAt = new Date(); - emit('session.updated', { permissionMode: nextMode }); - }, - ...(session.taskWorktree ? { worktreeActive: true } : {}), - confirmationHandler: { requestConfirmation }, - }; - const ensureAssistantMessage = (): string => { - if (!assistantMessageId) { - assistantMessageId = nanoid(12); - emit('message.created', { - messageId: assistantMessageId, - role: 'assistant', - content: '', - }); - } - return assistantMessageId; - }; - - // Phase 4: 使用 chatStream() + onEvent 事件驱动消费 - // message.complete 只在整个 run 结束时发一次(run-level 语义) - // stream_end 不外发给客户端(内部 per-turn 信号) - const handleLoopEvent = async (event: LoopEvent) => { - switch (event.kind) { - // --- 流式增量 --- - case 'content_delta': - if (event.delta.length > 0) outputStarted = true; - if (structuredOutputExpected) break; - emit('message.delta', { - messageId: ensureAssistantMessage(), - delta: event.delta, - }); - break; - case 'structured_output': - outputStarted = true; - emit('structured.output', { - messageId: ensureAssistantMessage(), - output: event.output, - schemaDigest: event.schemaDigest, - }); - break; - case 'thinking_delta': - if (event.delta.length > 0) outputStarted = true; - emit('thinking.delta', { - messageId: ensureAssistantMessage(), - delta: event.delta, - }); - break; - - // --- 工具事件 --- - case 'tool_start': - toolExecutionStarted = true; - if ('function' in event.toolCall) { - if (event.toolCall.function.name === STRUCTURED_OUTPUT_TOOL_NAME) break; - emit('tool.start', { - messageId: ensureAssistantMessage(), - toolName: event.toolCall.function.name, - toolCallId: event.toolCall.id, - arguments: event.toolCall.function.arguments, - toolKind: event.toolKind, - }); - } - break; - case 'tool_result': - toolExecutionStarted = true; - if ('function' in event.toolCall) { - if (event.toolCall.function.name === STRUCTURED_OUTPUT_TOOL_NAME) break; - emit('tool.result', { - messageId: ensureAssistantMessage(), - toolName: event.toolCall.function.name, - toolCallId: event.toolCall.id, - success: event.result.success, - summary: event.result.metadata?.summary, - output: renderToolDisplayToString( - fitToolDisplayForSurface( - formatToolDisplay(event.toolCall.function.name, event.result), - SERVER_TOOL_DETAIL_MAX_CHARS - ) - ), - metadata: sanitizeToolMetadata( - event.toolCall.function.name, - event.result.metadata - ), - }); - } - break; - case 'tool_progress': - toolExecutionStarted = true; - if ('function' in event.toolCall) { - if (event.toolCall.function.name === STRUCTURED_OUTPUT_TOOL_NAME) break; - emit('tool.progress', { - messageId: ensureAssistantMessage(), - toolName: event.toolCall.function.name, - toolCallId: event.toolCall.id, - ...event.update, - }); - } - break; - - // --- Token 使用 --- - case 'token_usage': - emit('token.usage', { ...event.usage }); - break; - case 'turn_start': - emit('turn.started', { turn: event.turn, maxTurns: event.maxTurns }); - break; - case 'turn_recovery': - emit('turn.recovery', { assessment: event.assessment }); - break; - case 'provider_admission': - emit('provider.admission', { - phase: event.phase, - requestClass: event.requestClass, - resource: event.resource, - scope: event.scope, - reason: event.reason, - queuePosition: event.queuePosition, - queueDepth: event.queueDepth, - inFlight: event.inFlight, - limit: event.limit, - waitMs: event.waitMs, - maxWaitMs: event.maxWaitMs, - recoveryRemainingMs: event.recoveryRemainingMs, - }); - break; - case 'provider_circuit': - emit('provider.circuit', { - phase: event.phase, - reason: event.reason, - statusCode: event.statusCode, - retryAfterMs: event.retryAfterMs, - nextProbeAt: event.nextProbeAt, - openDurationMs: event.openDurationMs, - sampleCount: event.sampleCount, - failureCount: event.failureCount, - recoveryRemainingMs: event.recoveryRemainingMs, - }); - break; - case 'provider_retry': - emit('provider.retry', { - phase: event.phase, - attempt: event.attempt, - maxRetries: event.maxRetries, - reason: event.reason, - statusCode: event.statusCode, - delayMs: event.delayMs, - nextRetryAt: event.nextRetryAt, - mode: event.mode, - recoveryBudgetMs: event.recoveryBudgetMs, - recoveryElapsedMs: event.recoveryElapsedMs, - recoveryRemainingMs: event.recoveryRemainingMs, - exhaustedBy: event.exhaustedBy, - }); - break; - case 'provider_stall': - emit('provider.stall', { - phase: event.phase, - stallCount: event.stallCount, - durationMs: event.durationMs, - warningAfterMs: event.warningAfterMs, - timeoutMs: event.timeoutMs, - outputStarted: event.outputStarted, - }); - break; - case 'provider_recovery': - // SessionRuntime already publishes the authoritative projection on the - // Session Bus. Do not emit a duplicate from this direct consumer. - break; - case 'turn_activity': - // SessionRuntime already publishes the authoritative projection on the - // Session Bus. Do not emit a duplicate from this direct consumer. - break; - case 'action_stationarity': - emit('action.stationarity', { - phase: event.phase, - toolName: event.toolName, - runLength: event.runLength, - nudgeThreshold: event.nudgeThreshold, - haltThreshold: event.haltThreshold, - progressAware: event.progressAware, - }); - break; - case 'mcp_catalog_changed': - emit('mcp.catalog.changed', { - messageId: ensureAssistantMessage(), - revision: event.revision, - serverName: event.serverName, - added: event.added, - removed: event.removed, - updated: event.updated, - }); - break; - case 'mcp_content_changed': - emit('mcp.content.changed', { - messageId: ensureAssistantMessage(), - revision: event.revision, - serverName: event.serverName, - contentKind: event.contentKind, - added: event.added, - removed: event.removed, - updated: event.updated, - }); - break; - case 'mcp_resource_updated': - emit('mcp.resource.updated', { - messageId: ensureAssistantMessage(), - revision: event.revision, - serverName: event.serverName, - uri: event.uri, - }); - break; - case 'mcp_connection_changed': - emit('mcp.connection.changed', { - messageId: ensureAssistantMessage(), - revision: event.revision, - serverName: event.serverName, - phase: event.phase, - reason: event.reason, - attempt: event.attempt, - maxAttempts: event.maxAttempts, - nextRetryAt: event.nextRetryAt, - error: event.error, - }); - break; - case 'mcp_log': - emit('mcp.log', { - messageId: ensureAssistantMessage(), - revision: event.revision, - serverName: event.serverName, - level: event.level, - logger: event.logger, - message: event.message, - projectedBytes: event.projectedBytes, - dataSha256: event.dataSha256, - truncated: event.truncated, - detailsOmitted: event.detailsOmitted, - timestamp: event.timestamp, - synthetic: event.synthetic, - }); - break; - case 'mcp_instructions_changed': - emit('mcp.instructions.changed', { - messageId: ensureAssistantMessage(), - revision: event.revision, - serverName: event.serverName, - action: event.action, - reason: event.reason, - text: event.text, - sourceBytes: event.sourceBytes, - projectedBytes: event.projectedBytes, - sha256: event.sha256, - truncated: event.truncated, - detailsOmitted: event.detailsOmitted, - }); - break; - case 'mcp_task_changed': - emit('mcp.task.changed', { - messageId: ensureAssistantMessage(), - revision: event.revision, - taskId: event.taskId, - serverName: event.serverName, - toolName: event.toolName, - status: event.status, - statusMessage: event.statusMessage, - createdAt: event.createdAt, - updatedAt: event.updatedAt, - completedAt: event.completedAt, - hasResult: event.hasResult, - error: event.error, - }); - break; - case 'project_rules_loaded': - emit('project.rules.loaded', { - messageId: ensureAssistantMessage(), - files: event.files, - triggerPaths: event.triggerPaths, - blockedWrite: event.blockedWrite, - }); - break; - case 'steering_applied': - for (const message of event.messages) { - if ((message.origin ?? 'user') !== 'user') continue; - if (message.persisted) continue; - if (projectedInboxMessageIds.has(message.id)) continue; - emit('message.created', { - messageId: message.id, - role: 'user', - content: getDisplayContent(message.content), - ...(message.metadata ? { metadata: message.metadata } : {}), - ...(message.recovered ? { recovered: true } : {}), - }); - rememberProjectedInboxMessageId(message.id); - } - emit('steering.applied', { - runId, - messageIds: event.messageIds, - count: event.count, - recovered: event.recovered, - delivery: event.delivery, - queued: runtimeOwner.getPendingSteeringCount(), - }); - emit('follow_up.queue.changed', { queue: event.queue }); - break; - case 'follow_up_started': { - if (assistantMessageId) { - emit('message.complete', { messageId: assistantMessageId }); - assistantMessageId = undefined; - } - emit('follow_up.started', { - runId, - queued: event.queued, - recovered: event.recovered, - }); - emit('follow_up.queue.changed', { queue: event.queue }); - ensureAssistantMessage(); - break; - } - case 'follow_up_queue_changed': - emit('follow_up.queue.changed', { queue: event.queue }); - break; - case 'goal_updated': - emit('goal.updated', { goal: event.goal }); - break; - case 'goal_continuation_started': - emit('goal.continuation.started', { - goal: event.goal, - continuation: event.continuation, - ...(event.prematureStopPattern - ? { prematureStopPattern: event.prematureStopPattern } - : {}), - ...(event.prematureStopCount !== undefined - ? { prematureStopCount: event.prematureStopCount } - : {}), - }); - ensureAssistantMessage(); - break; - case 'compaction': - emit( - event.phase === 'start' ? 'compaction.started' : 'compaction.completed', - { - ...(event.reason ? { reason: event.reason } : {}), - ...(event.strategy ? { strategy: event.strategy } : {}), - ...(event.outcome ? { outcome: event.outcome } : {}), - ...(event.preTokens !== undefined ? { preTokens: event.preTokens } : {}), - ...(event.preTokenSource ? { preTokenSource: event.preTokenSource } : {}), - ...(event.estimatedPendingTokens !== undefined - ? { estimatedPendingTokens: event.estimatedPendingTokens } - : {}), - ...(event.postTokens !== undefined - ? { postTokens: event.postTokens } - : {}), - ...(event.sampleAttempts !== undefined - ? { sampleAttempts: event.sampleAttempts } - : {}), - ...(event.inputReductions !== undefined - ? { inputReductions: event.inputReductions } - : {}), - ...(event.messagesOmitted !== undefined - ? { messagesOmitted: event.messagesOmitted } - : {}), - ...(event.filesOmitted !== undefined - ? { filesOmitted: event.filesOmitted } - : {}), - ...(event.imagesOmitted !== undefined - ? { imagesOmitted: event.imagesOmitted } - : {}), - ...(event.fallbackTargetTokens !== undefined - ? { fallbackTargetTokens: event.fallbackTargetTokens } - : {}), - ...(event.fallbackMessagesOmitted !== undefined - ? { fallbackMessagesOmitted: event.fallbackMessagesOmitted } - : {}), - ...(event.fallbackMessagesTruncated !== undefined - ? { fallbackMessagesTruncated: event.fallbackMessagesTruncated } - : {}), - ...(event.failureReason ? { failureReason: event.failureReason } : {}), - ...(event.memory ? { memory: event.memory } : {}), - } - ); - break; - case 'model_fallback': - emit('model.fallback', { - from: event.from, - to: event.to, - candidate: event.candidate, - candidateCount: event.candidateCount, - trigger: event.trigger, - }); - break; - - // --- 业务事件 --- - case 'task_update': - emit('task.updated', { tasks: event.tasks }); - break; - case 'goal_frontier_updated': - emit('goal.frontier.updated', { - goalId: event.goal.goalId, - goalStatus: event.goal.status, - frontier: event.frontier, - stall: event.goal.frontierStall, - }); - break; - - // stream_end is per-turn internal completion; clients consume run-level events. - default: - break; - } - }; - const runFailure = (result: LoopResult): WebAgentRunFailure => { - const toolCallsCount = result.metadata?.toolCallsCount; - return new WebAgentRunFailure({ - taskFailure: toTaskFailure( - result.error?.details ?? result.error?.message ?? 'Agent run failed' - ), - outputStarted, - toolExecutionStarted, - toolCallsCount: - typeof toolCallsCount === 'number' && Number.isInteger(toolCallsCount) - ? toolCallsCount - : -1, - }); - }; - let loopResult = await drainLoop( - agent.chatStream(content, chatContext, { - stream: true, - pendingInputOnly: options.pendingInputOnly, - preparedInputTurn: options.preparedInputTurn, - goalContinuationOnly: options.goalContinuationOnly, - outputSchema: options.outputSchema, - taskAdmission: options.taskAdmission, - }), - handleLoopEvent - ); - if (await settleRecoveryAttention(loopResult)) { - return; - } - if (!loopResult.success) { - throw runFailure(loopResult); - } - for (let followUpRun = 0; followUpRun < 20; followUpRun++) { - const requested = run.pendingFollowUpRequested === true; - run.pendingFollowUpRequested = false; - if (runtimeOwner.getPendingSteeringCount() === 0) { - if (!requested) break; - continue; - } - if (abortController.signal.aborted) break; - - loopResult = await drainLoop( - agent.chatStream('', chatContext, { - stream: true, - pendingInputOnly: true, - taskAdmission: options.taskAdmission, - }), - handleLoopEvent - ); - if (await settleRecoveryAttention(loopResult)) { - return; - } - if (!loopResult.success) { - throw runFailure(loopResult); - } - } - - await refreshSessionTaskMetadata(session); - - if ( - options.pendingResume && - abortController.signal.aborted && - abortController.signal.reason === WEB_PENDING_RESUME_DEADLINE_ABORT - ) { - throw new WebAgentRunFailure({ - taskFailure: taskFailureForCode('timeout'), - outputStarted, - toolExecutionStarted, - toolCallsCount: Number.isInteger(loopResult.metadata?.toolCallsCount) - ? (loopResult.metadata?.toolCallsCount ?? -1) - : -1, - }); - } - - if (abortController.signal.aborted || run.status === 'cancelled') { - await finalizeCancellation(); - emit('session.status', { status: 'idle' }); - return; - } - if (options.pendingResume) { - options.onPendingResumeSuccess?.(options.pendingResume); - } - - // message.complete 只在整个 run 结束时发一次(run-level 语义) - if (assistantMessageId) { - emit('message.complete', { messageId: assistantMessageId }); - } - // 保持 thinking.completed 向后兼容(Web 客户端注册了该事件,虽然当前是 no-op) - emit('thinking.completed', {}); - - run.status = 'completed'; - session.taskStatus = 'completed'; - session.taskCompletedAt ??= new Date().toISOString(); - emit('session.completed', { - runId, - outputTruncated: loopResult.metadata?.outputTruncated ?? false, - }); - emit('session.status', { status: 'idle' }); - } catch (error) { - if (runtime && options.preparedInputTurn) { - const recoveryAssessment = runtime.getTurnRecoveryAssessment(); - const cleanup = - recoveryAssessment.state === 'requires_attention' - ? runtime.finishTurn(options.preparedInputTurn.handle, { - preserveStartupRecovery: true, - }) - : runtime.finishTurn(options.preparedInputTurn.handle); - await cleanup.catch(() => undefined); - } - const deadlineExceeded = - options.pendingResume !== undefined && - abortController.signal.aborted && - abortController.signal.reason === WEB_PENDING_RESUME_DEADLINE_ABORT; - if ( - (abortController.signal.aborted && !deadlineExceeded) || - run.status === 'cancelled' - ) { - if (options.pendingResume) { - options.onPendingResumeCancelled?.(options.pendingResume); - } - cancelRun(run, 'runtime-abort'); - await finalizeCancellation(); - emit('session.status', { status: 'idle' }); - return; - } - const pendingResumeEvidence = - error instanceof WebAgentRunFailure - ? error.evidence - : deadlineExceeded - ? { - taskFailure: taskFailureForCode('timeout'), - outputStarted, - toolExecutionStarted, - toolCallsCount: -1, - } - : options.pendingResume - ? { - taskFailure: toTaskFailure(error), - outputStarted: true, - toolExecutionStarted: true, - toolCallsCount: -1, - } - : undefined; - const retryScheduled = - options.pendingResume !== undefined && - pendingResumeEvidence !== undefined && - options.onPendingResumeFailure?.( - options.pendingResume, - pendingResumeEvidence, - deadlineExceeded || (runtime?.getPendingSteeringCount() ?? 0) > 0, - deadlineExceeded - ) === true; - run.status = 'failed'; - if (retryScheduled) { - const runningMetadata = await runtime - ?.setTaskStatus('running') - .catch(() => undefined); - if (runningMetadata) syncSessionTaskMetadata(session, runningMetadata); - session.taskStatus = 'running'; - session.taskStatusReason = undefined; - session.taskFailure = undefined; - session.taskCompletedAt = undefined; - emit('session.status', { status: 'running' }); - return; - } - await refreshSessionTaskMetadata(session).catch(() => undefined); - logger.error('[SessionRoutes] Agent execution error:', error); - session.taskStatus = 'failed'; - session.taskCompletedAt ??= new Date().toISOString(); - const taskFailure = pendingResumeEvidence?.taskFailure ?? toTaskFailure(error); - if (!session.taskFailure) { - const failedMetadata = runtime - ? await runtime.setTaskStatus('failed', error).catch(() => undefined) - : await SessionService.updateSessionMetadata(session.id, session.projectPath, { - taskStatus: 'failed', - taskStatusReason: taskFailure.message, - taskFailure, - taskCompletedAt: session.taskCompletedAt, - taskOwnerPid: null, - taskQueuePosition: null, - taskQueueDepth: null, - }).catch(() => undefined); - if (failedMetadata) syncSessionTaskMetadata(session, failedMetadata); - } - session.taskStatusReason ??= taskFailure.message; - session.taskFailure ??= taskFailure; - emit('session.error', { - error: session.taskFailure.message, - taskFailure: session.taskFailure, - }); - emit('session.status', { status: 'error' }); - } finally { - if (pendingResumeDeadlineTimer) clearTimeout(pendingResumeDeadlineTimer); - options.taskAdmission?.release(); - if (options.taskAdmission) { - const stats = taskRunScheduler.getStats(); - emit('task.status', { - taskStatus: session.taskStatus, - ...(session.taskStatusReason - ? { taskStatusReason: session.taskStatusReason } - : {}), - ...(session.taskFailure ? { taskFailure: session.taskFailure } : {}), - ...(session.taskStartedAt ? { taskStartedAt: session.taskStartedAt } : {}), - ...(session.taskCompletedAt - ? { taskCompletedAt: session.taskCompletedAt } - : {}), - ...(session.taskDiffStat ? { taskDiffStat: session.taskDiffStat } : {}), - taskQueueDepth: stats.queued, - taskConcurrencyLimit: stats.maxConcurrent, - taskInFlight: stats.inFlight, - taskAdmissionPaused: stats.paused, - updatedAt: new Date().toISOString(), - }); - } - await agent?.destroy().catch(() => undefined); - runtimeLease?.release(); - projectionLease?.release(); - if (run.disposeRuntimeOnSettle && options.disposeRuntime) { - await options.disposeRuntime(session, runtime).catch((error) => { - logger.warn( - `[SessionRoutes] Failed to dispose terminal task runtime ${session.id}:`, - error - ); - }); - } - settleRun(run); - } -} - export async function respondToPermission( ref: SessionRef, permissionId: string, @@ -6816,31 +5066,28 @@ export async function respondToPermission( logger.info( `[SessionRoutes] Looking for permission ${permissionId} in session ${ref.sessionId}` ); - logger.info(`[SessionRoutes] Active runs: ${activeRuns.size}`); + logger.info(`[SessionRoutes] Active runs: ${activeRunCount()}`); - for (const [runId, run] of activeRuns.entries()) { - logger.info( - `[SessionRoutes] Checking run ${runId}: sessionId=${run.sessionId}, projectPath=${run.projectPath}, pendingPermission=${run.pendingPermission?.permissionId}` - ); - if ( - run.sessionId === ref.sessionId && - run.projectPath === ref.projectPath && - run.pendingPermission?.permissionId === permissionId - ) { - if (run.pendingPermission.details.interactionRequestId) { - await SessionInteractionService.respond( - ref.projectPath, - ref.sessionId, - permissionId, - response - ); - } - run.pendingPermission.resolve(response); - logger.info( - `[SessionRoutes] Permission ${permissionId} responded, runId: ${run.id}` + const run = findActivePermissionRun(ref, permissionId); + if (run) { + logger.info(`[SessionRoutes] Found permission ${permissionId} in run ${run.id}`); + const pendingPermission = run.pendingPermission; + if (!pendingPermission) { + throw new Error('Active permission run lost its pending interaction'); + } + if (pendingPermission.details.interactionRequestId) { + await SessionInteractionService.respond( + ref.projectPath, + ref.sessionId, + permissionId, + response ); - return true; } + pendingPermission.resolve(response); + logger.info( + `[SessionRoutes] Permission ${permissionId} responded, runId: ${run.id}` + ); + return true; } const durablePending = await SessionInteractionService.findPending( diff --git a/packages/cli/src/server/routes/sessionLoopEventProjection.ts b/packages/cli/src/server/routes/sessionLoopEventProjection.ts new file mode 100644 index 000000000..bc4fc6d1f --- /dev/null +++ b/packages/cli/src/server/routes/sessionLoopEventProjection.ts @@ -0,0 +1,264 @@ +import type { LoopEvent } from '../../agent/loop/types.js'; +import { + fitToolDisplayForSurface, + SERVER_TOOL_DETAIL_MAX_CHARS, +} from '../../tools/display/ToolResultProjector.js'; +import { + formatToolDisplay, + renderToolDisplayToString, +} from '../../ui/utils/toolFormatters.js'; +import { sanitizeToolMetadata } from './sessionToolMetadata.js'; + +export interface SessionLoopEventProjection { + type: string; + messageScoped: boolean; + properties: Record; + toolName?: string; +} + +const projectProperties = ( + event: Event, + keys: readonly Key[] +): Record => + Object.fromEntries(keys.map((key) => [String(key), event[key]])); + +const project = ( + event: Event, + type: string, + keys: readonly Key[], + messageScoped = false +): SessionLoopEventProjection => ({ + type, + messageScoped, + properties: projectProperties(event, keys), +}); + +function projectRawSessionLoopEvent( + event: LoopEvent +): SessionLoopEventProjection | undefined { + switch (event.kind) { + case 'tool_start': + if (!('function' in event.toolCall)) return undefined; + return { + type: 'tool.start', + messageScoped: true, + toolName: event.toolCall.function.name, + properties: { + toolName: event.toolCall.function.name, + toolCallId: event.toolCall.id, + arguments: event.toolCall.function.arguments, + toolKind: event.toolKind, + }, + }; + case 'tool_progress': + if (!('function' in event.toolCall)) return undefined; + return { + type: 'tool.progress', + messageScoped: true, + toolName: event.toolCall.function.name, + properties: { + toolName: event.toolCall.function.name, + toolCallId: event.toolCall.id, + ...event.update, + }, + }; + case 'tool_result': { + if (!('function' in event.toolCall)) return undefined; + const toolName = event.toolCall.function.name; + return { + type: 'tool.result', + messageScoped: true, + toolName, + properties: { + toolName, + toolCallId: event.toolCall.id, + success: event.result.success, + summary: event.result.metadata?.summary, + output: renderToolDisplayToString( + fitToolDisplayForSurface( + formatToolDisplay(toolName, event.result), + SERVER_TOOL_DETAIL_MAX_CHARS + ) + ), + metadata: sanitizeToolMetadata(toolName, event.result.metadata), + }, + }; + } + case 'provider_admission': + return project(event, 'provider.admission', [ + 'phase', + 'requestClass', + 'resource', + 'scope', + 'reason', + 'queuePosition', + 'queueDepth', + 'inFlight', + 'limit', + 'waitMs', + 'maxWaitMs', + 'recoveryRemainingMs', + ]); + case 'provider_circuit': + return project(event, 'provider.circuit', [ + 'phase', + 'reason', + 'statusCode', + 'retryAfterMs', + 'nextProbeAt', + 'openDurationMs', + 'sampleCount', + 'failureCount', + 'recoveryRemainingMs', + ]); + case 'provider_retry': + return project(event, 'provider.retry', [ + 'phase', + 'attempt', + 'maxRetries', + 'reason', + 'statusCode', + 'delayMs', + 'nextRetryAt', + 'mode', + 'recoveryBudgetMs', + 'recoveryElapsedMs', + 'recoveryRemainingMs', + 'exhaustedBy', + ]); + case 'provider_stall': + return project(event, 'provider.stall', [ + 'phase', + 'stallCount', + 'durationMs', + 'warningAfterMs', + 'timeoutMs', + 'outputStarted', + ]); + case 'action_stationarity': + return project(event, 'action.stationarity', [ + 'phase', + 'toolName', + 'runLength', + 'nudgeThreshold', + 'haltThreshold', + 'progressAware', + ]); + case 'mcp_catalog_changed': + return project( + event, + 'mcp.catalog.changed', + ['revision', 'serverName', 'added', 'removed', 'updated'], + true + ); + case 'mcp_content_changed': + return project( + event, + 'mcp.content.changed', + ['revision', 'serverName', 'contentKind', 'added', 'removed', 'updated'], + true + ); + case 'mcp_resource_updated': + return project( + event, + 'mcp.resource.updated', + ['revision', 'serverName', 'uri'], + true + ); + case 'mcp_connection_changed': + return project( + event, + 'mcp.connection.changed', + [ + 'revision', + 'serverName', + 'phase', + 'reason', + 'attempt', + 'maxAttempts', + 'nextRetryAt', + 'error', + ], + true + ); + case 'mcp_log': + return project( + event, + 'mcp.log', + [ + 'revision', + 'serverName', + 'level', + 'logger', + 'message', + 'projectedBytes', + 'dataSha256', + 'truncated', + 'detailsOmitted', + 'timestamp', + 'synthetic', + ], + true + ); + case 'mcp_instructions_changed': + return project( + event, + 'mcp.instructions.changed', + [ + 'revision', + 'serverName', + 'action', + 'reason', + 'text', + 'sourceBytes', + 'projectedBytes', + 'sha256', + 'truncated', + 'detailsOmitted', + ], + true + ); + case 'mcp_task_changed': + return project( + event, + 'mcp.task.changed', + [ + 'revision', + 'taskId', + 'serverName', + 'toolName', + 'status', + 'statusMessage', + 'createdAt', + 'updatedAt', + 'completedAt', + 'hasResult', + 'error', + ], + true + ); + case 'project_rules_loaded': + return project( + event, + 'project.rules.loaded', + ['files', 'triggerPaths', 'blockedWrite'], + true + ); + default: + return undefined; + } +} + +export function projectSessionLoopEvent( + event: LoopEvent, + options: { omitUndefined?: boolean } = {} +): SessionLoopEventProjection | undefined { + const projection = projectRawSessionLoopEvent(event); + if (!projection || !options.omitUndefined) return projection; + return { + ...projection, + properties: Object.fromEntries( + Object.entries(projection.properties).filter(([, value]) => value !== undefined) + ), + }; +} diff --git a/packages/cli/src/server/routes/sessionRunExecutor.ts b/packages/cli/src/server/routes/sessionRunExecutor.ts new file mode 100644 index 000000000..c95cc0485 --- /dev/null +++ b/packages/cli/src/server/routes/sessionRunExecutor.ts @@ -0,0 +1,737 @@ +import { nanoid } from 'nanoid'; +import { Agent } from '../../agent/Agent.js'; +import { drainLoop } from '../../agent/loop/index.js'; +import type { LoopEvent } from '../../agent/loop/types.js'; +import type { PreparedInputTurn } from '../../agent/runtime/ActiveTurnMailbox.js'; +import type { PendingResumeFailureEvidence } from '../../agent/runtime/PendingResumeRecoveryPolicy.js'; +import { SessionRuntime } from '../../agent/runtime/SessionRuntime.js'; +import type { SessionRuntimeResidencyLease } from '../../agent/runtime/SessionRuntimeResidency.js'; +import { + type TaskAdmissionHandle, + taskRunScheduler, +} from '../../agent/runtime/TaskRunScheduler.js'; +import type { ChatContext, LoopResult, UserMessageContent } from '../../agent/types.js'; +import { FOLLOW_UP_QUEUE_MAX_ITEMS } from '../../api/schemas.js'; +import type { PermissionMode } from '../../config/types.js'; +import { taskFailureForCode, toTaskFailure } from '../../context/taskFailure.js'; +import type { SessionTaskDispatch } from '../../context/types.js'; +import { createLogger, LogCategory } from '../../logging/Logger.js'; +import { SessionService } from '../../services/SessionService.js'; +import { STRUCTURED_OUTPUT_TOOL_NAME } from '../../services/StructuredOutputService.js'; +import { + CONFIRMATION_ABORTED_REASON, + type ConfirmationDetails, + type ConfirmationResponse, +} from '../../tools/types/ExecutionTypes.js'; +import { Bus } from '../bus.js'; +import type { SessionProjectionLease } from '../SessionProjectionResidency.js'; +import { projectSessionLoopEvent } from './sessionLoopEventProjection.js'; +import { + buildPendingInteractionEvent, + cancelRun, + type RunState, + refreshSessionTaskMetadata, + type SessionInfo, + sessionRefFromSession, + settleRun, + syncSessionTaskMetadata, + type WebPendingResumeAttempt, +} from './sessionRunState.js'; + +const logger = createLogger(LogCategory.SERVICE); +const WEB_PENDING_RESUME_DEADLINE_ABORT = + 'web-pending-resume-recovery-budget-exhausted'; + +class WebAgentRunFailure extends Error { + constructor(readonly evidence: PendingResumeFailureEvidence) { + super(evidence.taskFailure.message); + this.name = 'WebAgentRunFailure'; + } +} + +export interface SessionRunOptions { + pendingInputOnly?: boolean; + preparedInputTurn?: PreparedInputTurn; + goalContinuationOnly?: boolean; + outputSchema?: SessionTaskDispatch['outputSchema']; + taskAdmission?: TaskAdmissionHandle; + runtimeLease?: SessionRuntimeResidencyLease; + projectionLease?: SessionProjectionLease; + disposeRuntime?: (session: SessionInfo, runtime?: SessionRuntime) => Promise; + pendingResume?: WebPendingResumeAttempt; + onPendingResumeFailure?: ( + attempt: WebPendingResumeAttempt, + evidence: PendingResumeFailureEvidence, + workStillPending: boolean, + deadlineExceeded: boolean + ) => boolean; + onPendingResumeSuccess?: (attempt: WebPendingResumeAttempt) => boolean; + onPendingResumeCancelled?: (attempt: WebPendingResumeAttempt) => void; +} + +function getDisplayContent(content: UserMessageContent): string { + if (typeof content === 'string') return content; + return content + .filter((part) => part.type === 'text') + .map((part) => part.text) + .join('\n'); +} + +export async function executeRunAsync( + run: RunState, + session: SessionInfo, + content: UserMessageContent, + permissionMode: PermissionMode, + acquireRuntime: ( + session: SessionInfo + ) => Promise>, + options: SessionRunOptions = {} +): Promise { + const { abortController, sessionId, id: runId } = run; + const userMessageId = options.preparedInputTurn?.messageId ?? nanoid(12); + const startsFromPending = + options.pendingInputOnly === true || + options.preparedInputTurn?.mode === 'pending' || + options.goalContinuationOnly === true; + let assistantMessageId: string | undefined = startsFromPending + ? undefined + : nanoid(12); + let runtimeLease = options.runtimeLease; + const projectionLease = options.projectionLease; + let runtime: SessionRuntime | undefined; + let agent: Agent | undefined; + let outputStarted = false; + let toolExecutionStarted = false; + let pendingResumeDeadlineTimer: ReturnType | undefined; + const sessionRef = sessionRefFromSession(session); + const projectedInboxMessageIds = + options.pendingResume?.projectedInputIds ?? new Set(); + const rememberProjectedInboxMessageId = (messageId: string): void => { + if (projectedInboxMessageIds.has(messageId)) return; + if (projectedInboxMessageIds.size >= FOLLOW_UP_QUEUE_MAX_ITEMS) { + const oldest = projectedInboxMessageIds.values().next().value; + if (oldest !== undefined) projectedInboxMessageIds.delete(oldest); + } + projectedInboxMessageIds.add(messageId); + }; + const emit = (type: string, properties: Record) => { + Bus.publish(sessionRef, type, properties); + }; + + const settleRecoveryAttention = async (result: LoopResult): Promise => { + const assessment = result.metadata?.recoveryAttention; + if (!assessment || !runtime) return false; + if (options.preparedInputTurn) { + await runtime + .finishTurn(options.preparedInputTurn.handle, { + preserveStartupRecovery: true, + outcome: { + status: 'aborted', + cause: 'failed', + turnsCount: 0, + toolCallsCount: 0, + durationMs: 0, + }, + }) + .catch(() => undefined); + } + const reason = `Turn recovery requires attention: ${assessment.reason}`; + const metadata = await runtime + .setTaskStatus('interrupted', reason) + .catch(() => undefined); + if (metadata) syncSessionTaskMetadata(session, metadata); + else { + session.taskStatus = 'interrupted'; + session.taskStatusReason = reason; + session.taskCompletedAt = undefined; + } + run.status = 'attention_required'; + emit('session.status', { status: 'idle' }); + return true; + }; + + const finalizeCancellation = async (): Promise => { + const reason = String(abortController.signal.reason || 'Task run cancelled'); + if (session.taskIsolation) { + if (!runtimeLease) { + runtimeLease = await acquireRuntime(session).catch(() => undefined); + } + const taskRuntime = runtime ?? runtimeLease?.value; + if (reason === 'user-cancel') { + await taskRuntime?.discardPendingInput().catch((error) => { + logger.warn( + `[SessionRoutes] Failed to discard cancelled input for ${session.id}:`, + error + ); + }); + } + const metadata = await taskRuntime + ?.setTaskStatus('cancelled', reason) + .catch(() => undefined); + if (metadata) { + syncSessionTaskMetadata(session, metadata); + } else { + await refreshSessionTaskMetadata(session).catch(() => undefined); + } + } + session.taskStatus = 'cancelled'; + session.taskStatusReason = reason; + session.taskCompletedAt ??= new Date().toISOString(); + }; + + try { + if (options.pendingResume) { + const remainingMs = options.pendingResume.deadlineAt - Date.now(); + if (remainingMs <= 0) { + abortController.abort(WEB_PENDING_RESUME_DEADLINE_ABORT); + throw new WebAgentRunFailure({ + taskFailure: taskFailureForCode('timeout'), + outputStarted: false, + toolExecutionStarted: false, + toolCallsCount: 0, + }); + } + pendingResumeDeadlineTimer = setTimeout(() => { + abortController.abort(WEB_PENDING_RESUME_DEADLINE_ABORT); + const pendingPermission = run.pendingPermission; + run.pendingPermission = undefined; + pendingPermission?.resolve({ + approved: false, + reason: CONFIRMATION_ABORTED_REASON, + }); + if (pendingPermission) { + emit('interaction.resolved', { + requestId: pendingPermission.permissionId, + }); + } + }, remainingMs); + pendingResumeDeadlineTimer.unref?.(); + } + if (options.taskAdmission) { + await options.taskAdmission.ready; + await run.taskAdmissionUpdate; + if (abortController.signal.aborted) { + throw new Error(String(abortController.signal.reason || 'Task run cancelled')); + } + } + + session.taskStatus = 'running'; + session.taskStatusReason = undefined; + session.taskStartedAt = new Date().toISOString(); + session.taskCompletedAt = undefined; + if (!options.pendingInputOnly && !options.goalContinuationOnly) { + emit('message.created', { + messageId: userMessageId, + role: 'user', + content: getDisplayContent(content), + ...(options.preparedInputTurn?.metadata + ? { metadata: options.preparedInputTurn.metadata } + : {}), + }); + } + emit('session.status', { status: 'running' }); + if (assistantMessageId) { + emit('message.created', { + messageId: assistantMessageId, + role: 'assistant', + content: '', + }); + } + + runtimeLease ??= await acquireRuntime(session); + runtime = runtimeLease.value; + const runtimeOwner = runtime; + const structuredOutputExpected = Boolean( + options.outputSchema ?? + runtimeOwner + .getPendingSteeringMessages() + .find((pending) => pending.outputSchema)?.outputSchema + ); + agent = await Agent.createWithRuntime(runtimeOwner, { + sessionId, + ...(session.taskWorktree + ? { toolBlacklist: ['EnterWorktree', 'ExitWorktree'] } + : {}), + }); + + const requestConfirmation = async ( + details: ConfirmationDetails + ): Promise => { + const permissionId = details.interactionRequestId ?? nanoid(12); + const permissionTimeoutMs = 5 * 60 * 1_000; + + run.status = 'waiting_permission'; + + const resultPromise = new Promise((resolve) => { + const timeout = setTimeout(() => { + logger.warn( + `[SessionRoutes] Permission ${permissionId} timed out after ${permissionTimeoutMs}ms` + ); + emit('permission.timeout', { requestId: permissionId }); + resolve({ approved: false, reason: 'timeout' }); + }, permissionTimeoutMs); + + run.pendingPermission = { + permissionId, + resolve: (response) => { + clearTimeout(timeout); + resolve(response); + }, + details, + }; + }); + + const pendingInteraction = run.pendingPermission; + if (!pendingInteraction) { + throw new Error('Permission request was not registered'); + } + const interaction = buildPendingInteractionEvent(pendingInteraction); + emit(interaction.type, interaction.properties); + + logger.info( + `[SessionRoutes] Permission request created: ${permissionId}, runId: ${runId}` + ); + + const response = await resultPromise; + logger.info( + `[SessionRoutes] Permission response received: ${permissionId}, approved: ${response.approved}` + ); + if (!abortController.signal.aborted) { + run.status = 'running'; + } + if (run.pendingPermission === pendingInteraction) { + run.pendingPermission = undefined; + } + emit('interaction.resolved', { requestId: permissionId }); + + return response; + }; + + const modelContext = await SessionService.loadSessionModelContext( + session.id, + session.projectPath + ); + const chatContext: ChatContext = { + messages: modelContext, + userId: 'web-user', + sessionId, + workspaceRoot: session.projectPath, + signal: abortController.signal, + permissionMode, + onPermissionModeChange: async (nextMode) => { + session.permissionMode = nextMode; + session.updatedAt = new Date(); + emit('session.updated', { permissionMode: nextMode }); + }, + ...(session.taskWorktree ? { worktreeActive: true } : {}), + confirmationHandler: { requestConfirmation }, + }; + const ensureAssistantMessage = (): string => { + if (!assistantMessageId) { + assistantMessageId = nanoid(12); + emit('message.created', { + messageId: assistantMessageId, + role: 'assistant', + content: '', + }); + } + return assistantMessageId; + }; + + const handleLoopEvent = async (event: LoopEvent) => { + if ( + event.kind === 'tool_start' || + event.kind === 'tool_progress' || + event.kind === 'tool_result' + ) { + toolExecutionStarted = true; + } + const projection = projectSessionLoopEvent(event); + if (projection) { + if (projection.toolName === STRUCTURED_OUTPUT_TOOL_NAME) return; + emit(projection.type, { + ...(projection.messageScoped ? { messageId: ensureAssistantMessage() } : {}), + ...projection.properties, + }); + return; + } + switch (event.kind) { + case 'content_delta': + if (event.delta.length > 0) outputStarted = true; + if (structuredOutputExpected) break; + emit('message.delta', { + messageId: ensureAssistantMessage(), + delta: event.delta, + }); + break; + case 'structured_output': + outputStarted = true; + emit('structured.output', { + messageId: ensureAssistantMessage(), + output: event.output, + schemaDigest: event.schemaDigest, + }); + break; + case 'thinking_delta': + if (event.delta.length > 0) outputStarted = true; + emit('thinking.delta', { + messageId: ensureAssistantMessage(), + delta: event.delta, + }); + break; + case 'token_usage': + emit('token.usage', { ...event.usage }); + break; + case 'turn_start': + emit('turn.started', { turn: event.turn, maxTurns: event.maxTurns }); + break; + case 'turn_recovery': + emit('turn.recovery', { assessment: event.assessment }); + break; + case 'provider_recovery': + case 'turn_activity': + // SessionRuntime already publishes these authoritative projections. + break; + case 'steering_applied': + for (const message of event.messages) { + if ((message.origin ?? 'user') !== 'user') continue; + if (message.persisted) continue; + if (projectedInboxMessageIds.has(message.id)) continue; + emit('message.created', { + messageId: message.id, + role: 'user', + content: getDisplayContent(message.content), + ...(message.metadata ? { metadata: message.metadata } : {}), + ...(message.recovered ? { recovered: true } : {}), + }); + rememberProjectedInboxMessageId(message.id); + } + emit('steering.applied', { + runId, + messageIds: event.messageIds, + count: event.count, + recovered: event.recovered, + delivery: event.delivery, + queued: runtimeOwner.getPendingSteeringCount(), + }); + emit('follow_up.queue.changed', { queue: event.queue }); + break; + case 'follow_up_started': + if (assistantMessageId) { + emit('message.complete', { messageId: assistantMessageId }); + assistantMessageId = undefined; + } + emit('follow_up.started', { + runId, + queued: event.queued, + recovered: event.recovered, + }); + emit('follow_up.queue.changed', { queue: event.queue }); + ensureAssistantMessage(); + break; + case 'follow_up_queue_changed': + emit('follow_up.queue.changed', { queue: event.queue }); + break; + case 'goal_updated': + emit('goal.updated', { goal: event.goal }); + break; + case 'goal_continuation_started': + emit('goal.continuation.started', { + goal: event.goal, + continuation: event.continuation, + ...(event.prematureStopPattern + ? { prematureStopPattern: event.prematureStopPattern } + : {}), + ...(event.prematureStopCount !== undefined + ? { prematureStopCount: event.prematureStopCount } + : {}), + }); + ensureAssistantMessage(); + break; + case 'compaction': + emit( + event.phase === 'start' ? 'compaction.started' : 'compaction.completed', + { + ...(event.reason ? { reason: event.reason } : {}), + ...(event.strategy ? { strategy: event.strategy } : {}), + ...(event.outcome ? { outcome: event.outcome } : {}), + ...(event.preTokens !== undefined ? { preTokens: event.preTokens } : {}), + ...(event.preTokenSource ? { preTokenSource: event.preTokenSource } : {}), + ...(event.estimatedPendingTokens !== undefined + ? { estimatedPendingTokens: event.estimatedPendingTokens } + : {}), + ...(event.postTokens !== undefined + ? { postTokens: event.postTokens } + : {}), + ...(event.sampleAttempts !== undefined + ? { sampleAttempts: event.sampleAttempts } + : {}), + ...(event.inputReductions !== undefined + ? { inputReductions: event.inputReductions } + : {}), + ...(event.messagesOmitted !== undefined + ? { messagesOmitted: event.messagesOmitted } + : {}), + ...(event.filesOmitted !== undefined + ? { filesOmitted: event.filesOmitted } + : {}), + ...(event.imagesOmitted !== undefined + ? { imagesOmitted: event.imagesOmitted } + : {}), + ...(event.fallbackTargetTokens !== undefined + ? { fallbackTargetTokens: event.fallbackTargetTokens } + : {}), + ...(event.fallbackMessagesOmitted !== undefined + ? { fallbackMessagesOmitted: event.fallbackMessagesOmitted } + : {}), + ...(event.fallbackMessagesTruncated !== undefined + ? { fallbackMessagesTruncated: event.fallbackMessagesTruncated } + : {}), + ...(event.failureReason ? { failureReason: event.failureReason } : {}), + ...(event.memory ? { memory: event.memory } : {}), + } + ); + break; + case 'model_fallback': + emit('model.fallback', { + from: event.from, + to: event.to, + candidate: event.candidate, + candidateCount: event.candidateCount, + trigger: event.trigger, + }); + break; + case 'task_update': + emit('task.updated', { tasks: event.tasks }); + break; + case 'goal_frontier_updated': + emit('goal.frontier.updated', { + goalId: event.goal.goalId, + goalStatus: event.goal.status, + frontier: event.frontier, + stall: event.goal.frontierStall, + }); + break; + default: + break; + } + }; + const runFailure = (result: LoopResult): WebAgentRunFailure => { + const toolCallsCount = result.metadata?.toolCallsCount; + return new WebAgentRunFailure({ + taskFailure: toTaskFailure( + result.error?.details ?? result.error?.message ?? 'Agent run failed' + ), + outputStarted, + toolExecutionStarted, + toolCallsCount: + typeof toolCallsCount === 'number' && Number.isInteger(toolCallsCount) + ? toolCallsCount + : -1, + }); + }; + + let loopResult = await drainLoop( + agent.chatStream(content, chatContext, { + stream: true, + pendingInputOnly: options.pendingInputOnly, + preparedInputTurn: options.preparedInputTurn, + goalContinuationOnly: options.goalContinuationOnly, + outputSchema: options.outputSchema, + taskAdmission: options.taskAdmission, + }), + handleLoopEvent + ); + if (await settleRecoveryAttention(loopResult)) return; + if (!loopResult.success) throw runFailure(loopResult); + + for (let followUpRun = 0; followUpRun < 20; followUpRun++) { + const requested = run.pendingFollowUpRequested === true; + run.pendingFollowUpRequested = false; + if (runtimeOwner.getPendingSteeringCount() === 0) { + if (!requested) break; + continue; + } + if (abortController.signal.aborted) break; + + loopResult = await drainLoop( + agent.chatStream('', chatContext, { + stream: true, + pendingInputOnly: true, + taskAdmission: options.taskAdmission, + }), + handleLoopEvent + ); + if (await settleRecoveryAttention(loopResult)) return; + if (!loopResult.success) throw runFailure(loopResult); + } + + await refreshSessionTaskMetadata(session); + + if ( + options.pendingResume && + abortController.signal.aborted && + abortController.signal.reason === WEB_PENDING_RESUME_DEADLINE_ABORT + ) { + throw new WebAgentRunFailure({ + taskFailure: taskFailureForCode('timeout'), + outputStarted, + toolExecutionStarted, + toolCallsCount: Number.isInteger(loopResult.metadata?.toolCallsCount) + ? (loopResult.metadata?.toolCallsCount ?? -1) + : -1, + }); + } + + if (abortController.signal.aborted || run.status === 'cancelled') { + await finalizeCancellation(); + emit('session.status', { status: 'idle' }); + return; + } + if (options.pendingResume) { + options.onPendingResumeSuccess?.(options.pendingResume); + } + + if (assistantMessageId) { + emit('message.complete', { messageId: assistantMessageId }); + } + emit('thinking.completed', {}); + + run.status = 'completed'; + session.taskStatus = 'completed'; + session.taskCompletedAt ??= new Date().toISOString(); + emit('session.completed', { + runId, + outputTruncated: loopResult.metadata?.outputTruncated ?? false, + }); + emit('session.status', { status: 'idle' }); + } catch (error) { + if (runtime && options.preparedInputTurn) { + const recoveryAssessment = runtime.getTurnRecoveryAssessment(); + const cleanup = + recoveryAssessment.state === 'requires_attention' + ? runtime.finishTurn(options.preparedInputTurn.handle, { + preserveStartupRecovery: true, + }) + : runtime.finishTurn(options.preparedInputTurn.handle); + await cleanup.catch(() => undefined); + } + const deadlineExceeded = + options.pendingResume !== undefined && + abortController.signal.aborted && + abortController.signal.reason === WEB_PENDING_RESUME_DEADLINE_ABORT; + if ( + (abortController.signal.aborted && !deadlineExceeded) || + run.status === 'cancelled' + ) { + if (options.pendingResume) { + options.onPendingResumeCancelled?.(options.pendingResume); + } + cancelRun(run, 'runtime-abort'); + await finalizeCancellation(); + emit('session.status', { status: 'idle' }); + return; + } + const pendingResumeEvidence = + error instanceof WebAgentRunFailure + ? error.evidence + : deadlineExceeded + ? { + taskFailure: taskFailureForCode('timeout'), + outputStarted, + toolExecutionStarted, + toolCallsCount: -1, + } + : options.pendingResume + ? { + taskFailure: toTaskFailure(error), + outputStarted: true, + toolExecutionStarted: true, + toolCallsCount: -1, + } + : undefined; + const retryScheduled = + options.pendingResume !== undefined && + pendingResumeEvidence !== undefined && + options.onPendingResumeFailure?.( + options.pendingResume, + pendingResumeEvidence, + deadlineExceeded || (runtime?.getPendingSteeringCount() ?? 0) > 0, + deadlineExceeded + ) === true; + run.status = 'failed'; + if (retryScheduled) { + const runningMetadata = await runtime + ?.setTaskStatus('running') + .catch(() => undefined); + if (runningMetadata) syncSessionTaskMetadata(session, runningMetadata); + session.taskStatus = 'running'; + session.taskStatusReason = undefined; + session.taskFailure = undefined; + session.taskCompletedAt = undefined; + emit('session.status', { status: 'running' }); + return; + } + await refreshSessionTaskMetadata(session).catch(() => undefined); + logger.error('[SessionRoutes] Agent execution error:', error); + session.taskStatus = 'failed'; + session.taskCompletedAt ??= new Date().toISOString(); + const taskFailure = pendingResumeEvidence?.taskFailure ?? toTaskFailure(error); + if (!session.taskFailure) { + const failedMetadata = runtime + ? await runtime.setTaskStatus('failed', error).catch(() => undefined) + : await SessionService.updateSessionMetadata(session.id, session.projectPath, { + taskStatus: 'failed', + taskStatusReason: taskFailure.message, + taskFailure, + taskCompletedAt: session.taskCompletedAt, + taskOwnerPid: null, + taskQueuePosition: null, + taskQueueDepth: null, + }).catch(() => undefined); + if (failedMetadata) syncSessionTaskMetadata(session, failedMetadata); + } + session.taskStatusReason ??= taskFailure.message; + session.taskFailure ??= taskFailure; + emit('session.error', { + error: session.taskFailure.message, + taskFailure: session.taskFailure, + }); + emit('session.status', { status: 'error' }); + } finally { + if (pendingResumeDeadlineTimer) clearTimeout(pendingResumeDeadlineTimer); + options.taskAdmission?.release(); + if (options.taskAdmission) { + const stats = taskRunScheduler.getStats(); + emit('task.status', { + taskStatus: session.taskStatus, + ...(session.taskStatusReason + ? { taskStatusReason: session.taskStatusReason } + : {}), + ...(session.taskFailure ? { taskFailure: session.taskFailure } : {}), + ...(session.taskStartedAt ? { taskStartedAt: session.taskStartedAt } : {}), + ...(session.taskCompletedAt + ? { taskCompletedAt: session.taskCompletedAt } + : {}), + ...(session.taskDiffStat ? { taskDiffStat: session.taskDiffStat } : {}), + taskQueueDepth: stats.queued, + taskConcurrencyLimit: stats.maxConcurrent, + taskInFlight: stats.inFlight, + taskAdmissionPaused: stats.paused, + updatedAt: new Date().toISOString(), + }); + } + await agent?.destroy().catch(() => undefined); + runtimeLease?.release(); + projectionLease?.release(); + if (run.disposeRuntimeOnSettle && options.disposeRuntime) { + await options.disposeRuntime(session, runtime).catch((error) => { + logger.warn( + `[SessionRoutes] Failed to dispose terminal task runtime ${session.id}:`, + error + ); + }); + } + settleRun(run); + } +} diff --git a/packages/cli/src/server/routes/sessionRunState.ts b/packages/cli/src/server/routes/sessionRunState.ts new file mode 100644 index 000000000..79a054355 --- /dev/null +++ b/packages/cli/src/server/routes/sessionRunState.ts @@ -0,0 +1,298 @@ +import { LRUCache } from 'lru-cache'; +import type { PendingResumeFailureEvidence } from '../../agent/runtime/PendingResumeRecoveryPolicy.js'; +import type { TaskAdmissionHandle } from '../../agent/runtime/TaskRunScheduler.js'; +import type { + CommunicationStyleSelection, + PermissionMode, + ReasoningEffortSelection, + ResponseVerbositySelection, + ServiceTierSelection, +} from '../../config/types.js'; +import type { + SessionTaskDelivery, + SessionTaskKind, + SessionTaskPriority, + SessionTaskRetryRef, + SessionTaskWorktree, +} from '../../context/types.js'; +import type { SessionMetadata } from '../../services/SessionService.js'; +import { SessionService } from '../../services/SessionService.js'; +import { + CONFIRMATION_ABORTED_REASON, + type ConfirmationDetails, + type ConfirmationResponse, +} from '../../tools/types/ExecutionTypes.js'; +import { Bus } from '../bus.js'; +import type { SessionProjectionLease } from '../SessionProjectionResidency.js'; +import type { SessionRef } from '../sessionRef.js'; + +export interface WebPendingResumeAttempt { + attempt: number; + deadlineAt: number; + generation: number; + projectedInputIds: Set; +} + +export interface SessionInfo { + id: string; + projectPath: string; + title: string; + createdAt: Date; + updatedAt: Date; + rootId: string; + parentId?: string; + messageCount: number; + currentRunId?: string; + relationType?: 'subagent' | 'fork'; + taskStatus: SessionMetadata['taskStatus']; + taskStatusReason?: string; + taskFailure?: SessionMetadata['taskFailure']; + taskStartedAt?: string; + taskCompletedAt?: string; + taskPromptSummary?: string; + taskPriority?: SessionTaskPriority; + taskKind?: SessionTaskKind; + taskDueAt?: string; + taskModelId?: string; + selectedModelId?: string; + permissionMode?: PermissionMode; + reasoningEffort?: ReasoningEffortSelection; + serviceTier?: ServiceTierSelection; + responseVerbosity?: ResponseVerbositySelection; + communicationStyle?: CommunicationStyleSelection; + communicationStyleDigest?: string; + projectInstructionsDigest?: string; + pendingInteraction?: SessionMetadata['pendingInteraction']; + taskRetryAvailable?: boolean; + taskRetriedFrom?: SessionTaskRetryRef; + taskDelivery?: SessionTaskDelivery; + taskIsolation?: SessionMetadata['taskIsolation']; + taskSourceProjectPath?: string; + taskWorktreePath?: string; + taskWorktreeBranch?: string; + taskBaseCommit?: string; + taskDiffStat?: SessionMetadata['taskDiffStat']; + taskQueuePosition?: number; + taskQueueDepth?: number; + taskConcurrencyLimit?: number; + archivedAt?: string; + archivedBySessionId?: string; + taskWorktree?: SessionTaskWorktree; +} + +export interface RunState { + id: string; + sessionId: string; + projectPath: string; + status: + | 'queued' + | 'running' + | 'waiting_permission' + | 'attention_required' + | 'completed' + | 'failed' + | 'cancelled'; + abortController: AbortController; + pendingPermission?: { + permissionId: string; + resolve: (response: ConfirmationResponse) => void; + details: ConfirmationDetails; + }; + pendingFollowUpRequested?: boolean; + taskAdmission?: TaskAdmissionHandle; + taskAdmissionUpdate?: Promise; + disposeRuntimeOnSettle?: boolean; + pendingResume?: WebPendingResumeAttempt; + projectionLease: SessionProjectionLease; + completion?: Promise; + createdAt: Date; +} + +const activeRuns = new Map(); +const recentRuns = new LRUCache({ + max: 100, + ttl: 30 * 60 * 1_000, +}); + +export function sessionRefFromSession(session: SessionInfo): SessionRef { + return { sessionId: session.id, projectPath: session.projectPath }; +} + +function runRef(run: RunState): SessionRef { + return { sessionId: run.sessionId, projectPath: run.projectPath }; +} + +export function registerRun(run: RunState): void { + activeRuns.set(run.id, run); +} + +export function getRun(runId: string | undefined): RunState | undefined { + if (!runId) return undefined; + return activeRuns.get(runId) ?? recentRuns.get(runId); +} + +export function settleRun(run: RunState): void { + if (activeRuns.get(run.id) !== run) return; + activeRuns.delete(run.id); + recentRuns.set(run.id, run); +} + +export function forgetRun(runId: string): void { + activeRuns.delete(runId); + recentRuns.delete(runId); +} + +export function isActiveRun(run: RunState | undefined): run is RunState { + return ( + run?.status === 'queued' || + run?.status === 'running' || + run?.status === 'waiting_permission' + ); +} + +export function cancelRun(run: RunState, reason = 'user-cancel'): boolean { + if ( + run.status === 'cancelled' || + run.status === 'completed' || + run.status === 'failed' || + run.status === 'attention_required' + ) { + return false; + } + + const pendingPermission = run.pendingPermission; + run.pendingPermission = undefined; + pendingPermission?.resolve({ + approved: false, + reason: CONFIRMATION_ABORTED_REASON, + }); + if (pendingPermission) { + Bus.publish(runRef(run), 'interaction.resolved', { + requestId: pendingPermission.permissionId, + }); + } + run.abortController.abort(reason); + run.status = 'cancelled'; + Bus.publish(runRef(run), 'run.cancelled', { runId: run.id }); + return true; +} + +export function resetSessionRuns(reason: string): void { + for (const run of activeRuns.values()) { + cancelRun(run, reason); + } + activeRuns.clear(); + recentRuns.clear(); +} + +export function listActiveRuns(): RunState[] { + return [...activeRuns.values()]; +} + +export function activeRunCount(): number { + return activeRuns.size; +} + +export function findActivePermissionRun( + ref: SessionRef, + permissionId: string +): RunState | undefined { + return [...activeRuns.values()].find( + (run) => + run.sessionId === ref.sessionId && + run.projectPath === ref.projectPath && + run.pendingPermission?.permissionId === permissionId + ); +} + +export function buildPendingInteractionEvent( + pending: NonNullable, + replayed = false +): { type: string; properties: Record } { + const { permissionId, details } = pending; + if (details.type === 'askUserQuestion' && details.questions) { + return { + type: 'question.required', + properties: { + requestId: permissionId, + toolCallId: details.toolCallId ?? permissionId, + questions: details.questions, + details, + ...(replayed ? { replayed: true } : {}), + }, + }; + } + if (details.type === 'mcpElicitation' && details.mcpElicitation) { + return { + type: 'elicitation.required', + properties: { + requestId: permissionId, + toolCallId: details.toolCallId ?? permissionId, + elicitation: details.mcpElicitation, + ...(replayed ? { replayed: true } : {}), + }, + }; + } + + return { + type: 'permission.asked', + properties: { + requestId: permissionId, + toolName: details.toolName, + description: details.message, + args: details.args, + details, + ...(replayed ? { replayed: true } : {}), + }, + }; +} + +export function syncSessionTaskMetadata( + session: SessionInfo, + metadata: SessionMetadata +): void { + session.title = metadata.title ?? session.title; + session.taskStatus = metadata.taskStatus; + session.taskStatusReason = metadata.taskStatusReason; + session.taskFailure = metadata.taskFailure; + session.taskStartedAt = metadata.taskStartedAt; + session.taskCompletedAt = metadata.taskCompletedAt; + session.taskPromptSummary = metadata.taskPromptSummary; + session.taskPriority = metadata.taskPriority; + session.taskKind = metadata.taskKind; + session.taskDueAt = metadata.taskDueAt; + session.taskModelId = metadata.taskModelId; + session.selectedModelId = metadata.selectedModelId; + session.permissionMode = metadata.permissionMode as PermissionMode | undefined; + session.reasoningEffort = metadata.reasoningEffort; + session.serviceTier = metadata.serviceTier; + session.responseVerbosity = metadata.responseVerbosity; + session.communicationStyle = metadata.communicationStyle; + session.communicationStyleDigest = metadata.communicationStyleDigest; + session.projectInstructionsDigest = metadata.projectInstructionsDigest; + session.pendingInteraction = metadata.pendingInteraction; + session.taskRetryAvailable = metadata.taskRetryAvailable; + session.taskRetriedFrom = metadata.taskRetriedFrom; + session.taskDelivery = metadata.taskDelivery; + session.taskIsolation = metadata.taskIsolation; + session.taskSourceProjectPath = metadata.taskSourceProjectPath; + session.taskWorktreePath = metadata.taskWorktreePath; + session.taskWorktreeBranch = metadata.taskWorktreeBranch; + session.taskBaseCommit = metadata.taskBaseCommit; + session.taskDiffStat = metadata.taskDiffStat; + session.taskQueuePosition = metadata.taskQueuePosition; + session.taskQueueDepth = metadata.taskQueueDepth; + session.taskConcurrencyLimit = metadata.taskConcurrencyLimit; + session.archivedAt = metadata.archivedAt; + session.archivedBySessionId = metadata.archivedBySessionId; + session.messageCount = metadata.messageCount; + session.updatedAt = new Date(metadata.lastMessageTime); +} + +export async function refreshSessionTaskMetadata(session: SessionInfo): Promise { + const metadata = await SessionService.findSessionMetadata( + session.id, + session.projectPath + ); + if (metadata) syncSessionTaskMetadata(session, metadata); +} diff --git a/packages/cli/src/server/routes/sessionToolMetadata.ts b/packages/cli/src/server/routes/sessionToolMetadata.ts new file mode 100644 index 000000000..986e0b9ea --- /dev/null +++ b/packages/cli/src/server/routes/sessionToolMetadata.ts @@ -0,0 +1,400 @@ +import { + MAX_BROWSER_DIAGNOSTIC_RESULT_ENTRIES, + MAX_BROWSER_ID_BYTES, + MAX_BROWSER_ORIGIN_BYTES, + MAX_BROWSER_PROJECTED_URL_BYTES, + MAX_BROWSER_REF_BYTES, + MAX_BROWSER_SCREENSHOT_BYTES, + MAX_BROWSER_TITLE_BYTES, +} from '../../browser/constants.js'; +import { isBrowserToolName } from '../../browser/types.js'; +import type { ToolResultMetadata } from '../../tools/types/ToolTypes.js'; + +function sanitizeToolAdmissionMetadata( + value: unknown +): Record | undefined { + if (!value || typeof value !== 'object' || Array.isArray(value)) return undefined; + const admission = value as Record; + const { code, kind, limit, reason, scope } = admission; + if ( + (code !== 'tool_busy' && code !== 'tool_batch_full') || + (reason !== 'queue_full' && reason !== 'wait_timeout' && reason !== 'turn_limit') || + (scope !== 'global' && scope !== 'session') || + typeof admission.retryable !== 'boolean' || + !Number.isSafeInteger(limit) || + (limit as number) <= 0 || + (kind !== undefined && + kind !== 'readonly' && + kind !== 'write' && + kind !== 'execute') + ) { + return undefined; + } + return { + code, + reason, + scope, + retryable: admission.retryable, + ...(kind === undefined ? {} : { kind }), + limit, + }; +} + +export function sanitizeToolMetadata( + toolName: string, + metadata: ToolResultMetadata | undefined +): ToolResultMetadata | undefined { + if (!metadata || typeof metadata !== 'object') return metadata; + const sanitized = { ...(metadata as Record) }; + const toolAdmission = sanitizeToolAdmissionMetadata(sanitized.tool_admission); + if (toolAdmission) sanitized.tool_admission = toolAdmission; + else delete sanitized.tool_admission; + if (isBrowserToolName(toolName)) { + const source = + sanitized.browser && + typeof sanitized.browser === 'object' && + !Array.isArray(sanitized.browser) + ? (sanitized.browser as Record) + : {}; + const projected: Record = {}; + const boundedString = (key: string, maximum: number, pattern?: RegExp): void => { + const value = source[key]; + if ( + typeof value === 'string' && + Buffer.byteLength(value) <= maximum && + (!pattern || pattern.test(value)) + ) { + projected[key] = value; + } + }; + boundedString('action', 64); + boundedString('status', 16, /^(?:ok|warning|error)$/); + boundedString('pageId', MAX_BROWSER_ID_BYTES, /^browser_page_[a-f0-9-]+$/); + boundedString('snapshotId', MAX_BROWSER_ID_BYTES, /^browser_snapshot_[a-f0-9-]+$/); + boundedString('origin', MAX_BROWSER_ORIGIN_BYTES); + boundedString('candidateOrigin', MAX_BROWSER_ORIGIN_BYTES); + boundedString('url', MAX_BROWSER_PROJECTED_URL_BYTES); + boundedString('title', MAX_BROWSER_TITLE_BYTES); + boundedString('errorCode', 64, /^browser_[a-z_]+$/); + if (typeof source.truncated === 'boolean') { + projected.truncated = source.truncated; + } + if ( + typeof source.actionApplied === 'boolean' || + source.actionApplied === 'unknown' + ) { + projected.actionApplied = source.actionApplied; + } + if (typeof source.sideEffectsUncertain === 'boolean') { + projected.sideEffectsUncertain = source.sideEffectsUncertain; + } + if ( + typeof source.diagnosticCount === 'number' && + Number.isSafeInteger(source.diagnosticCount) && + source.diagnosticCount >= 0 && + source.diagnosticCount <= MAX_BROWSER_DIAGNOSTIC_RESULT_ENTRIES + ) { + projected.diagnosticCount = source.diagnosticCount; + } + if ( + source.interaction && + typeof source.interaction === 'object' && + !Array.isArray(source.interaction) + ) { + const interaction = source.interaction as Record; + const allowedActions = new Set([ + 'click', + 'hover', + 'fill', + 'type', + 'press', + 'select', + 'check', + 'uncheck', + 'scroll', + ]); + if ( + typeof interaction.action === 'string' && + allowedActions.has(interaction.action) + ) { + const projectedInteraction: Record = { + action: interaction.action, + }; + if ( + typeof interaction.ref === 'string' && + Buffer.byteLength(interaction.ref) <= MAX_BROWSER_REF_BYTES && + /^[a-z][a-z0-9]*$/.test(interaction.ref) + ) { + projectedInteraction.ref = interaction.ref; + } + const boundedNumber = ( + value: unknown, + minimum: number, + maximum: number + ): value is number => + typeof value === 'number' && + Number.isFinite(value) && + value >= minimum && + value <= maximum; + if ( + interaction.viewport && + typeof interaction.viewport === 'object' && + !Array.isArray(interaction.viewport) + ) { + const viewport = interaction.viewport as Record; + if ( + boundedNumber(viewport.width, 1, 16_384) && + boundedNumber(viewport.height, 1, 16_384) + ) { + projectedInteraction.viewport = { + width: viewport.width, + height: viewport.height, + }; + } + } + if ( + interaction.targetBox && + typeof interaction.targetBox === 'object' && + !Array.isArray(interaction.targetBox) + ) { + const targetBox = interaction.targetBox as Record; + if ( + boundedNumber(targetBox.x, -16_384, 32_768) && + boundedNumber(targetBox.y, -16_384, 32_768) && + boundedNumber(targetBox.width, 0, 16_384) && + boundedNumber(targetBox.height, 0, 16_384) + ) { + projectedInteraction.targetBox = { + x: targetBox.x, + y: targetBox.y, + width: targetBox.width, + height: targetBox.height, + }; + } + } + projected.interaction = projectedInteraction; + } + } + if ( + source.artifact && + typeof source.artifact === 'object' && + !Array.isArray(source.artifact) + ) { + const artifact = source.artifact as Record; + if ( + typeof artifact.id === 'string' && + /^[a-f0-9]{64}$/.test(artifact.id) && + artifact.sha256 === artifact.id && + artifact.kind === 'image' && + artifact.mimeType === 'image/png' && + typeof artifact.size === 'number' && + Number.isSafeInteger(artifact.size) && + artifact.size >= 0 && + artifact.size <= MAX_BROWSER_SCREENSHOT_BYTES && + artifact.persisted === true + ) { + projected.artifact = { + id: artifact.id, + sha256: artifact.sha256, + kind: artifact.kind, + mimeType: artifact.mimeType, + size: artifact.size, + persisted: true, + ...(typeof artifact.path === 'string' && + Buffer.byteLength(artifact.path) <= 8_192 + ? { path: artifact.path } + : {}), + }; + } + } + return { + ...(typeof sanitized.summary === 'string' + ? { summary: sanitized.summary.slice(0, 512) } + : {}), + browser: projected, + ...(toolAdmission ? { tool_admission: toolAdmission } : {}), + } as ToolResultMetadata; + } + if (toolName === 'Bash') { + const projected: Record = {}; + const stringFields = ['message', 'signal', 'status', 'summary'] as const; + const booleanFields = [ + 'aborted', + 'acp_mode', + 'admission_failed', + 'auto_backgrounded', + 'background', + 'capture_truncated', + 'finalization_failed', + 'has_stderr', + 'output_accounting_complete', + 'output_truncated', + 'projection_truncated', + 'sandbox_required', + 'sandboxed', + 'stderr_projection_truncated', + 'stdout_projection_truncated', + 'terminal_output_merged', + 'timeout', + ] as const; + const numberFields = [ + 'execution_time', + 'foreground_budget_ms', + 'pid', + 'raw_output_bytes', + 'stderr_length', + 'stderr_omitted_bytes', + 'stderr_retained_bytes', + 'stderr_total_bytes', + 'stdout_length', + 'stdout_omitted_bytes', + 'stdout_retained_bytes', + 'stdout_total_bytes', + ] as const; + for (const field of stringFields) { + const value = sanitized[field]; + if (typeof value === 'string') projected[field] = value.slice(0, 8_192); + } + for (const field of booleanFields) { + const value = sanitized[field]; + if (typeof value === 'boolean') projected[field] = value; + } + for (const field of numberFields) { + const value = sanitized[field]; + if (typeof value === 'number' && Number.isSafeInteger(value) && value >= 0) { + projected[field] = value; + } + } + if ( + sanitized.exit_code === null || + (typeof sanitized.exit_code === 'number' && + Number.isSafeInteger(sanitized.exit_code)) + ) { + projected.exit_code = sanitized.exit_code; + } + if ( + sanitized.terminal_transport === 'local' || + sanitized.terminal_transport === 'acp' || + sanitized.terminal_transport === 'local_fallback' + ) { + projected.terminal_transport = sanitized.terminal_transport; + } + if ( + sanitized.background_reason === 'explicit' || + sanitized.background_reason === 'foreground_budget' + ) { + projected.background_reason = sanitized.background_reason; + } + for (const field of ['bash_id', 'shell_id'] as const) { + const value = sanitized[field]; + if ( + typeof value === 'string' && + value.length <= 128 && + /^bash_[A-Za-z0-9-]+$/.test(value) + ) { + projected[field] = value; + } + } + if (toolAdmission) projected.tool_admission = toolAdmission; + const backgroundAdmission = sanitized.background_shell_admission; + if ( + backgroundAdmission && + typeof backgroundAdmission === 'object' && + !Array.isArray(backgroundAdmission) + ) { + const value = backgroundAdmission as Record; + if ( + value.code === 'background_shell_busy' && + (value.scope === 'session' || value.scope === 'global') && + value.retryable === true && + Number.isSafeInteger(value.limit) && + (value.limit as number) > 0 + ) { + projected.background_shell_admission = { + code: value.code, + scope: value.scope, + retryable: value.retryable, + limit: value.limit, + }; + } + } + return projected as ToolResultMetadata; + } + const MAX_INLINE_CONTENT = 200000; + const safeInteger = (value: unknown, maximum: number): number => + typeof value === 'number' && + Number.isSafeInteger(value) && + value >= 0 && + value <= maximum + ? value + : 0; + if ( + typeof sanitized.oldContent === 'string' && + sanitized.oldContent.length > MAX_INLINE_CONTENT + ) { + delete sanitized.oldContent; + } + if ( + typeof sanitized.newContent === 'string' && + sanitized.newContent.length > MAX_INLINE_CONTENT + ) { + delete sanitized.newContent; + } + if ( + sanitized.mcpResult && + typeof sanitized.mcpResult === 'object' && + !Array.isArray(sanitized.mcpResult) + ) { + const result = sanitized.mcpResult as Record; + const artifacts = Array.isArray(result.artifacts) + ? result.artifacts.slice(0, 64).flatMap((value) => { + if (!value || typeof value !== 'object' || Array.isArray(value)) return []; + const artifact = value as Record; + const artifactKinds = new Set(['text', 'image', 'audio', 'resource']); + if ( + typeof artifact.id !== 'string' || + !/^[a-f0-9]{64}$/.test(artifact.id) || + typeof artifact.sha256 !== 'string' || + artifact.sha256 !== artifact.id || + typeof artifact.kind !== 'string' || + !artifactKinds.has(artifact.kind) || + safeInteger(artifact.size, 64 * 1024 * 1024) !== artifact.size || + typeof artifact.persisted !== 'boolean' + ) { + return []; + } + return [ + { + id: artifact.id.slice(0, 128), + sha256: artifact.sha256.slice(0, 128), + kind: artifact.kind, + size: artifact.size, + persisted: artifact.persisted, + ...(typeof artifact.mimeType === 'string' + ? { mimeType: artifact.mimeType.slice(0, 256) } + : {}), + ...(typeof artifact.sourceUri === 'string' + ? { sourceUri: artifact.sourceUri.slice(0, 8_192) } + : {}), + ...(typeof artifact.path === 'string' + ? { path: artifact.path.slice(0, 8_192) } + : {}), + }, + ]; + }) + : []; + sanitized.mcpResult = { + isError: result.isError === true, + contentCount: safeInteger(result.contentCount, 64), + textBytes: safeInteger(result.textBytes, 4 * 1024 * 1024), + structuredBytes: safeInteger(result.structuredBytes, 4 * 1024 * 1024), + artifactCount: safeInteger(result.artifactCount, 64), + truncated: result.truncated === true, + binaryOmitted: result.binaryOmitted === true, + artifacts, + }; + } else { + delete sanitized.mcpResult; + } + return sanitized as ToolResultMetadata; +} diff --git a/packages/cli/src/services/ChatServiceInterface.ts b/packages/cli/src/services/ChatServiceInterface.ts index ae36c56a0..979da3d03 100644 --- a/packages/cli/src/services/ChatServiceInterface.ts +++ b/packages/cli/src/services/ChatServiceInterface.ts @@ -1,7 +1,4 @@ -/** - * ChatService 接口抽象 - * 定义统一的聊天服务接口,支持多种 API 提供商 - */ +/** ChatService 接口抽象 定义统一的聊天服务接口,支持多种 API 提供商 */ import type { ConstrainedSamplingConfig, @@ -49,17 +46,12 @@ import type { ProviderStallEvent } from './pi/providerStall.js'; const logger = createLogger(LogCategory.SERVICE); -/** - * Anthropic Prompt Caching 配置 - * 用于标记可缓存的内容,减少 token 消耗(成本降低 90%,延迟降低 85%) - */ +/** Anthropic Prompt Caching 配置 用于标记可缓存的内容,减少 token 消耗(成本降低 90%,延迟降低 85%) */ export interface AnthropicCacheControl { type: 'ephemeral'; } -/** - * Provider 特定选项 - */ +/** Provider 特定选项 */ export interface ProviderOptions { anthropic?: { cacheControl?: AnthropicCacheControl; @@ -80,18 +72,14 @@ export interface ChatFallbackModel extends ModelRef { }; } -/** - * 多模态内容部分 - 文本 - */ +/** 多模态内容部分 - 文本 */ interface TextContentPart { type: 'text'; text: string; providerOptions?: ProviderOptions; } -/** - * 多模态内容部分 - 图片 (OpenAI Vision API 格式) - */ +/** 多模态内容部分 - 图片 (OpenAI Vision API 格式) */ interface ImageContentPart { type: 'image_url'; image_url: { @@ -99,15 +87,10 @@ interface ImageContentPart { }; } -/** - * 多模态内容部分 - */ +/** 多模态内容部分 */ export type ContentPart = TextContentPart | ImageContentPart; -/** - * 消息类型 - * content 支持纯文本或多模态内容(文本+图片) - */ +/** 消息类型 content 支持纯文本或多模态内容(文本+图片) */ export type Message = { id?: string; role: MessageRole; @@ -119,10 +102,7 @@ export type Message = { metadata?: JsonValue; }; -/** - * ChatConfig - 聊天服务所需的配置 - * 注意:这些字段现在从 ModelConfig 中获取,而非直接从 BladeConfig - */ +/** ChatConfig - 聊天服务所需的配置 注意:这些字段现在从 ModelConfig 中获取,而非直接从 BladeConfig */ export interface ChatConfig { provider: ProviderType; apiKey?: string; @@ -157,9 +137,7 @@ export interface ChatConfig { modelCatalog?: PiModelCatalog; } -/** - * 聊天响应 - */ +/** 聊天响应 */ export interface UsageInfo { promptTokens: number; completionTokens: number; @@ -237,9 +215,7 @@ export interface ChatToolDefinition { */ export type StreamToolCall = ChatCompletionMessageToolCall | StreamToolCallDelta; -/** - * 流式响应块 - */ +/** 流式响应块 */ export interface StreamChunk { content?: string; reasoningContent?: string; @@ -253,14 +229,9 @@ export interface StreamChunk { providerStall?: ProviderStallEvent; } -/** - * 聊天服务接口 - * 所有 Provider 实现必须实现此接口 - */ +/** 聊天服务接口 所有 Provider 实现必须实现此接口 */ export interface IChatService { - /** - * 发送聊天请求(非流式) - */ + /** 发送聊天请求(非流式) */ chat( messages: Message[], tools?: ChatToolDefinition[], @@ -268,9 +239,7 @@ export interface IChatService { options?: ChatRequestOptions ): Promise; - /** - * 发送聊天请求(流式) - */ + /** 发送聊天请求(流式) */ streamChat( messages: Message[], tools?: ChatToolDefinition[], @@ -278,14 +247,10 @@ export interface IChatService { options?: ChatRequestOptions ): AsyncGenerator; - /** - * 获取当前配置 - */ + /** 获取当前配置 */ getConfig(): ChatConfig; - /** - * 更新配置 - */ + /** 更新配置 */ updateConfig(newConfig: Partial): void; } diff --git a/packages/cli/src/services/FileSystemService.ts b/packages/cli/src/services/FileSystemService.ts index 643e2948f..c90845829 100644 --- a/packages/cli/src/services/FileSystemService.ts +++ b/packages/cli/src/services/FileSystemService.ts @@ -1,15 +1,8 @@ -/** - * 文件系统服务 - * - * 抽象文件操作,支持本地和远程(ACP)两种实现。 - * 工具层统一通过此接口访问文件系统。 - */ +/** 文件系统服务 抽象文件操作,支持本地和远程(ACP)两种实现。 工具层统一通过此接口访问文件系统。 */ import * as fs from 'fs/promises'; -/** - * 文件统计信息 - */ +/** 文件统计信息 */ export interface FileStat { size: number; isDirectory: boolean; @@ -17,9 +10,7 @@ export interface FileStat { mtime: Date; } -/** - * 文件系统服务接口 - */ +/** 文件系统服务接口 */ export interface FileSystemService { // 基础操作 readTextFile(filePath: string): Promise; @@ -35,9 +26,7 @@ export interface FileSystemService { ): Promise; } -/** - * 本地文件系统服务(默认实现) - */ +/** 本地文件系统服务(默认实现) */ export class LocalFileSystemService implements FileSystemService { async readTextFile(filePath: string): Promise { return fs.readFile(filePath, 'utf-8'); @@ -89,10 +78,7 @@ export class LocalFileSystemService implements FileSystemService { // ==================== 服务获取 ==================== -/** - * 当前活跃的文件系统服务 - * 默认使用本地文件系统,ACP 模式下会被替换 - */ +/** 当前活跃的文件系统服务 默认使用本地文件系统,ACP 模式下会被替换 */ let currentFileSystemService: FileSystemService = new LocalFileSystemService(); /** diff --git a/packages/cli/src/services/GracefulShutdown.ts b/packages/cli/src/services/GracefulShutdown.ts index ec0d05de3..df03815ad 100644 --- a/packages/cli/src/services/GracefulShutdown.ts +++ b/packages/cli/src/services/GracefulShutdown.ts @@ -1,12 +1,6 @@ /** - * 优雅退出管理器 - * - * 负责: - * 1. 全局崩溃捕获 (uncaughtException/unhandledRejection) - * 2. 信号处理 (SIGINT/SIGTERM) - * 3. 资源清理和会话保存 - * 4. 恢复终端状态(光标等) - * 5. 执行 SessionEnd hooks + * 优雅退出管理器

负责: 1. 全局崩溃捕获 (uncaughtException/unhandledRejection) 2. 信号处理 + * (SIGINT/SIGTERM) 3. 资源清理和会话保存 4. 恢复终端状态(光标等) 5. 执行 SessionEnd hooks */ import { PermissionMode } from '../config/types.js'; @@ -16,10 +10,7 @@ import { createLogger, LogCategory, shutdownLogger } from '../logging/Logger.js' import { getState } from '../store/vanilla.js'; import { getCwd } from '../utils/cwd.js'; -/** - * 恢复终端状态 - * 确保退出时光标可见、终端模式和键盘协议正常 - */ +/** 恢复终端状态 确保退出时光标可见、终端模式和键盘协议正常 */ function restoreTerminal(): void { if (process.stdin.isTTY && typeof process.stdin.setRawMode === 'function') { try { @@ -57,9 +48,7 @@ type ExitReason = | 'esc' | 'normal'; -/** - * 将 ExitReason 映射到 SessionEnd hook 的 reason - */ +/** 将 ExitReason 映射到 SessionEnd hook 的 reason */ function mapExitReasonToHookReason(reason: ExitReason): SessionEndInput['reason'] { switch (reason) { case 'SIGINT': @@ -76,10 +65,7 @@ function mapExitReasonToHookReason(reason: ExitReason): SessionEndInput['reason' } } -/** - * 优雅退出管理器 - * 单例模式,确保全局只有一个实例处理退出逻辑 - */ +/** 优雅退出管理器 单例模式,确保全局只有一个实例处理退出逻辑 */ class GracefulShutdownManager { private static instance: GracefulShutdownManager | null = null; @@ -98,10 +84,7 @@ class GracefulShutdownManager { return GracefulShutdownManager.instance; } - /** - * 初始化全局错误处理器 - * 应该在应用启动时调用一次 - */ + /** 初始化全局错误处理器 应该在应用启动时调用一次 */ initialize(): void { if (this.initialized) { logger.debug('[GracefulShutdown] 已初始化,跳过重复初始化'); @@ -125,8 +108,7 @@ class GracefulShutdownManager { this.shutdown('SIGTERM', 0); }); - // 处理 SIGINT(Ctrl+C 或 kill -2) - // 在交互模式下实现双击退出逻辑,与键盘 Ctrl+C 行为一致 + // 处理 SIGINT(Ctrl+C 或 kill -2) 在交互模式下实现双击退出逻辑,与键盘 Ctrl+C 行为一致 process.on('SIGINT', () => { if (!process.stdin.isTTY || !process.stdout.isTTY) { void this.shutdown('SIGINT', 0); @@ -151,10 +133,7 @@ class GracefulShutdownManager { logger.debug('[GracefulShutdown] 全局错误处理器已初始化'); } - /** - * 注册清理函数 - * 在退出时按注册的逆序执行(后注册的先执行) - */ + /** 注册清理函数 在退出时按注册的逆序执行(后注册的先执行) */ registerCleanup(handler: CleanupHandler): () => void { this.cleanupHandlers.push(handler); logger.debug( @@ -173,9 +152,7 @@ class GracefulShutdownManager { }; } - /** - * 处理致命错误 - */ + /** 处理致命错误 */ private handleFatalError(type: ExitReason, error: Error): void { // 防止递归错误 if (this.isShuttingDown) { @@ -202,9 +179,7 @@ class GracefulShutdownManager { this.shutdown(type, 1); } - /** - * 执行优雅退出 - */ + /** 执行优雅退出 */ async shutdown(reason: ExitReason, exitCode: number = 0): Promise { if (this.isShuttingDown) { logger.debug('[GracefulShutdown] 已在退出过程中,跳过重复退出'); @@ -278,9 +253,7 @@ class GracefulShutdownManager { }, 100); } - /** - * 执行所有清理函数 - */ + /** 执行所有清理函数 */ private async runCleanupHandlers(): Promise { // 逆序执行 const handlers = [...this.cleanupHandlers].reverse(); @@ -298,16 +271,12 @@ class GracefulShutdownManager { } } - /** - * 检查是否正在退出 - */ + /** 检查是否正在退出 */ isExiting(): boolean { return this.isShuttingDown; } - /** - * 重置状态(仅用于测试) - */ + /** 重置状态(仅用于测试) */ reset(): void { this.isShuttingDown = false; this.cleanupHandlers = []; diff --git a/packages/cli/src/services/SessionService.ts b/packages/cli/src/services/SessionService.ts index fc3a6424c..97741ada5 100644 --- a/packages/cli/src/services/SessionService.ts +++ b/packages/cli/src/services/SessionService.ts @@ -1,7 +1,4 @@ -/** - * 会话管理服务 - * 负责加载和恢复历史会话 - */ +/** 会话管理服务 负责加载和恢复历史会话 */ import type { BigIntStats } from 'node:fs'; import { readdir, readFile, rm, stat } from 'node:fs/promises'; @@ -621,6 +618,69 @@ export type RemoteSessionMetadataUpdate = Pick< | 'projectInstructionsDigest' >; +type SessionUpdatedData = Extract['data']; + +const SHARED_METADATA_UPDATE_KEYS = [ + 'title', + 'taskStatus', + 'taskStatusReason', + 'taskFailure', + 'taskStartedAt', + 'taskCompletedAt', + 'taskOwnerPid', + 'taskPromptSummary', + 'taskPriority', + 'taskKind', + 'taskDueAt', + 'taskModelId', + 'taskQueuePosition', + 'taskQueueDepth', + 'taskConcurrencyLimit', + 'selectedModelId', + 'permissionMode', + 'reasoningEffort', + 'serviceTier', + 'responseVerbosity', + 'communicationStyle', + 'communicationStyleDigest', + 'projectInstructionsDigest', +] as const satisfies readonly (keyof SessionMetadataUpdate)[]; + +const LOCAL_METADATA_UPDATE_KEYS = [ + 'taskDispatch', + 'taskRetriedFrom', + 'taskDelivery', + 'taskIsolation', + 'taskSourceProjectPath', + 'taskWorktree', + 'taskDiffStat', +] as const satisfies readonly (keyof SessionMetadataUpdate)[]; + +function buildSessionUpdatedData( + sessionId: string, + update: SessionMetadataUpdate, + updatedAt: string, + includeLocalFields: boolean +): SessionUpdatedData { + const data: SessionUpdatedData = { sessionId, updatedAt }; + const keys: readonly (keyof SessionMetadataUpdate)[] = includeLocalFields + ? [...SHARED_METADATA_UPDATE_KEYS, ...LOCAL_METADATA_UPDATE_KEYS] + : SHARED_METADATA_UPDATE_KEYS; + for (const key of keys) { + const value = update[key]; + if (value !== undefined) { + Reflect.set( + data, + key, + key === 'taskDueAt' && value !== null + ? new Date(String(value)).toISOString() + : value + ); + } + } + return data; +} + export interface SessionPage { sessions: SessionMetadata[]; nextCursor?: string; @@ -930,13 +990,10 @@ function filterRemoteArchiveState( return projected.filter((session) => Boolean(session.archivedAt) === archived); } -/** - * 会话管理服务 - */ +/** 会话管理服务 */ export class SessionService { /** - * 将加载到的会话消息转换为 UI 安全的 SessionMessage。 - * 过滤掉 tool / system 等内部消息,仅从 ContentPart[] 中提取文本, + * 将加载到的会话消息转换为 UI 安全的 SessionMessage。 过滤掉 tool / system 等内部消息,仅从 ContentPart[] 中提取文本, * 避免把 、工具调用 JSON、summary 等内部内容泄露给用户或污染历史。 */ static toUISafeMessages(messages: Message[]): SessionMessage[] { @@ -1026,10 +1083,7 @@ export class SessionService { const projected = await this.listRemoteSessionPageFromProjection(normalized); if (projected) return projected; - const stored = await this.scanRemoteStoredSessions( - normalized, - normalized.cursor ? 5_000 : 0 - ); + const stored = await this.scanRemoteStoredSessions(normalized); const filtered = this.toRemoteCatalogEntries(stored).sort( compareRemoteSessionCatalogItems ); @@ -1164,7 +1218,7 @@ export class SessionService { ): Promise { resolveRemoteSessionCursorBoundary(options); try { - const sessions = await this.scanRemoteStoredSessionsFromProjection(options, 0); + const sessions = await this.scanRemoteStoredSessionsFromProjection(options); if (!sessions) return null; const page = paginateRemoteSessionCatalog( this.toRemoteCatalogEntries(sessions).sort(compareRemoteSessionCatalogItems), @@ -1179,10 +1233,7 @@ export class SessionService { } } - /** - * 列出所有可用会话 - * 扫描 ~/.blade/projects/ 目录下的所有 JSONL 文件 - */ + /** 列出所有可用会话 扫描 ~/.blade/projects/ 目录下的所有 JSONL 文件 */ static async listSessions( options: SessionScanOptions = {} ): Promise { @@ -1215,7 +1266,7 @@ export class SessionService { } : options ); - const stored = await this.scanRemoteStoredSessions(normalized, 0); + const stored = await this.scanRemoteStoredSessions(normalized); const seenSessions = new Set(); return this.toRemoteCatalogEntries(stored) .sort(compareRemoteSessionCatalogItems) @@ -1849,155 +1900,17 @@ export class SessionService { const rootId = sourceCreated.data.rootId || sourceSessionId; const gitBranch = detectGitBranch(targetProjectPath); const version = getVersion(); - const { - status: _sourceStatus, - taskStatus: _sourceTaskStatus, - taskStatusReason: _sourceTaskStatusReason, - taskFailure: _sourceTaskFailure, - taskStartedAt: _sourceTaskStartedAt, - taskCompletedAt: _sourceTaskCompletedAt, - taskOwnerPid: _sourceTaskOwnerPid, - taskPromptSummary: _sourceTaskPromptSummary, - taskPriority: _sourceTaskPriority, - taskKind: _sourceTaskKind, - taskDueAt: _sourceTaskDueAt, - taskDispatch: _sourceTaskDispatch, - taskModelId: _sourceTaskModelId, - taskRetriedFrom: _sourceTaskRetriedFrom, - taskDelivery: _sourceTaskDelivery, - taskIsolation: _sourceTaskIsolation, - taskSourceProjectPath: _sourceTaskSourceProjectPath, - taskWorktree: _sourceTaskWorktree, - taskDiffStat: _sourceTaskDiffStat, - taskQueuePosition: _sourceTaskQueuePosition, - taskQueueDepth: _sourceTaskQueueDepth, - taskConcurrencyLimit: _sourceTaskConcurrencyLimit, - pendingInteraction: _sourcePendingInteraction, - ...sourceCreatedData - } = sourceCreated.data; - const childCreated: Extract = { - id: nanoid(), - sessionId: targetSessionId, - timestamp: now, - type: 'session_created', - cwd: targetProjectPath, - gitBranch, - version, - data: { - ...sourceCreatedData, - sessionId: targetSessionId, - rootId, - parentId: sourceSessionId, - relationType: 'fork', - taskStatus: 'completed', - taskCompletedAt: now, - taskIsolation: 'local', - taskSourceProjectPath: targetProjectPath, - createdAt: now, - updatedAt: now, - }, - }; - const copiedEntries = sourceEntries - .filter( - (entry) => - entry.type !== 'session_created' && - entry.type !== 'token_budget_handoff_recorded' && - entry.type !== 'inbox_acknowledged' && - entry.type !== 'interaction_requested' && - entry.type !== 'interaction_responded' && - entry.type !== 'interaction_recovered' && - entry.type !== 'review_started' && - entry.type !== 'review_completed' - ) - .map((entry): SessionEvent => { - const base = { - ...entry, - id: nanoid(), - sessionId: targetSessionId, - cwd: targetProjectPath, - gitBranch, - version, - }; - if (entry.type === 'session_updated') { - const { - status: _status, - taskStatus: _taskStatus, - taskStatusReason: _taskStatusReason, - taskFailure: _taskFailure, - taskStartedAt: _taskStartedAt, - taskCompletedAt: _taskCompletedAt, - taskOwnerPid: _taskOwnerPid, - taskPromptSummary: _taskPromptSummary, - taskPriority: _taskPriority, - taskKind: _taskKind, - taskDueAt: _taskDueAt, - taskDispatch: _taskDispatch, - taskModelId: _taskModelId, - taskRetriedFrom: _taskRetriedFrom, - taskDelivery: _taskDelivery, - taskIsolation: _taskIsolation, - taskSourceProjectPath: _taskSourceProjectPath, - taskWorktree: _taskWorktree, - taskDiffStat: _taskDiffStat, - taskQueuePosition: _taskQueuePosition, - taskQueueDepth: _taskQueueDepth, - taskConcurrencyLimit: _taskConcurrencyLimit, - pendingInteraction: _pendingInteraction, - ...updatedData - } = entry.data; - return { - ...base, - type: 'session_updated', - data: { - ...updatedData, - sessionId: targetSessionId, - rootId, - parentId: sourceSessionId, - relationType: 'fork', - }, - }; - } - if (entry.type === 'message_created') { - const { inboxMessageId: _inboxMessageId, ...data } = entry.data; - return { - ...base, - type: 'message_created', - data, - }; - } - return base as SessionEvent; - }); - const forkBoundary: Extract = { - id: nanoid(), - sessionId: targetSessionId, - timestamp: now, - type: 'session_updated', - cwd: targetProjectPath, - gitBranch, + const childEntries = this.buildForkChildEntries({ + sourceEntries, + sourceCreated, + sourceSessionId, + targetSessionId, + targetProjectPath, + rootId, version, - data: { - sessionId: targetSessionId, - rootId, - parentId: sourceSessionId, - relationType: 'fork', - taskStatus: 'completed', - taskStatusReason: null, - taskFailure: null, - taskStartedAt: null, - taskCompletedAt: now, - taskOwnerPid: null, - taskIsolation: 'local', - taskSourceProjectPath: targetProjectPath, - taskWorktree: null, - taskDiffStat: null, - taskDelivery: null, - taskQueuePosition: null, - taskQueueDepth: null, - taskConcurrencyLimit: null, - updatedAt: now, - }, - }; - const childEntries: SessionEvent[] = [childCreated, ...copiedEntries, forkBoundary]; + now, + target: { kind: 'local', gitBranch }, + }); const targetFilePath = getSessionFilePath(targetProjectPath, targetSessionId); let targetCreated = false; @@ -2132,7 +2045,10 @@ export class SessionService { rootId, version, now, - remoteDescriptor: sourceMetadata.remoteWorkspace, + target: { + kind: 'remote', + descriptor: sourceMetadata.remoteWorkspace, + }, }); const targetFilePath = getAcpRemoteSessionFilePath(scope, targetSessionId); let targetCreated = false; @@ -2263,7 +2179,9 @@ export class SessionService { rootId: string; version: string; now: string; - remoteDescriptor: AcpRemoteWorkspaceDescriptorV1; + target: + | { kind: 'local'; gitBranch?: string } + | { kind: 'remote'; descriptor: AcpRemoteWorkspaceDescriptorV1 }; }): SessionEvent[] { const { sourceEntries, @@ -2274,8 +2192,28 @@ export class SessionService { rootId, version, now, - remoteDescriptor, + target, } = options; + const eventTarget = + target.kind === 'remote' + ? { projectPath: targetProjectPath } + : { gitBranch: target.gitBranch }; + const createdTarget = + target.kind === 'remote' + ? { remoteWorkspace: target.descriptor } + : { + taskIsolation: 'local' as const, + taskSourceProjectPath: targetProjectPath, + }; + const boundaryTarget = + target.kind === 'remote' + ? {} + : { + taskIsolation: 'local' as const, + taskSourceProjectPath: targetProjectPath, + taskWorktree: null, + taskDiffStat: null, + }; const { status: _sourceStatus, taskStatus: _sourceTaskStatus, @@ -2306,7 +2244,7 @@ export class SessionService { const childCreated: Extract = { id: nanoid(), sessionId: targetSessionId, - projectPath: targetProjectPath, + ...eventTarget, timestamp: now, type: 'session_created', cwd: targetProjectPath, @@ -2319,7 +2257,7 @@ export class SessionService { relationType: 'fork', taskStatus: 'completed', taskCompletedAt: now, - remoteWorkspace: remoteDescriptor, + ...createdTarget, createdAt: now, updatedAt: now, }, @@ -2339,10 +2277,10 @@ export class SessionService { .map((entry): SessionEvent => { const { gitBranch: _sourceGitBranch, ...entryWithoutGitBranch } = entry; const base = { - ...entryWithoutGitBranch, + ...(target.kind === 'remote' ? entryWithoutGitBranch : entry), id: nanoid(), sessionId: targetSessionId, - projectPath: targetProjectPath, + ...eventTarget, cwd: targetProjectPath, version, }; @@ -2395,7 +2333,7 @@ export class SessionService { const forkBoundary: Extract = { id: nanoid(), sessionId: targetSessionId, - projectPath: targetProjectPath, + ...eventTarget, timestamp: now, type: 'session_updated', cwd: targetProjectPath, @@ -2411,6 +2349,7 @@ export class SessionService { taskStartedAt: null, taskCompletedAt: now, taskOwnerPid: null, + ...boundaryTarget, taskDelivery: null, taskQueuePosition: null, taskQueueDepth: null, @@ -2523,6 +2462,19 @@ export class SessionService { update: SessionMetadataUpdate, sessionId: string ): void { + if ( + update.taskStatus !== undefined && + !SESSION_TASK_STATUSES.has(update.taskStatus) + ) { + throw new Error(`Invalid session task status: ${String(update.taskStatus)}`); + } + if ( + update.taskOwnerPid !== undefined && + update.taskOwnerPid !== null && + (!Number.isInteger(update.taskOwnerPid) || update.taskOwnerPid <= 0) + ) { + throw new Error('Session task owner PID must be a positive integer'); + } if ( update.taskPromptSummary !== undefined && update.taskPromptSummary !== null && @@ -2757,62 +2709,10 @@ export class SessionService { gitBranch: detectGitBranch(resolvedProjectPath), version: getVersion(), data: { + ...buildSessionUpdatedData(sessionId, initial, now, true), sessionId, rootId: sessionId, - ...(initial.title !== undefined ? { title: initial.title } : {}), taskStatus: initial.taskStatus ?? 'queued', - ...(initial.taskPromptSummary !== undefined - ? { taskPromptSummary: initial.taskPromptSummary } - : {}), - ...(initial.taskPriority !== undefined - ? { taskPriority: initial.taskPriority } - : {}), - ...(initial.taskKind !== undefined ? { taskKind: initial.taskKind } : {}), - ...(typeof initial.taskDueAt === 'string' - ? { taskDueAt: new Date(initial.taskDueAt).toISOString() } - : {}), - ...(initial.taskDispatch !== undefined - ? { taskDispatch: initial.taskDispatch } - : {}), - ...(initial.taskModelId !== undefined - ? { taskModelId: initial.taskModelId } - : {}), - ...(initial.taskRetriedFrom !== undefined - ? { taskRetriedFrom: initial.taskRetriedFrom } - : {}), - ...(initial.taskIsolation !== undefined - ? { taskIsolation: initial.taskIsolation } - : {}), - ...(initial.taskSourceProjectPath !== undefined - ? { taskSourceProjectPath: initial.taskSourceProjectPath } - : {}), - ...(initial.taskWorktree !== undefined - ? { taskWorktree: initial.taskWorktree } - : {}), - ...(initial.selectedModelId !== undefined - ? { selectedModelId: initial.selectedModelId } - : {}), - ...(initial.permissionMode !== undefined - ? { permissionMode: initial.permissionMode } - : {}), - ...(initial.reasoningEffort !== undefined - ? { reasoningEffort: initial.reasoningEffort } - : {}), - ...(initial.serviceTier !== undefined - ? { serviceTier: initial.serviceTier } - : {}), - ...(initial.responseVerbosity !== undefined - ? { responseVerbosity: initial.responseVerbosity } - : {}), - ...(initial.communicationStyle !== undefined - ? { communicationStyle: initial.communicationStyle } - : {}), - ...(initial.communicationStyleDigest !== undefined - ? { communicationStyleDigest: initial.communicationStyleDigest } - : {}), - ...(initial.projectInstructionsDigest !== undefined - ? { projectInstructionsDigest: initial.projectInstructionsDigest } - : {}), createdAt: now, updatedAt: now, }, @@ -2881,48 +2781,11 @@ export class SessionService { cwd: hostStateRoot, version: getVersion(), data: { + ...buildSessionUpdatedData(sessionId, initial, now, false), sessionId, rootId: sessionId, remoteWorkspace: validatedDescriptor, - ...(initial.title !== undefined ? { title: initial.title } : {}), taskStatus: initial.taskStatus ?? 'queued', - ...(initial.taskPromptSummary !== undefined - ? { taskPromptSummary: initial.taskPromptSummary } - : {}), - ...(initial.taskPriority !== undefined - ? { taskPriority: initial.taskPriority } - : {}), - ...(initial.taskKind !== undefined ? { taskKind: initial.taskKind } : {}), - ...(typeof initial.taskDueAt === 'string' - ? { taskDueAt: new Date(initial.taskDueAt).toISOString() } - : {}), - ...(initial.taskModelId !== undefined - ? { taskModelId: initial.taskModelId } - : {}), - ...(initial.selectedModelId !== undefined - ? { selectedModelId: initial.selectedModelId } - : {}), - ...(initial.permissionMode !== undefined - ? { permissionMode: initial.permissionMode } - : {}), - ...(initial.reasoningEffort !== undefined - ? { reasoningEffort: initial.reasoningEffort } - : {}), - ...(initial.serviceTier !== undefined - ? { serviceTier: initial.serviceTier } - : {}), - ...(initial.responseVerbosity !== undefined - ? { responseVerbosity: initial.responseVerbosity } - : {}), - ...(initial.communicationStyle !== undefined - ? { communicationStyle: initial.communicationStyle } - : {}), - ...(initial.communicationStyleDigest !== undefined - ? { communicationStyleDigest: initial.communicationStyleDigest } - : {}), - ...(initial.projectInstructionsDigest !== undefined - ? { projectInstructionsDigest: initial.projectInstructionsDigest } - : {}), createdAt: now, updatedAt: now, }, @@ -3238,19 +3101,6 @@ export class SessionService { ): Promise { assertValidSessionId(sessionId); SessionService.validateTaskMetadataUpdate(update, sessionId); - if ( - update.taskStatus !== undefined && - !SESSION_TASK_STATUSES.has(update.taskStatus) - ) { - throw new Error(`Invalid session task status: ${String(update.taskStatus)}`); - } - if ( - update.taskOwnerPid !== undefined && - update.taskOwnerPid !== null && - (!Number.isInteger(update.taskOwnerPid) || update.taskOwnerPid <= 0) - ) { - throw new Error('Session task owner PID must be a positive integer'); - } const resolvedProjectPath = SessionService.resolveCatalogWorkspace(projectPath); if ( update.taskWorktree && @@ -3305,101 +3155,7 @@ export class SessionService { cwd: resolvedProjectPath, gitBranch: detectGitBranch(resolvedProjectPath), version: getVersion(), - data: { - sessionId, - ...(update.title !== undefined ? { title: update.title } : {}), - ...(update.taskStatus !== undefined - ? { taskStatus: update.taskStatus } - : {}), - ...(update.taskStatusReason !== undefined - ? { taskStatusReason: update.taskStatusReason } - : {}), - ...(update.taskFailure !== undefined - ? { taskFailure: update.taskFailure } - : {}), - ...(update.taskStartedAt !== undefined - ? { taskStartedAt: update.taskStartedAt } - : {}), - ...(update.taskCompletedAt !== undefined - ? { taskCompletedAt: update.taskCompletedAt } - : {}), - ...(update.taskOwnerPid !== undefined - ? { taskOwnerPid: update.taskOwnerPid } - : {}), - ...(update.taskPromptSummary !== undefined - ? { taskPromptSummary: update.taskPromptSummary } - : {}), - ...(update.taskPriority !== undefined - ? { taskPriority: update.taskPriority } - : {}), - ...(update.taskKind !== undefined ? { taskKind: update.taskKind } : {}), - ...(update.taskDueAt !== undefined - ? { - taskDueAt: - update.taskDueAt === null - ? null - : new Date(update.taskDueAt).toISOString(), - } - : {}), - ...(update.taskDispatch !== undefined - ? { taskDispatch: update.taskDispatch } - : {}), - ...(update.taskModelId !== undefined - ? { taskModelId: update.taskModelId } - : {}), - ...(update.taskRetriedFrom !== undefined - ? { taskRetriedFrom: update.taskRetriedFrom } - : {}), - ...(update.taskDelivery !== undefined - ? { taskDelivery: update.taskDelivery } - : {}), - ...(update.taskIsolation !== undefined - ? { taskIsolation: update.taskIsolation } - : {}), - ...(update.taskSourceProjectPath !== undefined - ? { taskSourceProjectPath: update.taskSourceProjectPath } - : {}), - ...(update.taskWorktree !== undefined - ? { taskWorktree: update.taskWorktree } - : {}), - ...(update.taskDiffStat !== undefined - ? { taskDiffStat: update.taskDiffStat } - : {}), - ...(update.taskQueuePosition !== undefined - ? { taskQueuePosition: update.taskQueuePosition } - : {}), - ...(update.taskQueueDepth !== undefined - ? { taskQueueDepth: update.taskQueueDepth } - : {}), - ...(update.taskConcurrencyLimit !== undefined - ? { taskConcurrencyLimit: update.taskConcurrencyLimit } - : {}), - ...(update.selectedModelId !== undefined - ? { selectedModelId: update.selectedModelId } - : {}), - ...(update.permissionMode !== undefined - ? { permissionMode: update.permissionMode } - : {}), - ...(update.reasoningEffort !== undefined - ? { reasoningEffort: update.reasoningEffort } - : {}), - ...(update.serviceTier !== undefined - ? { serviceTier: update.serviceTier } - : {}), - ...(update.responseVerbosity !== undefined - ? { responseVerbosity: update.responseVerbosity } - : {}), - ...(update.communicationStyle !== undefined - ? { communicationStyle: update.communicationStyle } - : {}), - ...(update.communicationStyleDigest !== undefined - ? { communicationStyleDigest: update.communicationStyleDigest } - : {}), - ...(update.projectInstructionsDigest !== undefined - ? { projectInstructionsDigest: update.projectInstructionsDigest } - : {}), - updatedAt: now, - }, + data: buildSessionUpdatedData(sessionId, update, now, true), }; persistedEntries = [...entries, next]; return next; @@ -3426,19 +3182,6 @@ export class SessionService { ): Promise { assertValidSessionId(sessionId); SessionService.validateTaskMetadataUpdate(update, sessionId); - if ( - update.taskStatus !== undefined && - !SESSION_TASK_STATUSES.has(update.taskStatus) - ) { - throw new Error(`Invalid session task status: ${String(update.taskStatus)}`); - } - if ( - update.taskOwnerPid !== undefined && - update.taskOwnerPid !== null && - (!Number.isInteger(update.taskOwnerPid) || update.taskOwnerPid <= 0) - ) { - throw new Error('Session task owner PID must be a positive integer'); - } let descriptor: AcpRemoteWorkspaceDescriptorV1; try { @@ -3491,80 +3234,7 @@ export class SessionService { type: 'session_updated', cwd: hostStateRoot, version: getVersion(), - data: { - sessionId, - ...(update.title !== undefined ? { title: update.title } : {}), - ...(update.taskStatus !== undefined - ? { taskStatus: update.taskStatus } - : {}), - ...(update.taskStatusReason !== undefined - ? { taskStatusReason: update.taskStatusReason } - : {}), - ...(update.taskFailure !== undefined - ? { taskFailure: update.taskFailure } - : {}), - ...(update.taskStartedAt !== undefined - ? { taskStartedAt: update.taskStartedAt } - : {}), - ...(update.taskCompletedAt !== undefined - ? { taskCompletedAt: update.taskCompletedAt } - : {}), - ...(update.taskOwnerPid !== undefined - ? { taskOwnerPid: update.taskOwnerPid } - : {}), - ...(update.taskPromptSummary !== undefined - ? { taskPromptSummary: update.taskPromptSummary } - : {}), - ...(update.taskPriority !== undefined - ? { taskPriority: update.taskPriority } - : {}), - ...(update.taskKind !== undefined ? { taskKind: update.taskKind } : {}), - ...(update.taskDueAt !== undefined - ? { - taskDueAt: - update.taskDueAt === null - ? null - : new Date(update.taskDueAt).toISOString(), - } - : {}), - ...(update.taskModelId !== undefined - ? { taskModelId: update.taskModelId } - : {}), - ...(update.taskQueuePosition !== undefined - ? { taskQueuePosition: update.taskQueuePosition } - : {}), - ...(update.taskQueueDepth !== undefined - ? { taskQueueDepth: update.taskQueueDepth } - : {}), - ...(update.taskConcurrencyLimit !== undefined - ? { taskConcurrencyLimit: update.taskConcurrencyLimit } - : {}), - ...(update.selectedModelId !== undefined - ? { selectedModelId: update.selectedModelId } - : {}), - ...(update.permissionMode !== undefined - ? { permissionMode: update.permissionMode } - : {}), - ...(update.reasoningEffort !== undefined - ? { reasoningEffort: update.reasoningEffort } - : {}), - ...(update.serviceTier !== undefined - ? { serviceTier: update.serviceTier } - : {}), - ...(update.responseVerbosity !== undefined - ? { responseVerbosity: update.responseVerbosity } - : {}), - ...(update.communicationStyle !== undefined - ? { communicationStyle: update.communicationStyle } - : {}), - ...(update.communicationStyleDigest !== undefined - ? { communicationStyleDigest: update.communicationStyleDigest } - : {}), - ...(update.projectInstructionsDigest !== undefined - ? { projectInstructionsDigest: update.projectInstructionsDigest } - : {}), - updatedAt: now, - }, + data: buildSessionUpdatedData(sessionId, update, now, false), }; persistedEntries = [...entries, next]; return next; @@ -3681,9 +3351,7 @@ export class SessionService { } } - /** - * 从 JSONL 文件加载并转换消息 - */ + /** 从 JSONL 文件加载并转换消息 */ private static async loadSessionFromFile( filePath: string, sessionId: string, @@ -3770,9 +3438,7 @@ export class SessionService { return [...replacementMessages, ...suffix]; } - /** - * 将 JSONL 条目转换为 OpenAI Message 格式 - */ + /** 将 JSONL 条目转换为 OpenAI Message 格式 */ static convertJSONLToMessages( entries: SessionEvent[], options: { includeTokenBudgetHandoffs?: boolean } = {} @@ -3979,9 +3645,7 @@ export class SessionService { return messages; } - /** - * 元数据聚合器(注入投影层,复用 projectMetadataFromEntries 保证与 JSONL 逐条一致)。 - */ + /** 元数据聚合器(注入投影层,复用 projectMetadataFromEntries 保证与 JSONL 逐条一致)。 */ private static projectionDeriver(): MetadataDeriver { return (entries, sessionId, projectPath, sourceKind, actualFilePath) => { try { @@ -4086,7 +3750,7 @@ export class SessionService { const projected = await this.scanStoredSessionsFromProjection( scopedProjectPath, includeSubagents, - 0, + projectionSyncMaxAgeMs, archived, taskFilters ); @@ -4204,13 +3868,9 @@ export class SessionService { } private static async scanRemoteStoredSessions( - options: NormalizedRemoteSessionListOptions, - projectionSyncMaxAgeMs = 0 + options: NormalizedRemoteSessionListOptions ): Promise { - const projected = await this.scanRemoteStoredSessionsFromProjection( - options, - projectionSyncMaxAgeMs - ); + const projected = await this.scanRemoteStoredSessionsFromProjection(options); if (projected) return projected; const scopes = await this.listRemoteSessionScopes(options); @@ -4264,8 +3924,7 @@ export class SessionService { } private static async scanRemoteStoredSessionsFromProjection( - options: NormalizedRemoteSessionListOptions, - projectionSyncMaxAgeMs = 0 + options: NormalizedRemoteSessionListOptions ): Promise { try { const db = await getProjectionDb(); @@ -4669,7 +4328,6 @@ export class SessionService { : path.resolve(committedProjectPath); const remoteWorkspace = this.parseRemoteWorkspaceFromCreated( created, - sessionId, resolvedProjectPath ); const parsedTaskWorktree = parseTaskWorktree(durable.taskWorktree); @@ -4851,7 +4509,6 @@ export class SessionService { private static parseRemoteWorkspaceFromCreated( created: Extract, - sessionId: string, resolvedProjectPath: string ): AcpRemoteWorkspaceDescriptorV1 | undefined { if (!Object.hasOwn(created.data, 'remoteWorkspace')) { @@ -4971,9 +4628,7 @@ export class SessionService { return publicSession; } - /** - * 获取会话文件路径 - */ + /** 获取会话文件路径 */ private static getSessionFilePath(projectPath: string, sessionId: string): string { return getSessionFilePath(projectPath, sessionId); } diff --git a/packages/cli/src/services/StructuredOutputService.ts b/packages/cli/src/services/StructuredOutputService.ts index 91f832557..f4cd408bc 100644 --- a/packages/cli/src/services/StructuredOutputService.ts +++ b/packages/cli/src/services/StructuredOutputService.ts @@ -1,9 +1,9 @@ import { createHash } from 'node:crypto'; import Ajv, { type ErrorObject, type ValidateFunction } from 'ajv'; import type { JSONSchema7 } from 'json-schema'; -import type { Message } from './ChatServiceInterface.js'; import type { JsonObject, JsonValue } from '../store/types.js'; import type { FunctionDeclaration } from '../tools/types/index.js'; +import type { Message } from './ChatServiceInterface.js'; export const STRUCTURED_OUTPUT_TOOL_NAME = 'StructuredOutput'; export const MAX_STRUCTURED_OUTPUT_SCHEMA_BYTES = 64 * 1024; @@ -185,15 +185,6 @@ export function createStructuredOutputContract( }; } -export function isStructuredOutputSchema(value: unknown): value is JsonObject { - try { - createStructuredOutputContract(value); - return true; - } catch { - return false; - } -} - function outputFromMetadata( value: unknown, contract: StructuredOutputContract diff --git a/packages/cli/src/services/TranscriptSearch.ts b/packages/cli/src/services/TranscriptSearch.ts index 6b4be10d8..523383c39 100644 --- a/packages/cli/src/services/TranscriptSearch.ts +++ b/packages/cli/src/services/TranscriptSearch.ts @@ -1,8 +1,6 @@ /** - * Transcript Search — 历史会话搜索 - * - * 在过去的 session JSONL 文件中搜索关键词,返回匹配的上下文片段。 - * 用于 /search 命令和 Agent 工具(回忆过去做过的事情)。 + * Transcript Search — 历史会话搜索

在过去的 session JSONL 文件中搜索关键词,返回匹配的上下文片段。 用于 /search 命令和 + * Agent 工具(回忆过去做过的事情)。 */ import * as fs from 'node:fs/promises'; diff --git a/packages/cli/src/services/VersionChecker.ts b/packages/cli/src/services/VersionChecker.ts index 6f0b66128..1d216fae1 100644 --- a/packages/cli/src/services/VersionChecker.ts +++ b/packages/cli/src/services/VersionChecker.ts @@ -1,8 +1,4 @@ -/** - * 版本检查服务 - * - * 启动时检查 npm registry 获取最新版本,提供交互式更新选项 - */ +/** 版本检查服务 启动时检查 npm registry 获取最新版本,提供交互式更新选项 */ import * as fs from 'fs/promises'; import * as path from 'path'; @@ -45,9 +41,7 @@ export interface VersionCheckResult { error?: string; } -/** - * 获取当前安装的版本 - */ +/** 获取当前安装的版本 */ async function getCurrentVersion(): Promise { try { if (process.env.BLADE_VERSION) { @@ -59,9 +53,7 @@ async function getCurrentVersion(): Promise { } } -/** - * 从缓存读取版本信息 - */ +/** 从缓存读取版本信息 */ async function readCache(): Promise { try { const content = await fs.readFile(CACHE_FILE, 'utf-8'); @@ -78,9 +70,7 @@ async function readCache(): Promise { } } -/** - * 写入缓存 - */ +/** 写入缓存 */ async function writeCache(cache: VersionCache): Promise { try { await fs.mkdir(CACHE_DIR, { recursive: true, mode: 0o755 }); @@ -90,9 +80,7 @@ async function writeCache(cache: VersionCache): Promise { } } -/** - * 从 npm registry 获取最新版本 - */ +/** 从 npm registry 获取最新版本 */ async function fetchLatestVersion(): Promise { try { return await proxyFetch( @@ -204,9 +192,7 @@ export async function checkVersion(forceCheck = false): Promise { const cache = await readCache(); await writeCache({ @@ -225,9 +211,7 @@ export function getUpgradeCommand(): string { return `npm install -g ${PACKAGE_NAME}@latest --registry https://registry.npmjs.org`; } -/** - * 执行升级(返回 Promise) - */ +/** 执行升级(返回 Promise) */ export async function performUpgrade(): Promise<{ success: boolean; message: string }> { const { spawn } = await import('child_process'); @@ -262,9 +246,7 @@ export async function performUpgrade(): Promise<{ success: boolean; message: str }); } -/** - * 启动时版本检查(简化版,仅返回检查结果) - */ +/** 启动时版本检查(简化版,仅返回检查结果) */ export async function checkVersionOnStartup(): Promise { try { const cache = await readCache(); diff --git a/packages/cli/src/services/modelAlias.ts b/packages/cli/src/services/modelAlias.ts index 50acafa7b..6ec293840 100644 --- a/packages/cli/src/services/modelAlias.ts +++ b/packages/cli/src/services/modelAlias.ts @@ -1,8 +1,6 @@ /** - * Model Alias Resolution - * - * Maps short/convenient names to full model IDs. - * Used by config loading, /model command, and BLADE_MODEL env var. + * Model Alias Resolution

Maps short/convenient names to full model IDs. Used by + * config loading, /model command, and BLADE_MODEL env var. */ const MODEL_ALIASES: Record = { @@ -43,17 +41,3 @@ export function resolveModelAlias(nameOrAlias: string): string { const lower = nameOrAlias.toLowerCase().trim(); return MODEL_ALIASES[lower] ?? nameOrAlias; } - -/** - * Check if a string is a known alias. - */ -export function isModelAlias(name: string): boolean { - return name.toLowerCase().trim() in MODEL_ALIASES; -} - -/** - * Get all available aliases for display. - */ -export function getModelAliases(): Array<{ alias: string; model: string }> { - return Object.entries(MODEL_ALIASES).map(([alias, model]) => ({ alias, model })); -} diff --git a/packages/cli/src/services/pi/PiModelCatalog.ts b/packages/cli/src/services/pi/PiModelCatalog.ts index 8e4d82c6d..268bd2c55 100644 --- a/packages/cli/src/services/pi/PiModelCatalog.ts +++ b/packages/cli/src/services/pi/PiModelCatalog.ts @@ -212,13 +212,6 @@ export class PiModelCatalog { this.installCustomProvider(providerId); } - unregisterModelProvider(providerId: string): void { - if (!this.customProviderConfigs.has(providerId)) return; - this.customProviderConfigs.delete(providerId); - this.customProviderModels.delete(providerId); - this.models.deleteProvider(providerId); - } - private requireProvider(providerId: string): void { if (!this.models.getProvider(providerId)) { throw new Error(`Unknown pi-ai provider: ${providerId}`); diff --git a/packages/cli/src/services/pi/providerCircuitBreaker.ts b/packages/cli/src/services/pi/providerCircuitBreaker.ts index 612bb4a65..c5f314e09 100644 --- a/packages/cli/src/services/pi/providerCircuitBreaker.ts +++ b/packages/cli/src/services/pi/providerCircuitBreaker.ts @@ -16,11 +16,9 @@ export { export const PROVIDER_CIRCUIT_WINDOW_MS = 60_000; export const PROVIDER_CIRCUIT_MIN_SAMPLES = 4; export const PROVIDER_CIRCUIT_ERROR_RATE_THRESHOLD = 0.8; -export const PROVIDER_CIRCUIT_HALF_OPEN_MAX_PROBES = 1; export const MIN_PROVIDER_CIRCUIT_PROBE_LEASE_MS = 300_000; export const MAX_PROVIDER_CIRCUIT_PROBE_LEASE_MS = 600_000; -export const PROVIDER_CIRCUIT_HEARTBEAT_MS = 15_000; export const MAX_PROVIDER_CIRCUIT_REGISTRY_ENTRIES = 128; export const MAX_PROVIDER_CIRCUIT_WINDOW_ENTRIES = 256; diff --git a/packages/cli/src/skills/SkillInstaller.ts b/packages/cli/src/skills/SkillInstaller.ts index 6636fe93f..6d501f143 100644 --- a/packages/cli/src/skills/SkillInstaller.ts +++ b/packages/cli/src/skills/SkillInstaller.ts @@ -57,17 +57,13 @@ export function skillNameFromRepositoryUrl(url: string): string { return path.posix.basename(pathname.replace(/\/+$/, '')).replace(/\.git$/, ''); } -/** - * 官方 Skills 仓库信息 - */ +/** 官方 Skills 仓库信息 */ const OFFICIAL_SKILLS_REPO = { url: 'https://github.com/anthropics/skills.git', branch: 'main', }; -/** - * SkillInstaller 类 - */ +/** SkillInstaller 类 */ export class SkillInstaller { private skillsDir: string; @@ -75,9 +71,7 @@ export class SkillInstaller { this.skillsDir = skillsDir || path.join(homedir(), '.blade', 'skills'); } - /** - * 检查 git 是否可用 - */ + /** 检查 git 是否可用 */ private async isGitAvailable(): Promise { try { await this.runGit(['--version'], 5000); @@ -114,8 +108,7 @@ export class SkillInstaller { // 确保目录存在 await fs.mkdir(this.skillsDir, { recursive: true, mode: 0o755 }); - // 使用 git clone --depth 1 --filter 克隆指定目录 - // 方法:克隆整个仓库(浅克隆),然后只复制需要的目录 + // 使用 git clone --depth 1 --filter 克隆指定目录 方法:克隆整个仓库(浅克隆),然后只复制需要的目录 await this.runGit( [ 'clone', @@ -317,9 +310,7 @@ export class SkillInstaller { } } - /** - * 安装所有官方 Skills - */ + /** 安装所有官方 Skills */ async installAllOfficialSkills(): Promise<{ installed: string[]; failed: string[] }> { const { url, branch } = OFFICIAL_SKILLS_REPO; const tempDir = path.join(this.skillsDir, `.tmp-all-${Date.now()}`); @@ -402,9 +393,7 @@ export class SkillInstaller { } } -/** - * 获取 SkillInstaller 单例 - */ +/** 获取 SkillInstaller 单例 */ let installerInstance: SkillInstaller | null = null; export function getSkillInstaller(skillsDir?: string): SkillInstaller { diff --git a/packages/cli/src/skills/SkillLoader.ts b/packages/cli/src/skills/SkillLoader.ts index 9521fbe17..8a2d7bcb7 100644 --- a/packages/cli/src/skills/SkillLoader.ts +++ b/packages/cli/src/skills/SkillLoader.ts @@ -1,8 +1,6 @@ /** - * SkillLoader - SKILL.md 文件解析器 - * - * 负责解析 SKILL.md 文件的 YAML 前置数据和 Markdown 正文内容。 - * 支持 Progressive Disclosure:可以只加载元数据,或加载完整内容。 + * SkillLoader - SKILL.md 文件解析器

负责解析 SKILL.md 文件的 YAML 前置数据和 Markdown 正文内容。 支持 + * Progressive Disclosure:可以只加载元数据,或加载完整内容。 */ import * as fs from 'node:fs/promises'; @@ -19,10 +17,7 @@ const NAME_REGEX = /^[a-z0-9][a-z0-9-]{0,62}[a-z0-9]?$/; /** 描述最大长度 */ const MAX_DESCRIPTION_LENGTH = 1024; -/** - * 解析 SKILL.md 的 YAML 前置数据 - * 完全对齐 Claude Code Skills 规范 - */ +/** 解析 SKILL.md 的 YAML 前置数据 完全对齐 Claude Code Skills 规范 */ interface RawFrontmatter { name?: string; description?: string; @@ -40,9 +35,7 @@ interface RawFrontmatter { when_to_use?: string; } -/** - * 验证并规范化 allowed-tools 字段 - */ +/** 验证并规范化 allowed-tools 字段 */ function parseAllowedTools(raw: string | string[] | undefined): string[] | undefined { if (!raw) return undefined; @@ -61,9 +54,7 @@ function parseAllowedTools(raw: string | string[] | undefined): string[] | undef return undefined; } -/** - * 解析布尔值字段(支持 true/false 字符串) - */ +/** 解析布尔值字段(支持 true/false 字符串) */ function parseBoolean(value: boolean | string | undefined): boolean | undefined { if (value === undefined) return undefined; if (typeof value === 'boolean') return value; @@ -75,9 +66,7 @@ function parseBoolean(value: boolean | string | undefined): boolean | undefined return undefined; } -/** - * 验证 Skill 元数据 - */ +/** 验证 Skill 元数据 */ function validateMetadata( frontmatter: RawFrontmatter, filePath: string @@ -130,9 +119,7 @@ function validateMetadata( }; } -/** - * 解析 SKILL.md 文件内容 - */ +/** 解析 SKILL.md 文件内容 */ function parseSkillContent( content: string, filePath: string, @@ -185,9 +172,7 @@ function parseSkillContent( }; } -/** - * 从文件加载 Skill(仅元数据) - */ +/** 从文件加载 Skill(仅元数据) */ export async function loadSkillMetadata( filePath: string, source: 'user' | 'project' | 'builtin' @@ -209,9 +194,7 @@ export async function loadSkillMetadata( } } -/** - * 加载完整 Skill 内容 - */ +/** 加载完整 Skill 内容 */ export async function loadSkillContent( metadata: SkillMetadata ): Promise { @@ -224,9 +207,7 @@ export async function loadSkillContent( } } -/** - * 检查目录中是否存在 SKILL.md - */ +/** 检查目录中是否存在 SKILL.md */ export async function hasSkillFile(dirPath: string): Promise { try { await fs.access(path.join(dirPath, 'SKILL.md')); diff --git a/packages/cli/src/skills/SkillRegistry.ts b/packages/cli/src/skills/SkillRegistry.ts index d47b0e36a..ec91a5b99 100644 --- a/packages/cli/src/skills/SkillRegistry.ts +++ b/packages/cli/src/skills/SkillRegistry.ts @@ -1,8 +1,6 @@ /** - * SkillRegistry - Skill 注册表 - * - * 负责发现、加载、管理所有可用的 Skills。 - * 使用 Progressive Disclosure:启动时仅加载元数据,执行时才加载完整内容。 + * SkillRegistry - Skill 注册表

负责发现、加载、管理所有可用的 Skills。 使用 Progressive + * Disclosure:启动时仅加载元数据,执行时才加载完整内容。 */ import * as fs from 'node:fs/promises'; @@ -27,11 +25,7 @@ import type { SkillRegistryConfig, } from './types.js'; -/** - * 默认配置 - * 注意:cwd 不在这里求值,而是在构造函数中延迟求值, - * 因为此常量在模块加载阶段就会被执行,那时 setCwd() 可能尚未调用。 - */ +/** 默认配置 注意:cwd 不在这里求值,而是在构造函数中延迟求值, 因为此常量在模块加载阶段就会被执行,那时 setCwd() 可能尚未调用。 */ const DEFAULT_CONFIG_BASE = { userSkillsDir: path.join(homedir(), '.blade', 'skills'), projectSkillsDir: '.blade/skills', @@ -40,18 +34,14 @@ const DEFAULT_CONFIG_BASE = { claudeProjectSkillsDir: '.claude/skills', }; -/** - * SkillRegistry 单例 - */ +/** SkillRegistry 单例 */ const instances = new Map(); function registryKey(config?: SkillRegistryConfig): string { return path.resolve(config?.cwd ?? getCwd()); } -/** - * Skill 注册表 - */ +/** Skill 注册表 */ export class SkillRegistry { private skills: Map = new Map(); /** Plugin skills stored with namespaced names */ @@ -63,9 +53,7 @@ export class SkillRegistry { this.config = { ...DEFAULT_CONFIG_BASE, cwd: getCwd(), ...config }; } - /** - * 获取单例实例 - */ + /** 获取单例实例 */ static getInstance(config?: SkillRegistryConfig): SkillRegistry { const key = registryKey(config); let instance = instances.get(key); @@ -79,9 +67,7 @@ export class SkillRegistry { return instance; } - /** - * 重置单例(用于测试) - */ + /** 重置单例(用于测试) */ static resetInstance(): void { instances.clear(); } @@ -167,9 +153,7 @@ export class SkillRegistry { }; } - /** - * 加载内置 Skills - */ + /** 加载内置 Skills */ private loadBuiltinSkills(): void { // 注册 skill-creator this.skills.set(skillCreatorMetadata.name, skillCreatorMetadata); @@ -177,9 +161,7 @@ export class SkillRegistry { this.skills.set(updateConfigMetadata.name, updateConfigMetadata); } - /** - * 扫描指定目录下的所有 skills - */ + /** 扫描指定目录下的所有 skills */ private async scanDirectory( dirPath: string, source: 'user' | 'project' @@ -228,30 +210,22 @@ export class SkillRegistry { return { skills, errors }; } - /** - * 获取所有已注册的 skills 元数据 - */ + /** 获取所有已注册的 skills 元数据 */ getAll(): SkillMetadata[] { return Array.from(this.skills.values()); } - /** - * 根据名称获取 skill 元数据 - */ + /** 根据名称获取 skill 元数据 */ get(name: string): SkillMetadata | undefined { return this.skills.get(name); } - /** - * 检查 skill 是否存在 - */ + /** 检查 skill 是否存在 */ has(name: string): boolean { return this.skills.has(name); } - /** - * 加载 skill 的完整内容(懒加载) - */ + /** 加载 skill 的完整内容(懒加载) */ async loadContent(name: string): Promise { const metadata = this.skills.get(name); if (!metadata) return null; @@ -265,9 +239,7 @@ export class SkillRegistry { return loadSkillContent(metadata); } - /** - * 加载内置 Skill 的完整内容 - */ + /** 加载内置 Skill 的完整内容 */ private loadBuiltinContent(name: string): SkillContent | null { switch (name) { case 'skill-creator': @@ -331,16 +303,12 @@ export class SkillRegistry { return lines.join('\n'); } - /** - * 获取 skills 数量 - */ + /** 获取 skills 数量 */ get size(): number { return this.skills.size; } - /** - * 重新扫描并刷新注册表 - */ + /** 重新扫描并刷新注册表 */ async refresh(): Promise { this.skills.clear(); this.pluginSkills.clear(); @@ -401,48 +369,7 @@ export class SkillRegistry { this.skills.set(skill.namespacedName, skill.metadata); } - /** - * 查找插件技能 - * - * Supports both: - * - Full namespaced name: "plugin:skill" - * - Short name if unique: "skill" - * - * @param name - Skill name to find - * @returns Plugin skill or undefined - */ - findPluginSkill(name: string): PluginSkill | undefined { - // Try exact namespaced match first - const exact = this.pluginSkills.get(name); - if (exact) return exact; - - // Try short name match (if unique) - const matches: PluginSkill[] = []; - for (const skill of this.pluginSkills.values()) { - if (skill.originalName === name) { - matches.push(skill); - } - } - - // Only return if exactly one match - if (matches.length === 1) { - return matches[0]; - } - - return undefined; - } - - /** - * 获取所有插件技能 - */ - getAllPluginSkills(): PluginSkill[] { - return Array.from(this.pluginSkills.values()); - } - - /** - * 清除所有插件技能 - * Called when refreshing plugins - */ + /** 清除所有插件技能 Called when refreshing plugins */ clearPluginSkills(): void { // Remove from main skills map for (const skill of this.pluginSkills.values()) { @@ -450,28 +377,9 @@ export class SkillRegistry { } this.pluginSkills.clear(); } - - /** - * 获取插件技能数量 - */ - getPluginSkillCount(): number { - return this.pluginSkills.size; - } } -/** - * 获取 SkillRegistry 单例 - */ +/** 获取 SkillRegistry 单例 */ export function getSkillRegistry(config?: SkillRegistryConfig): SkillRegistry { return SkillRegistry.getInstance(config); } - -/** - * 初始化并获取所有 skills - */ -export async function discoverSkills( - config?: SkillRegistryConfig -): Promise { - const registry = getSkillRegistry(config); - return registry.initialize(); -} diff --git a/packages/cli/src/skills/builtin/skill-creator.md b/packages/cli/src/skills/builtin/skill-creator.md new file mode 100644 index 000000000..41f1e6ead --- /dev/null +++ b/packages/cli/src/skills/builtin/skill-creator.md @@ -0,0 +1,183 @@ +# Skill Creator + +帮助用户创建新的 Blade Skills。 + +## Instructions + +当用户想要创建新 Skill 时,按以下步骤进行: + +### 1. 了解需求 + +询问用户: +- Skill 的目的是什么?解决什么问题? +- 什么场景下应该使用这个 Skill? +- 需要访问哪些工具?(Read, Write, Bash, Grep, Glob, Task, WebFetch 等) +- 是否需要支持用户通过 `/skill-name` 命令调用? + +### 2. 设计 Skill + +根据用户需求,设计以下内容: + +**必填字段:** +- `name`: kebab-case 格式,≤64 字符(如 `code-review`, `commit-helper`) +- `description`: 简洁描述,≤1024 字符,包含"什么"和"何时使用" + +**可选字段:** +- `allowed-tools`: 限制可用工具列表(提高安全性) +- `argument-hint`: 参数提示(如 ``) +- `user-invocable`: 是否支持 `/skill-name` 调用(默认 false) +- `disable-model-invocation`: 是否禁止 AI 自动调用(默认 false) +- `version`: 版本号 + +**系统提示词设计要点:** +- 清晰的任务描述 +- 具体的执行步骤 +- 边界条件和错误处理 +- 输出格式规范 + +### 3. 确认保存位置 + +询问用户希望将 Skill 保存到: +- **项目级** (`.blade/skills/`): 与团队共享,通过 git 同步(**默认推荐**) +- **用户级** (`~/.blade/skills/`): 个人使用,跨项目可用 + +如果用户没有明确指定,默认使用 **项目级** (`.blade/skills/`)。 + +### 4. 生成文件 + +**重要:直接使用 Write 工具创建文件,不要使用外部脚本。** + +使用 Write 工具创建 SKILL.md 文件: + +``` +.blade/skills/{name}/SKILL.md # 项目级(默认) +~/.blade/skills/{name}/SKILL.md # 用户级 +``` + +文件格式: +```yaml +--- +name: {name} +description: {description} +allowed-tools: + - {tool1} + - {tool2} +user-invocable: true # 如果需要 /skill-name 命令 +--- + +# {Skill Title} + +## Instructions + +{详细指令} + +## Examples + +{使用示例} +``` + +### 5. 刷新并验证 + +创建完成后,**必须提示用户执行 `/skills` 命令刷新 Skills 列表**,否则新创建的 Skill 不会立即生效。 + +验证步骤: +- 检查目录和文件是否创建成功 +- **重要**:告诉用户执行 `/skills` 刷新列表 +- 提示用户可以通过以下方式使用新 Skill: + - AI 自动调用(如果未禁用) + - `/skill-name` 命令(如果启用了 user-invocable) + +## Best Practices + +1. **名称规范** + - 使用 kebab-case:`code-review`, `commit-helper`, `test-generator` + - 名称应该简洁且描述性强 + +2. **描述要具体** + - 包含触发词,帮助 AI 识别何时使用 + - 例如:"Generate commit messages following conventional commits format. Use when the user wants to commit changes or asks for a commit message." + +3. **限制工具访问** + - 仅授予必要的工具权限,提高安全性 + - 例如:只读 Skill 只需 Read, Grep, Glob + +4. **提供清晰的指令** + - 使用 Markdown 格式组织内容 + - 包含具体步骤和示例 + - 处理边界情况 + +5. **考虑调用方式** + - 频繁使用的 Skill 可设置 `user-invocable: true` + - 仅限用户手动触发的 Skill 可设置 `disable-model-invocation: true` + +## Example Skills + +### 代码审查 Skill + +```yaml +--- +name: code-review +description: Review code for best practices, bugs, and improvements. Use when reviewing PRs, checking code quality, or before committing. +allowed-tools: + - Read + - Grep + - Glob +argument-hint: +user-invocable: true +--- + +# Code Review + +审查代码质量、最佳实践和潜在问题。 + +## Instructions + +1. 读取指定的文件或目录 +2. 检查以下方面: + - 代码风格一致性 + - 潜在的 bug 或错误 + - 性能问题 + - 安全漏洞 + - 可读性和可维护性 +3. 提供具体的改进建议 + +## Output Format + +- 问题严重程度:[CRITICAL] 严重 | [WARN] 警告 | [INFO] 建议 +- 具体位置和代码片段 +- 改进建议和示例代码 +``` + +### 提交消息生成 Skill + +```yaml +--- +name: commit-message +description: Generate conventional commit messages. Use when committing changes or when user asks for a commit message. +allowed-tools: + - Bash + - Read +user-invocable: true +--- + +# Commit Message Generator + +生成符合 Conventional Commits 规范的提交消息。 + +## Instructions + +1. 运行 `git diff --staged` 查看暂存的更改 +2. 分析更改类型(feat/fix/docs/style/refactor/test/chore) +3. 生成简洁的提交消息 +4. 询问用户是否满意或需要调整 + +## Output Format + +``` +(): + + + +