From a3c71b13a05ca62a5572b1c4b7de71e0a83cb6fe Mon Sep 17 00:00:00 2001 From: Hoang Nguyen Date: Tue, 22 Sep 2026 20:03:38 +0200 Subject: [PATCH 1/3] feat(cli): add Jev session compaction --- .../2026-09-22-feature-session-compact-jev.md | 142 +++++++++++++ .../2026-09-22-feature-session-compact-jev.md | 87 ++++++++ .../2026-09-22-feature-session-compact-jev.md | 112 ++++++++++ .../2026-09-22-feature-session-compact-jev.md | 73 +++++++ .../2026-09-22-feature-session-compact-jev.md | 103 ++++++++++ package-lock.json | 10 + packages/cli/package.json | 1 + .../cli/src/__tests__/commands/agent.test.ts | 191 ++++++++++++++++++ .../session-compact/jev-classifier.test.ts | 104 ++++++++++ .../session-compact.service.test.ts | 169 ++++++++++++++++ .../services/skill/skill-builtins.test.ts | 12 +- packages/cli/src/commands/agent.ts | 71 +++++++ .../session-compact/jev-classifier.ts | 99 +++++++++ .../session-compact.service.ts | 140 +++++++++++++ .../session-compact/session-compact.types.ts | 56 +++++ .../cli/src/services/skill/skill-builtins.ts | 1 + skills/built-in.json | 3 +- skills/session-compact/SKILL.md | 51 +++++ 18 files changed, 1419 insertions(+), 6 deletions(-) create mode 100644 docs/ai/design/2026-09-22-feature-session-compact-jev.md create mode 100644 docs/ai/implementation/2026-09-22-feature-session-compact-jev.md create mode 100644 docs/ai/planning/2026-09-22-feature-session-compact-jev.md create mode 100644 docs/ai/requirements/2026-09-22-feature-session-compact-jev.md create mode 100644 docs/ai/testing/2026-09-22-feature-session-compact-jev.md create mode 100644 packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts create mode 100644 packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts create mode 100644 packages/cli/src/services/session-compact/jev-classifier.ts create mode 100644 packages/cli/src/services/session-compact/session-compact.service.ts create mode 100644 packages/cli/src/services/session-compact/session-compact.types.ts create mode 100644 skills/session-compact/SKILL.md diff --git a/docs/ai/design/2026-09-22-feature-session-compact-jev.md b/docs/ai/design/2026-09-22-feature-session-compact-jev.md new file mode 100644 index 00000000..631c4ef5 --- /dev/null +++ b/docs/ai/design/2026-09-22-feature-session-compact-jev.md @@ -0,0 +1,142 @@ +--- +phase: design +title: Jev Session Compaction Design +description: Architecture for typed session-event classification and deterministic compact artifacts +--- + +# Jev Session Compaction Design + +## Architecture Overview + +The feature extends the existing `agent session` command group. The command owns session selection and output. A small session-compaction service owns availability, redaction, Jev classification, deterministic assembly, and rendering. Provider-specific transcript parsing stays in `agent-manager`. + +```mermaid +flowchart TD + CLI[agent session compact] --> Gate{TYPESAFE_API_KEY present?} + Gate -->|no| Unavailable[Jev unavailable result] + Gate -->|yes| Sessions[AgentManager.listSessions] + Sessions --> Resolve[Resolve ID and optional type] + Resolve --> Adapter[Adapter.getConversation verbose] + Adapter --> Redact[Local credential redaction] + Redact --> Jev[Jev classifier] + Jev --> Filter[Exclude discard, irrelevant, sensitive] + Filter --> Builder[Deterministic compact builder] + Builder --> Markdown[Markdown renderer] + Builder --> JSON[JSON serializer] +``` + +## Data Models + +```ts +type CompactCategory = + | "user_instruction" + | "decision" + | "code_change" + | "command_evidence" + | "validation_evidence" + | "blocker" + | "next_step" + | "memory_candidate" + | "discard"; + +type CompactImportance = "irrelevant" | "useful" | "important" | "critical"; + +interface ClassifiedSessionEvent { + role: "user" | "assistant" | "system"; + content: string; + timestamp?: string; + keep: boolean; + category: CompactCategory; + importance: CompactImportance; + sensitive: boolean; +} + +interface SessionCompact { + intent: string; + currentState: string; + decisions: string[]; + changedFiles: string[]; + commands: string[]; + validation: string[]; + openQuestions: string[]; + nextStep: string; + memoryCandidates: string[]; + resumePrompt: string; + jev: { available: true; model: string; classifiedEvents: number }; +} + +type JevUnavailable = { + jev: { available: false; reason: "TYPESAFE_API_KEY is not set" }; +}; +``` + +The builder preserves source wording rather than claiming generated facts. It uses the first retained user instruction as intent, the latest retained operational events as current state, the latest next-step event as next step, and category groups for array fields. Blockers populate open questions. The resume prompt is a fixed template composed from those fields. + +## API Design + +### CLI + +```text +ai-devkit agent session compact --id [--type ] [--format markdown|json] +``` + +- Default format: `markdown`. +- Unavailable Markdown is the single required sentence. +- Unavailable JSON is the `JevUnavailable` object. +- Availability is checked before `AgentManager.listSessions()` so no transcript is read without a usable configuration signal. + +### Internal boundaries + +```ts +interface SessionEventClassifier { + readonly model: string; + classify(message: ConversationMessage): Promise; +} + +async function compactSession( + messages: ConversationMessage[], + classifier: SessionEventClassifier, +): Promise; + +function redactSensitiveText(content: string): string; +function renderSessionCompactMarkdown(result: SessionCompact): string; +``` + +The default classifier wraps `@typesafe-ai/sdk` and asks one `choice` question for category, one `choice` for importance, and two `noul` questions for retention and sensitivity. It validates returned labels before constructing a classified event. + +## Component Breakdown + +- `packages/cli/src/services/session-compact/session-compact.types.ts`: stable feature types and enums. +- `packages/cli/src/services/session-compact/jev-classifier.ts`: SDK adapter and response validation. +- `packages/cli/src/services/session-compact/session-compact.service.ts`: redaction, filtering, deterministic assembly, and Markdown rendering. +- `packages/cli/src/commands/agent.ts`: Commander wiring, early availability gate, existing session resolution, output selection. +- `packages/cli/src/__tests__/services/session-compact/*`: classifier/service unit tests with no network. +- `packages/cli/src/__tests__/commands/agent.test.ts`: command-level unavailable and successful wiring tests. +- `skills/session-compact/SKILL.md` and `skills/built-in.json`: agent instructions and distribution manifest. + +## Design Decisions + +1. **Use `agent session compact`, not a new top-level `session`.** Historical discovery and detail already live under `agent`; extending that namespace minimizes concepts and shares resolution behavior. +2. **Select by session ID, not raw input path.** Adapters already own heterogeneous file/database formats and normalize them to `ConversationMessage[]`. +3. **No silent fallback.** Missing configuration is a typed, successful unavailable result; API/runtime failures are errors. +4. **Deterministic output after classification.** Jev is a decision model, not a prose generator. Keeping source content also makes the artifact auditable. +5. **One injectable classifier boundary.** Tests avoid network access and the SDK can be replaced without changing the builder or CLI. +6. **Sequential classification for MVP.** It is simple, deterministic, and avoids unverified API batch/concurrency limits. Measurement can justify later bounded concurrency. +7. **Redact before Jev and exclude Jev-sensitive events.** This reduces exposure and prevents sensitive events from entering the artifact, while acknowledging regex redaction is not comprehensive DLP. + +### Rejected alternatives + +- Top-level `session compact`: duplicates the existing `agent session` resource namespace. +- `--input ` first: leaks provider parsing into the command and cannot naturally represent database-backed sessions. +- Put compaction into `agent-manager`: provider-independent classification and presentation are CLI application concerns. +- Add a general AI-provider layer: no second current caller. +- Ask Jev to generate the artifact: unsupported by its typed-decision role. + +## Non-Functional Requirements + +- Do not log, serialize, interpolate into errors, or transmit `TYPESAFE_API_KEY` as state. +- Redact common bearer tokens, private-key blocks, and credential assignments before classification. +- Validate all Jev labels at the integration boundary; never coerce unknown labels to a normal category. +- Preserve script-friendly stdout. Normal JSON/Markdown output goes to stdout; runtime errors follow existing stderr/exit-1 handling. +- No filesystem writes by the command. +- Test all new branches without live credentials or network calls. diff --git a/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md b/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md new file mode 100644 index 00000000..ad7d0d10 --- /dev/null +++ b/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md @@ -0,0 +1,87 @@ +--- +phase: implementation +title: Implementation Guide +description: Technical implementation notes, patterns, and code guidelines +--- + +# Jev Session Compaction Implementation + +## Development Setup + +**How do we get started?** + +- Worktree: `.worktrees/feature-session-compact-jev` on `feature-session-compact-jev`. +- Install with `npm ci`; build with `npm run build`. +- Runtime Jev access requires `TYPESAFE_API_KEY`. Tests never require a live key. +- Official SDK dependency: `@typesafe-ai/sdk@0.6.0` in the CLI workspace. + +## Code Structure + +**How is the code organized?** + +- `services/session-compact/session-compact.types.ts`: compaction domain types and unavailable constants. +- `services/session-compact/session-compact.service.ts`: local redaction, filtering, assembly, and Markdown rendering. +- `services/session-compact/jev-classifier.ts`: official SDK adapter and typed answer validation. +- `__tests__/services/session-compact/`: network-free service and adapter tests. + +## Implementation Notes + +**Key technical details to remember:** + +### Core Features + +- Messages are locally redacted, classified sequentially, filtered, and grouped without generative rewriting. +- Jev asks typed category/importance choice questions and retention/sensitivity noul questions. +- Unknown category or importance labels throw explicit integration errors. + +### Patterns & Best Practices + +- `SessionEventClassifier` is the only injectable boundary needed by the deterministic service. +- `createJevSessionEventClassifier` constructs the production SDK client with logging disabled. +- Threshold `>= 0.5` converts Jev noul probabilities to keep/sensitive booleans. +- Category/importance labels and both noul probabilities are validated at the SDK boundary; blank optional model configuration normalizes to `jev-latest`. + +## Integration Points + +**How do pieces connect?** + +- `TypeSafeClient.systemOne` uses `jev-latest` unless `TYPESAFE_DEFAULT_MODEL` is configured. +- No database or persistent state is added. +- CLI wiring will pass adapter-normalized `ConversationMessage[]` to `compactSession`. +- `agent session compact --id [--type ] [--format markdown|json]` is registered beside historical session detail. +- The missing-key branch returns before `createAgentManager`, so it cannot list sessions or read a transcript. +- Successful resolution reuses `resolveListSessionsOptions`, `findSessionById`, adapter lookup, and `getConversation(..., { verbose: true })`. + +## Error Handling + +**How do we handle failures?** + +- Missing key will be handled before constructing the SDK client or reading a transcript. +- SDK/API failures propagate through existing CLI error handling; there is no fallback. +- SDK logging is disabled so request bodies cannot be emitted by this feature. + +## Performance Considerations + +**How do we keep it fast?** + +- MVP classification is sequential and uncached; measure before adding concurrency or caching. +- Each source message produces exactly one Jev request. + +## Security Notes + +**What security measures are in place?** + +- The API key is passed only to `TypeSafeClient` and never put in Jev state. +- Common bearer tokens, private-key blocks, and secret assignments are redacted locally. +- Jev-sensitive events are excluded from the compact artifact. +- Redaction is defense in depth, not complete DLP. + +## Delivered Skill + +- `skills/session-compact/SKILL.md` routes long-context, handoff, stale-resume, complex-close, memory-candidate, and task-progress use cases to the CLI. +- It requires agents to report missing `TYPESAFE_API_KEY` clearly and forbids claiming a manual fallback was Jev-backed. +- `skills/built-in.json` and the offline fallback list both include the skill. + +## Design Alignment + +The implementation matches the reviewed design with no material deviations. The only review-driven addition was stricter validation of Jev probabilities and blank model configuration. No provider/session discovery code, database state, automatic memory/task mutation, fallback summarizer, or output-file behavior was added. diff --git a/docs/ai/planning/2026-09-22-feature-session-compact-jev.md b/docs/ai/planning/2026-09-22-feature-session-compact-jev.md new file mode 100644 index 00000000..471a3b41 --- /dev/null +++ b/docs/ai/planning/2026-09-22-feature-session-compact-jev.md @@ -0,0 +1,112 @@ +--- +phase: planning +title: Jev Session Compaction Plan +description: Ordered implementation and validation tasks for session compaction +--- + +# Jev Session Compaction Plan + +## Milestones + +- [x] Milestone 1: Typed compaction domain and Jev adapter are covered by unit tests. +- [x] Milestone 2: `agent session compact` implements explicit unavailable and successful output flows. +- [x] Milestone 3: Built-in skill, lifecycle docs, and repository verification are complete. + +## Task Breakdown + +### Phase 1: Test-first service foundation + +- [x] Task 1.1: Add the official `@typesafe-ai/sdk` CLI dependency. + - Outcome: reproducible Jev SDK integration on Node 20. + - Dependencies: none. + - Evidence: lockfile diff and CLI package build. +- [x] Task 1.2: Write failing tests for compaction filtering, grouping, resume-prompt assembly, Markdown headings, and secret redaction. + - Outcome: deterministic artifact contract is executable before production code. + - Dependencies: requirements/design schemas. + - Evidence: focused Vitest failure for missing modules/behavior. + - Covers: service and security scenarios in the testing strategy. +- [x] Task 1.3: Implement compaction types and service until Task 1.2 passes. + - Outcome: provider-independent, network-independent compaction core. + - Dependencies: Task 1.2. + - Evidence: focused service tests pass. +- [x] Task 1.4: Write failing Jev adapter tests, then implement the injectable SDK boundary and strict answer validation. + - Outcome: typed questions map to validated domain events without exposing credentials. + - Dependencies: Tasks 1.1 and 1.3. + - Evidence: mocked-classifier tests pass; no network calls. + +### Phase 2: CLI integration + +- [x] Task 2.1: Write failing command tests for Markdown/JSON unavailable results, early key gate, format validation, session resolution, and successful rendering. + - Outcome: user-visible semantics are locked before wiring. + - Dependencies: Phase 1. + - Evidence: focused command tests initially fail for absent command, then pass. +- [x] Task 2.2: Register `agent session compact` using existing session ID/type resolution and verbose adapter conversation parsing. + - Outcome: discoverable command with no fallback and no transcript read when the key is absent. + - Dependencies: Task 2.1. + - Evidence: command tests and built `--help` smoke test. +- [x] Task 2.3: Verify built unavailable behavior in both formats. + - Outcome: status 0 and exact output from compiled CLI. + - Dependencies: Task 2.2 and CLI build. + - Evidence: shell smoke commands with `TYPESAFE_API_KEY` unset. + +### Phase 3: Skill, docs, and validation + +- [x] Task 3.1: Create `skills/session-compact/SKILL.md` and add it to the built-in manifest/fallback list with validation tests. + - Outcome: installed agents know when and how to use the command and report Jev unavailable. + - Dependencies: stable CLI syntax from Phase 2. + - Evidence: skill tests/lint and manifest assertions. +- [x] Task 3.2: Reconcile implementation and testing docs with actual files, decisions, and evidence. + - Outcome: lifecycle artifacts reflect delivered behavior rather than the initial forecast. + - Dependencies: implementation complete. + - Evidence: feature lint. +- [x] Task 3.3: Run focused tests, CLI package tests/build, workspace build/test as proportionate, formatting/lint, and final lifecycle review. + - Outcome: evidence-backed completion or documented blockers. + - Dependencies: all preceding tasks. + - Evidence: fresh command outputs recorded in testing docs and durable task. + +## Dependencies + +```mermaid +flowchart LR + T11[SDK dependency] --> T14[Jev adapter] + T12[Service tests] --> T13[Service implementation] + T13 --> T14 + T14 --> T21[CLI tests] + T21 --> T22[CLI wiring] + T22 --> T23[Built smoke] + T22 --> T31[Skill] + T23 --> T32[Docs reconciliation] + T31 --> T32 + T32 --> T33[Final verification] +``` + +- The TypeSafe API is external, but automated tests use mocks and do not require credentials. +- Live-key validation is optional and cannot block CI or local verification. +- No migration or persistent state dependency exists. + +## Timeline & Estimates + +- Service and adapter: small-to-medium, approximately half a day. +- CLI/test integration: small, approximately two hours. +- Skill/docs/final verification: small, approximately two hours. +- Buffer is reserved for SDK response-shape or existing command-test fixture differences. + +## Risks & Mitigation + +- SDK/API schema changes: isolate in `JevSessionEventClassifier`, validate labels, pin via lockfile. +- Long sessions cause many sequential calls: accept for MVP, measure before adding concurrency. +- Secret leakage: early key gate, local redaction before calls, sensitive-event exclusion, synthetic security tests. +- Command namespace confusion: reuse existing `agent session` and document discovery via `agent sessions`. +- Command tests are already broad: add focused cases and avoid changing shared behavior. +- Live API unavailable: mock integration boundary; distinguish configuration unavailability from API failure. + +## Resources Needed + +- Existing `@ai-devkit/agent-manager` session discovery/conversation APIs. +- Official `@typesafe-ai/sdk` package. +- Vitest, Commander test harness, lifecycle lint, and existing built-in skill validation. +- No database, new service, or deployment infrastructure. + +## Progress Summary + +All planned tasks are complete. Final evidence: CLI package 98 files / 1171 tests passed; feature coverage reached 97.1% statements, 96.77% branches, 90.9% functions, and 98.36% lines; all workspace test/build/lint targets passed; changed TypeScript files are formatted; feature lint and compiled unavailable smoke tests passed. A live paid-key call remains intentionally optional. diff --git a/docs/ai/requirements/2026-09-22-feature-session-compact-jev.md b/docs/ai/requirements/2026-09-22-feature-session-compact-jev.md new file mode 100644 index 00000000..af1b5ded --- /dev/null +++ b/docs/ai/requirements/2026-09-22-feature-session-compact-jev.md @@ -0,0 +1,73 @@ +--- +phase: requirements +title: Jev Session Compaction Requirements +description: Resume-ready historical session compaction with explicit Jev availability +--- + +# Jev Session Compaction Requirements + +## Problem Statement + +Long AI coding sessions contain operational facts that ordinary narrative summaries can lose: user constraints, decisions, changed files, commands, validation evidence, blockers, and the next action. AI DevKit can discover and read historical sessions, but users currently have to inspect the transcript and hand-write a continuation artifact. + +The feature serves developers and agent harnesses handing work to another agent, resuming stale work, or preserving the useful state before context pressure becomes severe. + +## Goals & Objectives + +- Add a discoverable command that compacts one historical session into a concise continuation artifact. +- Use Jev as a typed decision layer to decide which normalized session messages survive, their category and importance, and whether they contain sensitive material. +- Make Jev availability explicit by checking `TYPESAFE_API_KEY` before reading or classifying the transcript. +- Produce equivalent Markdown and JSON representations for humans and automation. +- Add a built-in `session-compact` skill explaining when and how agents should call the command. + +### Non-goals + +- No non-Jev summarization fallback. +- No automatic memory writes, task mutation, daemon, context-pressure hook, or session mutation. +- No new session-file discovery implementation or general provider abstraction. +- No `--current`, `--agent`, arbitrary `--input`, or output-file option in the MVP. +- No generative model call; retained event content is organized deterministically. + +## User Stories & Use Cases + +- As a developer, I can run `ai-devkit agent session compact --id ` and receive a resume-ready Markdown artifact. +- As an automation author, I can pass `--format json` and receive a stable structured result. +- As a user without TypeSafe credentials, I receive an explicit Jev-unavailable result instead of a fallback or stack trace. +- As a user with multiple providers sharing a session ID, I can pass `--type` using the same resolution rules as `agent session detail`. +- As an agent, I can use the built-in skill when handing off work, nearing context limits, resuming stale work, or extracting candidate durable memories. + +### Edge cases + +- Unknown and ambiguous session IDs use the existing clear session-resolution errors. +- An empty normalized conversation produces an empty but valid compact result. +- A message classified as sensitive is excluded from all compact fields. +- Obvious credential patterns are locally redacted before network transmission. +- Invalid Jev responses, authentication failures, rate limits, and network failures are runtime errors and are not mislabeled as unavailable. +- The API key is never included in output, logs, error details, or Jev state. + +## Success Criteria + +- `ai-devkit agent session compact --help` is discoverable under the existing historical-session namespace. +- `--id ` is required; optional `--type` narrows provider resolution. +- Markdown is the default; `--format markdown|json` rejects other values. +- With no `TYPESAFE_API_KEY`, the command exits with status 0, does not read the session or call Jev, and prints exactly `Jev is unavailable because TYPESAFE_API_KEY is not set.` in Markdown mode. +- JSON unavailable output contains `{ "jev": { "available": false, "reason": "TYPESAFE_API_KEY is not set" } }`. +- With a key, each normalized message is evaluated for retention, category, importance, and sensitivity using Jev typed questions. +- The result contains intent, current state, decisions, changed files, commands, validation, open questions, next step, memory candidates, resume prompt, and Jev metadata. +- Markdown contains the required `# Session Compact` and section headings. +- Unit tests cover unavailable behavior, classification mapping, secret redaction/exclusion, rendering, option validation, and command wiring. +- Existing `agent sessions` and `agent session detail` tests remain green. +- A built-in `session-compact` skill documents the supported command and explicit unavailable behavior. + +## Constraints & Assumptions + +- Node.js 20+ and Commander conventions remain unchanged. +- Historical-session discovery and provider parsing remain owned by `@ai-devkit/agent-manager`; the CLI resolves an ID, then calls the adapter's `getConversation(..., { verbose: true })`. +- `@typesafe-ai/sdk` is the narrow Jev integration dependency. `TypeSafeClient` reads `TYPESAFE_API_KEY`; production code still performs its own presence check first. +- Jev is classification-only. A deterministic builder groups retained source content into the output schema and builds the resume prompt. +- Secret redaction is defense in depth, not a complete data-loss-prevention system; users remain responsible for the transcript they send to the hosted service. +- The unavailable result is an expected capability probe and therefore exits 0. Malformed input and runtime/API failures exit non-zero through normal CLI error handling. + +## Questions & Open Items + +All MVP decisions are resolved. Potential follow-ups are `--current`, agent-name resolution, arbitrary normalized input, `--out`, bounded concurrency/batching after measurement, and optional non-Jev compaction under an explicitly different mode. diff --git a/docs/ai/testing/2026-09-22-feature-session-compact-jev.md b/docs/ai/testing/2026-09-22-feature-session-compact-jev.md new file mode 100644 index 00000000..35b5ed05 --- /dev/null +++ b/docs/ai/testing/2026-09-22-feature-session-compact-jev.md @@ -0,0 +1,103 @@ +--- +phase: testing +title: Jev Session Compaction Testing Strategy +description: Coverage for explicit availability, typed classification, rendering, security, and CLI integration +--- + +# Jev Session Compaction Testing Strategy + +## Test Coverage Goals + +- Cover 100% of new compaction service branches where practical. +- Exercise all acceptance paths without live Jev calls. +- Preserve command help and existing historical-session behavior. +- Treat missing-key behavior and non-disclosure of credentials as release-blocking tests. + +## Unit Tests + +### Session compact service + +- [x] Returns an empty valid compact for an empty conversation. +- [x] Groups retained events into the equivalent JSON fields. +- [x] Drops `keep: false`, `discard`, `irrelevant`, and `sensitive` events. +- [x] Uses the first retained user instruction for intent and latest next-step event for next step. +- [x] Builds current state and resume prompt deterministically. +- [x] Renders every required Markdown heading. +- [x] Redacts bearer tokens, private-key blocks, and common secret assignments. + +### Jev classifier + +- [x] Sends redacted role/content/timestamp state and all four typed questions. +- [x] Maps valid typed answers into a classified event. +- [x] Rejects unknown category or importance labels. +- [x] Never includes the API key in request state or thrown validation messages. + +### CLI helpers and wiring + +- [x] Rejects unsupported `--format` values. +- [x] Missing key prints exact Markdown unavailable text and does not create/list/read a session. +- [x] Missing key JSON emits the explicit unavailable object. +- [x] Present key resolves ID/type, reads verbose normalized conversation, invokes compaction, and emits selected format. +- [x] Unknown and ambiguous sessions reuse the tested `findSessionById` resolution path and add explicit compact errors. + +### Built-in skill + +- [x] Manifest includes `session-compact` exactly once. +- [x] Skill frontmatter is valid and documents missing-key behavior. + +## Integration Tests + +- [x] Run the focused CLI Vitest suites with a mocked SDK/classifier boundary. +- [x] Build the CLI package with the official SDK dependency. +- [x] Run `ai-devkit agent session compact --help` from built output and assert command/options are discoverable. +- [x] Run the built command with `TYPESAFE_API_KEY` absent and assert status 0 plus exact output. +- [ ] Run feature lint and repository lint. + +## End-to-End Tests + +- [ ] Manual live-key smoke test is documented but optional; CI must not require paid credentials. +- [x] Confirm an unavailable invocation never reads a real historical transcript. +- [x] Confirm existing `agent sessions` and `agent session detail` focused tests remain green. + +## Test Data + +- In-memory `ConversationMessage` fixtures spanning all categories and importance levels. +- Fake classifier implementations and mocked `@typesafe-ai/sdk` client responses. +- Secret-shaped strings that are synthetic and not usable credentials. +- Existing agent command manager/adapter mocks for session resolution. + +## Test Reporting & Coverage + +- Focused: `npx vitest run src/__tests__/services/session-compact src/__tests__/commands/agent.test.ts` from `packages/cli`. +- Package: `npm test --workspace packages/cli`. +- Build: `npm run build --workspace packages/cli` and final workspace `npm run build`. +- Lifecycle: `npx ai-devkit@latest lint --feature session-compact-jev`. +- Record fresh command evidence in this document and the durable task after implementation. + +## Manual Testing + +- Check Markdown readability and JSON parseability. +- Check exact missing-key output and successful exit status. +- Check help hierarchy and option descriptions. +- A live Jev call requires an operator-provided key and must never print it. + +## Performance Testing + +- Unit-test sequential call count equals the number of normalized messages. +- Record a follow-up if representative long sessions make sequential classification impractical; do not add batching without measured need. + +## Bug Tracking + +- Security disclosure, silent fallback, incorrect successful availability, or transcript reads before the key gate are release blockers. +- Schema drift and provider-specific parser failures are normal runtime errors and require regression fixtures when encountered. + +## Validation Results + +- CLI package: 98 test files and 1171 tests passed. +- Focused feature coverage: 97.1% statements, 96.77% branches, 90.9% functions, and 98.36% lines across the session-compaction service files. +- Workspace: all six lint, build, and test targets completed successfully. Lint retains four unrelated pre-existing warnings in channel/preview code. +- Compiled CLI: help lists `agent session compact`; missing-key Markdown and JSON outputs match the required contracts and exit successfully. +- Skill: `quick_validate.py skills/session-compact` passed; built-in manifest/fallback tests passed. +- Formatting: all nine changed TypeScript files pass `oxfmt --check`. The repository-wide formatter still reports two unrelated pre-existing agent-manager files, which were not modified. +- Lifecycle: base and `session-compact-jev` feature lint passed; `git diff --check` passed. +- Live Jev call: not run because no paid credential is required for automated validation. SDK request construction and response handling are covered with mocked boundary tests. diff --git a/package-lock.json b/package-lock.json index 550d5a78..f89d0c81 100644 --- a/package-lock.json +++ b/package-lock.json @@ -6721,6 +6721,15 @@ "dev": true, "license": "MIT" }, + "node_modules/@typesafe-ai/sdk": { + "version": "0.6.0", + "resolved": "https://registry.npmjs.org/@typesafe-ai/sdk/-/sdk-0.6.0.tgz", + "integrity": "sha512-IddX+Q0XM+VagOUZFeP7wZjaO4SHMdvnh2zEBdrZZnXedWI3BNK1lKhMx3ayrkFWvVLbVcUHJy6AVZlY+e6Jaw==", + "license": "MIT", + "engines": { + "node": ">=20" + } + }, "node_modules/@vitest/coverage-v8": { "version": "4.1.8", "resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.8.tgz", @@ -13604,6 +13613,7 @@ "@ai-devkit/channel-connector": "0.13.4", "@ai-devkit/memory": "0.20.1", "@inquirer/prompts": "^8.5.2", + "@typesafe-ai/sdk": "^0.6.0", "chalk": "^5.6.0", "commander": "^11.1.0", "debug": "^4.4.3", diff --git a/packages/cli/package.json b/packages/cli/package.json index 19612092..9e68b00c 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -40,6 +40,7 @@ "@ai-devkit/channel-connector": "0.13.4", "@ai-devkit/memory": "0.20.1", "@inquirer/prompts": "^8.5.2", + "@typesafe-ai/sdk": "^0.6.0", "chalk": "^5.6.0", "commander": "^11.1.0", "debug": "^4.4.3", diff --git a/packages/cli/src/__tests__/commands/agent.test.ts b/packages/cli/src/__tests__/commands/agent.test.ts index 47b314a5..25d98ded 100644 --- a/packages/cli/src/__tests__/commands/agent.test.ts +++ b/packages/cli/src/__tests__/commands/agent.test.ts @@ -59,6 +59,12 @@ const mockSpinner: any = { }; const mockSelect: any = vi.fn(); +const { mockCompactSession, mockRenderSessionCompactMarkdown, mockCreateJevClassifier } = + vi.hoisted(() => ({ + mockCompactSession: vi.fn(), + mockRenderSessionCompactMarkdown: vi.fn(), + mockCreateJevClassifier: vi.fn(), + })); const mockTtyWriterSend = vi .fn<(location: any, message: string) => Promise>() @@ -295,6 +301,15 @@ vi.mock("../../util/debug.js", () => ({ createLogger: () => mockDebugLogger, })); +vi.mock("../../services/session-compact/session-compact.service.js", () => ({ + compactSession: mockCompactSession, + renderSessionCompactMarkdown: mockRenderSessionCompactMarkdown, +})); + +vi.mock("../../services/session-compact/jev-classifier.js", () => ({ + createJevSessionEventClassifier: mockCreateJevClassifier, +})); + vi.mock("../../util/tmux.js", () => ({ resolveTmuxInstallInstructions: mockTmuxInstructions, })); @@ -379,6 +394,9 @@ describe("agent command", () => { mockManager.resolveAgent.mockReset(); mockManager.getAdapter.mockReset(); mockAgentAdapter.getConversation.mockReset(); + mockCompactSession.mockReset(); + mockRenderSessionCompactMarkdown.mockReset(); + mockCreateJevClassifier.mockReset(); mockDurableRepository.list.mockReset().mockResolvedValue([]); mockDurableRepository.resolve.mockReset().mockResolvedValue(null); mockDurableService.create.mockReset(); @@ -2770,6 +2788,179 @@ Waiting on user input`, }); }); + describe("session compact", () => { + const session = { + type: "codex", + sessionId: "sess-compact", + cwd: "/repo", + firstUserMessage: "implement it", + lastActive: new Date("2026-09-22T01:00:00Z"), + startedAt: new Date("2026-09-22T00:00:00Z"), + sessionFilePath: "/tmp/sess-compact.jsonl", + }; + + afterEach(() => { + vi.unstubAllEnvs(); + }); + + it("returns the exact Markdown unavailable result before reading sessions", async () => { + vi.stubEnv("TYPESAFE_API_KEY", ""); + const program = new Command(); + registerAgentCommand(program); + + await program.parseAsync([ + "node", + "test", + "agent", + "session", + "compact", + "--id", + "sess-compact", + ]); + + expect(logSpy).toHaveBeenCalledWith( + "Jev is unavailable because TYPESAFE_API_KEY is not set.", + ); + expect(mockManager.listSessions).not.toHaveBeenCalled(); + expect(mockCreateJevClassifier).not.toHaveBeenCalled(); + expect(process.exit).not.toHaveBeenCalled(); + }); + + it("returns a structured unavailable result in JSON mode", async () => { + vi.stubEnv("TYPESAFE_API_KEY", ""); + const program = new Command(); + registerAgentCommand(program); + + await program.parseAsync([ + "node", + "test", + "agent", + "session", + "compact", + "--id", + "sess-compact", + "--format", + "json", + ]); + + expect(JSON.parse(logSpy.mock.calls[0][0] as string)).toEqual({ + jev: { available: false, reason: "TYPESAFE_API_KEY is not set" }, + }); + expect(mockManager.listSessions).not.toHaveBeenCalled(); + }); + + it("resolves the historical session and renders a Jev compact", async () => { + vi.stubEnv("TYPESAFE_API_KEY", "test-key"); + const messages = [{ role: "user", content: "implement it" }]; + const classifier = { model: "jev-test", classify: vi.fn() }; + const compact = { + intent: "implement it", + currentState: "done", + decisions: [], + changedFiles: [], + commands: [], + validation: [], + openQuestions: [], + nextStep: "review", + memoryCandidates: [], + resumePrompt: "review", + jev: { available: true, model: "jev-test", classifiedEvents: 1 }, + }; + mockManager.listSessions.mockResolvedValue([session]); + mockManager.getAdapter.mockReturnValue(mockAgentAdapter); + mockAgentAdapter.getConversation.mockReturnValue(messages); + mockCreateJevClassifier.mockReturnValue(classifier); + mockCompactSession.mockResolvedValue(compact); + mockRenderSessionCompactMarkdown.mockReturnValue("# Session Compact\n"); + const program = new Command(); + registerAgentCommand(program); + + await program.parseAsync([ + "node", + "test", + "agent", + "session", + "compact", + "--id", + "sess-compact", + "--type", + "codex", + ]); + + expect(mockManager.listSessions).toHaveBeenCalledWith({ cwd: undefined, type: "codex" }); + expect(mockAgentAdapter.getConversation).toHaveBeenCalledWith("/tmp/sess-compact.jsonl", { + verbose: true, + }); + expect(mockCreateJevClassifier).toHaveBeenCalledWith("test-key"); + expect(mockCompactSession).toHaveBeenCalledWith(messages, classifier); + expect(logSpy).toHaveBeenCalledWith("# Session Compact\n"); + }); + + it("emits the successful compact as JSON without Markdown rendering", async () => { + vi.stubEnv("TYPESAFE_API_KEY", "test-key"); + const classifier = { model: "jev-test", classify: vi.fn() }; + const compact = { + intent: "implement it", + currentState: "done", + decisions: [], + changedFiles: [], + commands: [], + validation: [], + openQuestions: [], + nextStep: "review", + memoryCandidates: [], + resumePrompt: "review", + jev: { available: true, model: "jev-test", classifiedEvents: 1 }, + }; + mockManager.listSessions.mockResolvedValue([session]); + mockManager.getAdapter.mockReturnValue(mockAgentAdapter); + mockAgentAdapter.getConversation.mockReturnValue([]); + mockCreateJevClassifier.mockReturnValue(classifier); + mockCompactSession.mockResolvedValue(compact); + const program = new Command(); + registerAgentCommand(program); + + await program.parseAsync([ + "node", + "test", + "agent", + "session", + "compact", + "--id", + "sess-compact", + "--format", + "json", + ]); + + expect(JSON.parse(logSpy.mock.calls[0][0] as string)).toEqual(compact); + expect(mockRenderSessionCompactMarkdown).not.toHaveBeenCalled(); + }); + + it("rejects an unsupported format before reading sessions", async () => { + vi.stubEnv("TYPESAFE_API_KEY", "test-key"); + const program = new Command(); + registerAgentCommand(program); + + await program.parseAsync([ + "node", + "test", + "agent", + "session", + "compact", + "--id", + "sess-compact", + "--format", + "yaml", + ]); + + expect(ui.error).toHaveBeenCalledWith( + "Failed to compact session: Invalid --format. Expected markdown or json.", + ); + expect(mockManager.listSessions).not.toHaveBeenCalled(); + expect(process.exit).toHaveBeenCalledWith(1); + }); + }); + describe("agent rename", () => { it("calls registry.rename and prints success", async () => { mockRegistry.rename.mockReturnValue(undefined); diff --git a/packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts b/packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts new file mode 100644 index 00000000..d80c0e5a --- /dev/null +++ b/packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts @@ -0,0 +1,104 @@ +import type { ConversationMessage } from "@ai-devkit/agent-manager"; +import { + createJevSessionEventClassifier, + JevSessionEventClassifier, +} from "../../../services/session-compact/jev-classifier.js"; + +function choiceAnswer(choice: string) { + return { type: "choice" as const, choice, confidence: 0.9, probabilities: {} }; +} + +describe("JevSessionEventClassifier", () => { + afterEach(() => vi.unstubAllEnvs()); + + it("uses jev-latest when the optional model environment value is blank", () => { + vi.stubEnv("TYPESAFE_DEFAULT_MODEL", " "); + + expect(createJevSessionEventClassifier("test-key").model).toBe("jev-latest"); + }); + + it("asks typed questions and maps the answers into a classified event", async () => { + const systemOne = vi.fn().mockResolvedValue({ + model: "jev-1.13.0", + usage: { input_tokens: 10, output_tokens: 0 }, + answers: { + category: choiceAnswer("code_change"), + importance: choiceAnswer("critical"), + keep: { type: "noul", noul: 0.8 }, + sensitive: { type: "noul", noul: 0.1 }, + }, + }); + const classifier = new JevSessionEventClassifier({ systemOne }, "jev-latest"); + const message: ConversationMessage = { + role: "assistant", + content: "Authorization: Bearer secret-token\nChanged agent.ts", + timestamp: "2026-09-22T00:00:00.000Z", + }; + + const result = await classifier.classify(message); + + expect(result).toEqual({ + role: "assistant", + content: "Authorization: Bearer [REDACTED]\nChanged agent.ts", + timestamp: "2026-09-22T00:00:00.000Z", + category: "code_change", + importance: "critical", + keep: true, + sensitive: false, + }); + expect(systemOne).toHaveBeenCalledOnce(); + const request = systemOne.mock.calls[0][0]; + expect(request.model).toBe("jev-latest"); + expect(request.state).toEqual({ + role: "assistant", + content: "Authorization: Bearer [REDACTED]\nChanged agent.ts", + timestamp: "2026-09-22T00:00:00.000Z", + }); + expect(Object.keys(request.questions)).toEqual(["category", "importance", "keep", "sensitive"]); + expect(request.questions.category.type).toBe("choice"); + expect(request.questions.importance.type).toBe("choice"); + expect(request.questions.keep.type).toBe("noul"); + expect(request.questions.sensitive.type).toBe("noul"); + expect(JSON.stringify(request)).not.toContain("secret-token"); + }); + + it.each([ + ["category", "unexpected", "Unexpected Jev category: unexpected"], + ["importance", "urgent", "Unexpected Jev importance: urgent"], + ])("rejects an unknown %s label", async (field, value, expectedMessage) => { + const answers = { + category: choiceAnswer(field === "category" ? value : "decision"), + importance: choiceAnswer(field === "importance" ? value : "important"), + keep: { type: "noul" as const, noul: 0.9 }, + sensitive: { type: "noul" as const, noul: 0.1 }, + }; + const classifier = new JevSessionEventClassifier( + { systemOne: vi.fn().mockResolvedValue({ model: "jev-latest", usage: {}, answers }) }, + "jev-latest", + ); + + await expect(classifier.classify({ role: "assistant", content: "content" })).rejects.toThrow( + expectedMessage, + ); + }); + + it.each([ + ["keep", Number.NaN], + ["sensitive", 1.2], + ])("rejects an invalid %s noul probability", async (field, value) => { + const answers = { + category: choiceAnswer("decision"), + importance: choiceAnswer("important"), + keep: { type: "noul" as const, noul: field === "keep" ? value : 0.9 }, + sensitive: { type: "noul" as const, noul: field === "sensitive" ? value : 0.1 }, + }; + const classifier = new JevSessionEventClassifier( + { systemOne: vi.fn().mockResolvedValue({ model: "jev-latest", usage: {}, answers }) }, + "jev-latest", + ); + + await expect(classifier.classify({ role: "assistant", content: "content" })).rejects.toThrow( + `Unexpected Jev ${field} probability`, + ); + }); +}); diff --git a/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts b/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts new file mode 100644 index 00000000..cc25c5a4 --- /dev/null +++ b/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts @@ -0,0 +1,169 @@ +import type { ConversationMessage } from "@ai-devkit/agent-manager"; +import { + compactSession, + redactSensitiveText, + renderSessionCompactMarkdown, + type SessionEventClassifier, +} from "../../../services/session-compact/session-compact.service.js"; +import type { + ClassifiedSessionEvent, + CompactCategory, +} from "../../../services/session-compact/session-compact.types.js"; + +function event( + category: CompactCategory, + content: string, + overrides: Partial = {}, +): ClassifiedSessionEvent { + return { + role: "assistant", + content, + keep: true, + category, + importance: "important", + sensitive: false, + ...overrides, + }; +} + +function classifier(events: ClassifiedSessionEvent[]): SessionEventClassifier { + let index = 0; + return { + model: "jev-test", + classify: vi.fn(async () => events[index++]), + }; +} + +describe("session compaction", () => { + it("returns an empty but valid compact for an empty conversation", async () => { + const result = await compactSession([], classifier([])); + + expect(result).toEqual({ + intent: "", + currentState: "", + decisions: [], + changedFiles: [], + commands: [], + validation: [], + openQuestions: [], + nextStep: "", + memoryCandidates: [], + resumePrompt: "Continue the session from the compacted state above.", + jev: { available: true, model: "jev-test", classifiedEvents: 0 }, + }); + }); + + it("groups retained events and builds a deterministic continuation artifact", async () => { + const messages: ConversationMessage[] = Array.from({ length: 8 }, (_, index) => ({ + role: index === 0 ? "user" : "assistant", + content: `message-${index}`, + })); + const result = await compactSession( + messages, + classifier([ + event("user_instruction", "Keep the CLI explicit", { role: "user" }), + event("decision", "Use the existing namespace"), + event("code_change", "Changed packages/cli/src/commands/agent.ts"), + event("command_evidence", "npm run build"), + event("validation_evidence", "CLI tests passed"), + event("blocker", "Live key is unavailable"), + event("next_step", "Add the built-in skill"), + event("memory_candidate", "Adapters own transcript parsing"), + ]), + ); + + expect(result.intent).toBe("Keep the CLI explicit"); + expect(result.decisions).toEqual(["Use the existing namespace"]); + expect(result.changedFiles).toEqual(["Changed packages/cli/src/commands/agent.ts"]); + expect(result.commands).toEqual(["npm run build"]); + expect(result.validation).toEqual(["CLI tests passed"]); + expect(result.openQuestions).toEqual(["Live key is unavailable"]); + expect(result.nextStep).toBe("Add the built-in skill"); + expect(result.memoryCandidates).toEqual(["Adapters own transcript parsing"]); + expect(result.currentState).toContain("Changed packages/cli/src/commands/agent.ts"); + expect(result.resumePrompt).toContain("Intent: Keep the CLI explicit"); + expect(result.resumePrompt).toContain("Next step: Add the built-in skill"); + expect(result.jev.classifiedEvents).toBe(8); + }); + + it("excludes events that are discarded, irrelevant, sensitive, or not retained", async () => { + const messages: ConversationMessage[] = Array.from({ length: 4 }, () => ({ + role: "assistant", + content: "source", + })); + const result = await compactSession( + messages, + classifier([ + event("decision", "not retained", { keep: false }), + event("discard", "discarded"), + event("decision", "irrelevant", { importance: "irrelevant" }), + event("decision", "secret", { sensitive: true }), + ]), + ); + + expect(result.decisions).toEqual([]); + expect(result.currentState).toBe(""); + }); + + it("redacts common credentials before classification", () => { + const content = [ + "Authorization: Bearer abc.def.ghi", + "TYPESAFE_API_KEY=jv_live_example", + "-----BEGIN PRIVATE KEY-----\nprivate-material\n-----END PRIVATE KEY-----", + ].join("\n"); + + const redacted = redactSensitiveText(content); + + expect(redacted).not.toContain("abc.def.ghi"); + expect(redacted).not.toContain("jv_live_example"); + expect(redacted).not.toContain("private-material"); + expect(redacted.match(/\[REDACTED\]/g)?.length).toBeGreaterThanOrEqual(3); + }); + + it("renders all required Markdown sections", async () => { + const compact = await compactSession([], classifier([])); + + expect(renderSessionCompactMarkdown(compact)).toBe(`# Session Compact + +## User Intent + +_None recorded._ + +## Current State + +_None recorded._ + +## Decisions Made + +_None recorded._ + +## Files Touched + +_None recorded._ + +## Commands Run + +_None recorded._ + +## Validation Evidence + +_None recorded._ + +## Open Questions + +_None recorded._ + +## Next Step + +_None recorded._ + +## Memory Candidates + +_None recorded._ + +## Resume Prompt + +Continue the session from the compacted state above. +`); + }); +}); diff --git a/packages/cli/src/__tests__/services/skill/skill-builtins.test.ts b/packages/cli/src/__tests__/services/skill/skill-builtins.test.ts index 2f7b0d53..5de4ab6e 100644 --- a/packages/cli/src/__tests__/services/skill/skill-builtins.test.ts +++ b/packages/cli/src/__tests__/services/skill/skill-builtins.test.ts @@ -30,9 +30,10 @@ describe("getBuiltinSkillNames", () => { const { getBuiltinSkillNames } = await import("../../../services/skill/skill-builtins.js"); const names = await getBuiltinSkillNames(); - expect(names).toHaveLength(23); + expect(names).toHaveLength(24); expect(names).toContain("agent-communication"); expect(names).toContain("tdd"); + expect(names).toContain("session-compact"); }); it("falls back to the bundled list for an unsuccessful response", async () => { @@ -46,7 +47,7 @@ describe("getBuiltinSkillNames", () => { const { getBuiltinSkillNames } = await import("../../../services/skill/skill-builtins.js"); - await expect(getBuiltinSkillNames()).resolves.toHaveLength(23); + await expect(getBuiltinSkillNames()).resolves.toHaveLength(24); }); it("falls back to the bundled list when response JSON cannot be parsed", async () => { @@ -62,7 +63,7 @@ describe("getBuiltinSkillNames", () => { const { getBuiltinSkillNames } = await import("../../../services/skill/skill-builtins.js"); - await expect(getBuiltinSkillNames()).resolves.toHaveLength(23); + await expect(getBuiltinSkillNames()).resolves.toHaveLength(24); }); it.each([ @@ -83,7 +84,7 @@ describe("getBuiltinSkillNames", () => { const { getBuiltinSkillNames } = await import("../../../services/skill/skill-builtins.js"); - await expect(getBuiltinSkillNames()).resolves.toHaveLength(23); + await expect(getBuiltinSkillNames()).resolves.toHaveLength(24); }); }); @@ -92,7 +93,7 @@ describe("built-in skills manifest", () => { const manifestPath = new URL("../../../../../../skills/built-in.json", import.meta.url); const manifest = JSON.parse(await readFile(manifestPath, "utf8")); - expect(manifest).toHaveLength(23); + expect(manifest).toHaveLength(24); expect(manifest).toEqual( expect.arrayContaining([ "agent-communication", @@ -105,6 +106,7 @@ describe("built-in skills manifest", () => { "memory", "verify", "tdd", + "session-compact", ]), ); }); diff --git a/packages/cli/src/commands/agent.ts b/packages/cli/src/commands/agent.ts index 36c28877..498aee3a 100644 --- a/packages/cli/src/commands/agent.ts +++ b/packages/cli/src/commands/agent.ts @@ -67,6 +67,15 @@ import { resolveTmuxInstallInstructions } from "../util/tmux.js"; import { createTmuxInspectionDeps } from "../util/tmux-deps.js"; import { ConfigManager } from "../lib/Config.js"; import { getErrorMessage } from "../util/text.js"; +import { + compactSession, + renderSessionCompactMarkdown, +} from "../services/session-compact/session-compact.service.js"; +import { createJevSessionEventClassifier } from "../services/session-compact/jev-classifier.js"; +import { + JEV_UNAVAILABLE_MESSAGE, + JEV_UNAVAILABLE_REASON, +} from "../services/session-compact/session-compact.types.js"; // eslint-disable-next-line no-control-regex const ANSI_ESCAPE_PATTERN = /\x1b\[[0-9;]*m/g; @@ -607,6 +616,68 @@ export function registerAgentCommand(program: Command): void { }), ); + sessionCommand + .command("compact") + .description("Compact a historical session into a Jev-classified continuation artifact") + .requiredOption("--id ", "Session ID (as shown in agent sessions)") + .option( + "--type ", + "Filter to one of: claude, codex, gemini_cli, grok_cli, opencode, copilot, pi", + ) + .option("--format ", "Output format: markdown or json", "markdown") + .action( + withErrorHandler("compact session", async (options) => { + if (options.format !== "markdown" && options.format !== "json") { + throw new Error("Invalid --format. Expected markdown or json."); + } + + const apiKey = process.env.TYPESAFE_API_KEY?.trim(); + if (!apiKey) { + if (options.format === "json") { + console.log( + JSON.stringify( + { jev: { available: false, reason: JEV_UNAVAILABLE_REASON } }, + null, + 2, + ), + ); + } else { + console.log(JEV_UNAVAILABLE_MESSAGE); + } + return; + } + + const manager = createAgentManager(); + const listOptions = resolveListSessionsOptions({ + all: true, + type: options.type, + }).adapterOptions; + const sessions = await manager.listSessions(listOptions); + const resolved = findSessionById(sessions, options.id); + + if (!resolved) { + throw new Error(`No session found matching "${options.id}".`); + } + if (Array.isArray(resolved)) { + throw new Error( + `Multiple sessions match "${options.id}". Use --type to choose the intended session source.`, + ); + } + + const adapter = manager.getAdapter(resolved.type); + if (!adapter) throw new Error(`Unsupported agent type: ${resolved.type}`); + + const conversation = adapter.getConversation(resolved.sessionFilePath, { verbose: true }); + const classifier = createJevSessionEventClassifier(apiKey); + const result = await compactSession(conversation, classifier); + console.log( + options.format === "json" + ? JSON.stringify(result, null, 2) + : renderSessionCompactMarkdown(result), + ); + }), + ); + agentCommand .command("open ") .description("Focus a running agent terminal") diff --git a/packages/cli/src/services/session-compact/jev-classifier.ts b/packages/cli/src/services/session-compact/jev-classifier.ts new file mode 100644 index 00000000..b4220a16 --- /dev/null +++ b/packages/cli/src/services/session-compact/jev-classifier.ts @@ -0,0 +1,99 @@ +import type { ConversationMessage } from "@ai-devkit/agent-manager"; +import { choice, noul, TypeSafeClient, type SystemOneResult } from "@typesafe-ai/sdk"; +import { redactSensitiveText, type SessionEventClassifier } from "./session-compact.service.js"; +import { + COMPACT_CATEGORIES, + COMPACT_IMPORTANCE, + type ClassifiedSessionEvent, + type CompactCategory, + type CompactImportance, +} from "./session-compact.types.js"; + +const categoryCriteria = Object.fromEntries( + COMPACT_CATEGORIES.map((category) => [category, null]), +) as Record; + +const importanceCriteria = Object.fromEntries( + COMPACT_IMPORTANCE.map((importance) => [importance, null]), +) as Record; + +const classificationQuestions = { + category: choice( + "Which operational category best describes this coding-session event?", + categoryCriteria, + ), + importance: choice( + "How important is this event for accurately continuing the coding session?", + importanceCriteria, + ), + keep: noul("Should this event survive session compaction?"), + sensitive: noul("Does this event contain sensitive material that should not be stored?"), +}; + +interface JevClassificationClient { + systemOne(request: { + state: { role: string; content: string; timestamp?: string }; + questions: typeof classificationQuestions; + model: string; + }): Promise>; +} + +function isCategory(value: string): value is CompactCategory { + return (COMPACT_CATEGORIES as readonly string[]).includes(value); +} + +function isImportance(value: string): value is CompactImportance { + return (COMPACT_IMPORTANCE as readonly string[]).includes(value); +} + +function requireProbability(name: string, value: number): number { + if (!Number.isFinite(value) || value < 0 || value > 1) { + throw new Error(`Unexpected Jev ${name} probability`); + } + return value; +} + +export class JevSessionEventClassifier implements SessionEventClassifier { + constructor( + private readonly client: JevClassificationClient, + readonly model = "jev-latest", + ) {} + + async classify(message: ConversationMessage): Promise { + const redactedMessage = { ...message, content: redactSensitiveText(message.content) }; + const response = await this.client.systemOne({ + model: this.model, + state: redactedMessage, + questions: classificationQuestions, + }); + + const category = response.answers.category.choice; + if (!isCategory(category)) throw new Error(`Unexpected Jev category: ${category}`); + + const importance = response.answers.importance.choice; + if (!isImportance(importance)) throw new Error(`Unexpected Jev importance: ${importance}`); + + const keep = requireProbability("keep", response.answers.keep.noul); + const sensitive = requireProbability("sensitive", response.answers.sensitive.noul); + + return { + ...redactedMessage, + category, + importance, + keep: keep >= 0.5, + sensitive: sensitive >= 0.5, + }; + } +} + +export function createJevSessionEventClassifier( + apiKey: string, + model = process.env.TYPESAFE_DEFAULT_MODEL, +): JevSessionEventClassifier { + const resolvedModel = model?.trim() || "jev-latest"; + const sdk = new TypeSafeClient({ apiKey, defaultModel: resolvedModel, logLevel: "off" }); + const client: JevClassificationClient = { + systemOne: async (request) => sdk.systemOne(request), + }; + return new JevSessionEventClassifier(client, resolvedModel); +} diff --git a/packages/cli/src/services/session-compact/session-compact.service.ts b/packages/cli/src/services/session-compact/session-compact.service.ts new file mode 100644 index 00000000..39fb721c --- /dev/null +++ b/packages/cli/src/services/session-compact/session-compact.service.ts @@ -0,0 +1,140 @@ +import type { ConversationMessage } from "@ai-devkit/agent-manager"; +import type { ClassifiedSessionEvent, SessionCompact } from "./session-compact.types.js"; + +export interface SessionEventClassifier { + readonly model: string; + classify(message: ConversationMessage): Promise; +} + +const PRIVATE_KEY_PATTERN = + /-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g; +const BEARER_PATTERN = /(Authorization\s*:\s*Bearer\s+)[^\s]+/gi; +const SECRET_ASSIGNMENT_PATTERN = + /\b([A-Z][A-Z0-9_]*(?:API_KEY|TOKEN|SECRET|PASSWORD|PRIVATE_KEY))\s*=\s*([^\s]+)/g; + +export function redactSensitiveText(content: string): string { + return content + .replace(PRIVATE_KEY_PATTERN, "[REDACTED]") + .replace(BEARER_PATTERN, "$1[REDACTED]") + .replace(SECRET_ASSIGNMENT_PATTERN, "$1=[REDACTED]"); +} + +function includeEvent(event: ClassifiedSessionEvent): boolean { + return ( + event.keep && + !event.sensitive && + event.category !== "discard" && + event.importance !== "irrelevant" + ); +} + +function list( + events: ClassifiedSessionEvent[], + category: ClassifiedSessionEvent["category"], +): string[] { + return events.filter((event) => event.category === category).map((event) => event.content); +} + +function buildResumePrompt(intent: string, currentState: string, nextStep: string): string { + const lines = ["Continue the session from the compacted state above."]; + if (intent) lines.push(`Intent: ${intent}`); + if (currentState) lines.push(`Current state: ${currentState}`); + if (nextStep) lines.push(`Next step: ${nextStep}`); + return lines.join("\n"); +} + +export async function compactSession( + messages: ConversationMessage[], + classifier: SessionEventClassifier, +): Promise { + const classified: ClassifiedSessionEvent[] = []; + for (const message of messages) { + classified.push( + await classifier.classify({ ...message, content: redactSensitiveText(message.content) }), + ); + } + + const retained = classified.filter(includeEvent); + const intent = list(retained, "user_instruction")[0] ?? ""; + const decisions = list(retained, "decision"); + const changedFiles = list(retained, "code_change"); + const commands = list(retained, "command_evidence"); + const validation = list(retained, "validation_evidence"); + const openQuestions = list(retained, "blocker"); + const nextSteps = list(retained, "next_step"); + const nextStep = nextSteps.at(-1) ?? ""; + const memoryCandidates = list(retained, "memory_candidate"); + const currentState = retained + .filter((event) => + ["decision", "code_change", "validation_evidence", "blocker"].includes(event.category), + ) + .map((event) => event.content) + .join("\n"); + + return { + intent, + currentState, + decisions, + changedFiles, + commands, + validation, + openQuestions, + nextStep, + memoryCandidates, + resumePrompt: buildResumePrompt(intent, currentState, nextStep), + jev: { available: true, model: classifier.model, classifiedEvents: classified.length }, + }; +} + +function markdownValue(value: string): string { + return value || "_None recorded._"; +} + +function markdownList(values: string[]): string { + return values.length > 0 ? values.map((value) => `- ${value}`).join("\n") : "_None recorded._"; +} + +export function renderSessionCompactMarkdown(compact: SessionCompact): string { + return `# Session Compact + +## User Intent + +${markdownValue(compact.intent)} + +## Current State + +${markdownValue(compact.currentState)} + +## Decisions Made + +${markdownList(compact.decisions)} + +## Files Touched + +${markdownList(compact.changedFiles)} + +## Commands Run + +${markdownList(compact.commands)} + +## Validation Evidence + +${markdownList(compact.validation)} + +## Open Questions + +${markdownList(compact.openQuestions)} + +## Next Step + +${markdownValue(compact.nextStep)} + +## Memory Candidates + +${markdownList(compact.memoryCandidates)} + +## Resume Prompt + +${compact.resumePrompt} +`; +} diff --git a/packages/cli/src/services/session-compact/session-compact.types.ts b/packages/cli/src/services/session-compact/session-compact.types.ts new file mode 100644 index 00000000..bfd9bf93 --- /dev/null +++ b/packages/cli/src/services/session-compact/session-compact.types.ts @@ -0,0 +1,56 @@ +import type { ConversationMessage } from "@ai-devkit/agent-manager"; + +export const COMPACT_CATEGORIES = [ + "user_instruction", + "decision", + "code_change", + "command_evidence", + "validation_evidence", + "blocker", + "next_step", + "memory_candidate", + "discard", +] as const; + +export type CompactCategory = (typeof COMPACT_CATEGORIES)[number]; + +export const COMPACT_IMPORTANCE = ["irrelevant", "useful", "important", "critical"] as const; + +export type CompactImportance = (typeof COMPACT_IMPORTANCE)[number]; + +export interface ClassifiedSessionEvent extends ConversationMessage { + keep: boolean; + category: CompactCategory; + importance: CompactImportance; + sensitive: boolean; +} + +export interface SessionCompact { + intent: string; + currentState: string; + decisions: string[]; + changedFiles: string[]; + commands: string[]; + validation: string[]; + openQuestions: string[]; + nextStep: string; + memoryCandidates: string[]; + resumePrompt: string; + jev: { + available: true; + model: string; + classifiedEvents: number; + }; +} + +export interface JevUnavailableResult { + jev: { + available: false; + reason: "TYPESAFE_API_KEY is not set"; + }; +} + +export type SessionCompactResult = SessionCompact | JevUnavailableResult; + +export const JEV_UNAVAILABLE_REASON = "TYPESAFE_API_KEY is not set" as const; +export const JEV_UNAVAILABLE_MESSAGE = `Jev is unavailable because ${JEV_UNAVAILABLE_REASON}.`; diff --git a/packages/cli/src/services/skill/skill-builtins.ts b/packages/cli/src/services/skill/skill-builtins.ts index 827650ae..55382c7c 100644 --- a/packages/cli/src/services/skill/skill-builtins.ts +++ b/packages/cli/src/services/skill/skill-builtins.ts @@ -29,6 +29,7 @@ const FALLBACK_BUILTIN_SKILL_NAMES = [ "brainstorm", "verify", "tdd", + "session-compact", ] as const; let builtInSkillNamesPromise: Promise | undefined; diff --git a/skills/built-in.json b/skills/built-in.json index 259f8c2f..3e9e3ddd 100644 --- a/skills/built-in.json +++ b/skills/built-in.json @@ -21,5 +21,6 @@ "simplify-implementation", "brainstorm", "verify", - "tdd" + "tdd", + "session-compact" ] diff --git a/skills/session-compact/SKILL.md b/skills/session-compact/SKILL.md new file mode 100644 index 00000000..81c8a41b --- /dev/null +++ b/skills/session-compact/SKILL.md @@ -0,0 +1,51 @@ +--- +name: session-compact +description: AI DevKit · Compact a historical AI coding session with Jev when context is long, work is handed off or resumed, or durable continuation facts and memory candidates need extraction. +--- + +# Session Compact + +Use AI DevKit's Jev-backed CLI to create a structured continuation artifact from a historical session. Prefer this command over a hand-written compact when it is available. + +## When to Use + +- Context is getting long and useful state needs to survive compaction. +- Work is being handed to another agent or resumed after becoming stale. +- A complex implementation or debugging session is closing. +- Durable memory candidates or task-progress facts need to be identified after a long run. + +## Workflow + +1. Find the session ID when it is not already known: + ```bash + npx ai-devkit@latest agent sessions --all + ``` +2. Produce Markdown for a human handoff: + ```bash + npx ai-devkit@latest agent session compact --id + ``` +3. Use JSON for automation or structured inspection: + ```bash + npx ai-devkit@latest agent session compact --id --format json + ``` +4. Add `--type ` when the same session ID is ambiguous across providers. +5. Review the artifact before using its resume prompt, memory candidates, or validation claims. Compaction preserves selected transcript evidence; it does not independently verify that evidence. + +## Jev Availability + +The command requires `TYPESAFE_API_KEY`. If it reports: + +```text +Jev is unavailable because TYPESAFE_API_KEY is not set. +``` + +report that result clearly and stop the compaction attempt. Do not silently replace it with hand-written summarization or imply that Jev classified the session. A user may explicitly request a separate manual summary afterward. + +Never print, log, store, or pass the API key as a command argument. The CLI reads it from the environment. + +## Boundaries + +- The command is read-only and does not write memories or mutate tasks. +- Treat memory candidates as proposals until they pass the memory skill's quality gate. +- Record task progress separately when the task workflow requires it. +- API, authentication, schema, or network failures with a configured key are errors, not Jev-unavailable results. From 5f714f853ba7a7dfa256a3984f24427063b73a01 Mon Sep 17 00:00:00 2001 From: Hoang Nguyen Date: Wed, 23 Sep 2026 20:37:34 +0200 Subject: [PATCH 2/3] perf(session): accelerate Jev compaction --- .../2026-09-22-feature-session-compact-jev.md | 15 +++-- .../2026-09-22-feature-session-compact-jev.md | 9 +-- .../2026-09-22-feature-session-compact-jev.md | 14 ++++- .../2026-09-22-feature-session-compact-jev.md | 8 ++- .../2026-09-22-feature-session-compact-jev.md | 17 ++++-- packages/agent-manager/src/AgentManager.ts | 34 +++++++++++ .../src/__tests__/AgentManager.test.ts | 60 +++++++++++++++++++ .../adapters/ClaudeCodeAdapter.test.ts | 23 +++++++ .../__tests__/adapters/CodexAdapter.test.ts | 7 +++ .../__tests__/adapters/GrokCliAdapter.test.ts | 9 +++ .../providers/copilot/CopilotAdapter.test.ts | 12 ++++ .../providers/gemini/GeminiCliAdapter.test.ts | 21 +++++++ .../opencode/OpenCodeAdapter.test.ts | 18 ++++++ .../__tests__/providers/pi/PiAdapter.test.ts | 15 +++++ .../src/__tests__/utils/session.test.ts | 12 +++- .../src/adapters/AgentAdapter.ts | 9 +++ .../src/adapters/GrokCliAdapter.ts | 51 ++++++++++++---- .../src/providers/claude/ClaudeCodeAdapter.ts | 51 +++++++++------- .../providers/claude/ClaudeSessionLocator.ts | 20 +++++++ .../src/providers/codex/CodexAdapter.ts | 8 +++ .../src/providers/copilot/CopilotAdapter.ts | 36 +++++++---- .../copilot/CopilotSessionLocator.ts | 8 ++- .../src/providers/gemini/GeminiCliAdapter.ts | 8 +++ .../src/providers/opencode/OpenCodeAdapter.ts | 4 ++ .../opencode/OpenCodeSessionLocator.ts | 58 +++++++++++++----- .../src/providers/pi/PiAdapter.ts | 12 ++++ .../src/providers/pi/PiSessionLocator.ts | 17 +++++- packages/agent-manager/src/utils/session.ts | 11 ++++ .../cli/src/__tests__/commands/agent.test.ts | 13 +++- .../session-compact.service.test.ts | 48 +++++++++++++++ packages/cli/src/commands/agent.ts | 13 ++-- .../session-compact.service.ts | 31 ++++++++-- 32 files changed, 576 insertions(+), 96 deletions(-) diff --git a/docs/ai/design/2026-09-22-feature-session-compact-jev.md b/docs/ai/design/2026-09-22-feature-session-compact-jev.md index 631c4ef5..b979f491 100644 --- a/docs/ai/design/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/design/2026-09-22-feature-session-compact-jev.md @@ -14,11 +14,11 @@ The feature extends the existing `agent session` command group. The command owns flowchart TD CLI[agent session compact] --> Gate{TYPESAFE_API_KEY present?} Gate -->|no| Unavailable[Jev unavailable result] - Gate -->|yes| Sessions[AgentManager.listSessions] - Sessions --> Resolve[Resolve ID and optional type] + Gate -->|yes| Sessions[AgentManager.findSessionsById] + Sessions --> Resolve[Resolve exact ID and optional type] Resolve --> Adapter[Adapter.getConversation verbose] Adapter --> Redact[Local credential redaction] - Redact --> Jev[Jev classifier] + Redact --> Jev[Bounded Jev worker pool] Jev --> Filter[Exclude discard, irrelevant, sensitive] Filter --> Builder[Deterministic compact builder] Builder --> Markdown[Markdown renderer] @@ -83,7 +83,7 @@ ai-devkit agent session compact --id [--type ] [--format mark - Default format: `markdown`. - Unavailable Markdown is the single required sentence. - Unavailable JSON is the `JevUnavailable` object. -- Availability is checked before `AgentManager.listSessions()` so no transcript is read without a usable configuration signal. +- Availability is checked before `AgentManager.findSessionsById()` so no transcript is read without a usable configuration signal. ### Internal boundaries @@ -96,6 +96,7 @@ interface SessionEventClassifier { async function compactSession( messages: ConversationMessage[], classifier: SessionEventClassifier, + options?: { concurrency?: number }, ): Promise; function redactSensitiveText(content: string): string; @@ -109,7 +110,8 @@ The default classifier wraps `@typesafe-ai/sdk` and asks one `choice` question f - `packages/cli/src/services/session-compact/session-compact.types.ts`: stable feature types and enums. - `packages/cli/src/services/session-compact/jev-classifier.ts`: SDK adapter and response validation. - `packages/cli/src/services/session-compact/session-compact.service.ts`: redaction, filtering, deterministic assembly, and Markdown rendering. -- `packages/cli/src/commands/agent.ts`: Commander wiring, early availability gate, existing session resolution, output selection. +- `packages/agent-manager/src/AgentManager.ts` and built-in adapters: exact session-ID resolution using provider-native storage, with a compatibility fallback for external adapters. +- `packages/cli/src/commands/agent.ts`: Commander wiring, early availability gate, exact session resolution, output selection. - `packages/cli/src/__tests__/services/session-compact/*`: classifier/service unit tests with no network. - `packages/cli/src/__tests__/commands/agent.test.ts`: command-level unavailable and successful wiring tests. - `skills/session-compact/SKILL.md` and `skills/built-in.json`: agent instructions and distribution manifest. @@ -121,8 +123,9 @@ The default classifier wraps `@typesafe-ai/sdk` and asks one `choice` question f 3. **No silent fallback.** Missing configuration is a typed, successful unavailable result; API/runtime failures are errors. 4. **Deterministic output after classification.** Jev is a decision model, not a prose generator. Keeping source content also makes the artifact auditable. 5. **One injectable classifier boundary.** Tests avoid network access and the SDK can be replaced without changing the builder or CLI. -6. **Sequential classification for MVP.** It is simple, deterministic, and avoids unverified API batch/concurrency limits. Measurement can justify later bounded concurrency. +6. **Bounded classification concurrency.** Eight workers overlap independent Jev calls; each result is written to its source index so deterministic artifact ordering is preserved. A bounded pool avoids the request spike of unbounded `Promise.all`. 7. **Redact before Jev and exclude Jev-sensitive events.** This reduces exposure and prevents sensitive events from entering the artifact, while acknowledging regex redaction is not comprehensive DLP. +8. **Direct lookup is adapter-wide and additive.** All built-in adapters implement exact-ID lookup using their native filesystem or database shape. The interface method is optional so external adapters continue to work through manager-level list-and-filter fallback. ### Rejected alternatives diff --git a/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md b/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md index ad7d0d10..c7576c04 100644 --- a/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md @@ -30,7 +30,7 @@ description: Technical implementation notes, patterns, and code guidelines ### Core Features -- Messages are locally redacted, classified sequentially, filtered, and grouped without generative rewriting. +- Messages are locally redacted, classified by an eight-worker bounded pool, filtered, and grouped without generative rewriting. Indexed result placement preserves source order. - Jev asks typed category/importance choice questions and retention/sensitivity noul questions. - Unknown category or importance labels throw explicit integration errors. @@ -50,7 +50,8 @@ description: Technical implementation notes, patterns, and code guidelines - CLI wiring will pass adapter-normalized `ConversationMessage[]` to `compactSession`. - `agent session compact --id [--type ] [--format markdown|json]` is registered beside historical session detail. - The missing-key branch returns before `createAgentManager`, so it cannot list sessions or read a transcript. -- Successful resolution reuses `resolveListSessionsOptions`, `findSessionById`, adapter lookup, and `getConversation(..., { verbose: true })`. +- Successful resolution calls `AgentManager.findSessionsById`. Every built-in adapter implements provider-native lookup; optional adapter-method semantics preserve list-and-filter compatibility for external adapters. +- Codex, Claude, Grok, Copilot, Pi, and OpenCode exploit ID-addressable paths or SQL. Gemini metadata does not encode IDs in filenames, so it scans until the exact embedded ID is found without building summaries for later files. ## Error Handling @@ -64,7 +65,7 @@ description: Technical implementation notes, patterns, and code guidelines **How do we keep it fast?** -- MVP classification is sequential and uncached; measure before adding concurrency or caching. +- Classification defaults to eight concurrent requests and accepts an internal concurrency override for deterministic tests or future tuning. - Each source message produces exactly one Jev request. ## Security Notes @@ -84,4 +85,4 @@ description: Technical implementation notes, patterns, and code guidelines ## Design Alignment -The implementation matches the reviewed design with no material deviations. The only review-driven addition was stricter validation of Jev probabilities and blank model configuration. No provider/session discovery code, database state, automatic memory/task mutation, fallback summarizer, or output-file behavior was added. +The implementation now includes the measured performance follow-up: adapter-wide exact-ID lookup and bounded concurrent classification. It adds no persistent index/database state, automatic memory/task mutation, fallback summarizer, or output-file behavior. diff --git a/docs/ai/planning/2026-09-22-feature-session-compact-jev.md b/docs/ai/planning/2026-09-22-feature-session-compact-jev.md index 471a3b41..0066c742 100644 --- a/docs/ai/planning/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/planning/2026-09-22-feature-session-compact-jev.md @@ -11,6 +11,7 @@ description: Ordered implementation and validation tasks for session compaction - [x] Milestone 1: Typed compaction domain and Jev adapter are covered by unit tests. - [x] Milestone 2: `agent session compact` implements explicit unavailable and successful output flows. - [x] Milestone 3: Built-in skill, lifecycle docs, and repository verification are complete. +- [x] Milestone 4: Measured performance bottlenecks are addressed with adapter-wide direct lookup and bounded Jev concurrency. ## Task Breakdown @@ -64,6 +65,17 @@ description: Ordered implementation and validation tasks for session compaction - Dependencies: all preceding tasks. - Evidence: fresh command outputs recorded in testing docs and durable task. +### Phase 4: Measured compaction performance + +- [x] Task 4.1: Benchmark a representative Codex session and isolate lookup, parsing, classification, and rendering costs. + - Outcome: full historical enumeration and sequential remote classification identified as the dominant avoidable costs. +- [x] Task 4.2: Add exact-ID lookup to `AgentManager` and every built-in adapter, retaining fallback compatibility for external adapters. + - Outcome: the compact command no longer builds every historical session summary before opening one transcript. +- [x] Task 4.3: Replace sequential classification with an order-preserving bounded worker pool using default concurrency eight. + - Outcome: independent Jev round trips overlap without changing artifact order. +- [x] Task 4.4: Add adapter, manager, command, and concurrency regression tests and rerun repository verification. + - Outcome: direct lookup, provider narrowing, ambiguity, fallback compatibility, concurrency bounds, and ordering are executable contracts. + ## Dependencies ```mermaid @@ -94,7 +106,7 @@ flowchart LR ## Risks & Mitigation - SDK/API schema changes: isolate in `JevSessionEventClassifier`, validate labels, pin via lockfile. -- Long sessions cause many sequential calls: accept for MVP, measure before adding concurrency. +- Long sessions can still require many API calls: cap concurrency at eight, preserve ordering, and leave adaptive throttling/batching as a measured follow-up. - Secret leakage: early key gate, local redaction before calls, sensitive-event exclusion, synthetic security tests. - Command namespace confusion: reuse existing `agent session` and document discovery via `agent sessions`. - Command tests are already broad: add focused cases and avoid changing shared behavior. diff --git a/docs/ai/requirements/2026-09-22-feature-session-compact-jev.md b/docs/ai/requirements/2026-09-22-feature-session-compact-jev.md index af1b5ded..b69910fb 100644 --- a/docs/ai/requirements/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/requirements/2026-09-22-feature-session-compact-jev.md @@ -24,7 +24,7 @@ The feature serves developers and agent harnesses handing work to another agent, - No non-Jev summarization fallback. - No automatic memory writes, task mutation, daemon, context-pressure hook, or session mutation. -- No new session-file discovery implementation or general provider abstraction. +- No persistent session index, cache, or provider storage migration. - No `--current`, `--agent`, arbitrary `--input`, or output-file option in the MVP. - No generative model call; retained event content is organized deterministically. @@ -53,6 +53,8 @@ The feature serves developers and agent harnesses handing work to another agent, - With no `TYPESAFE_API_KEY`, the command exits with status 0, does not read the session or call Jev, and prints exactly `Jev is unavailable because TYPESAFE_API_KEY is not set.` in Markdown mode. - JSON unavailable output contains `{ "jev": { "available": false, "reason": "TYPESAFE_API_KEY is not set" } }`. - With a key, each normalized message is evaluated for retention, category, importance, and sensitivity using Jev typed questions. +- Exact ID resolution uses provider-native lookup in every built-in adapter instead of constructing every historical session summary; third-party adapters remain compatible through a list-and-filter fallback. +- Jev classification runs with bounded concurrency (eight requests by default) while preserving source-message order in the artifact. - The result contains intent, current state, decisions, changed files, commands, validation, open questions, next step, memory candidates, resume prompt, and Jev metadata. - Markdown contains the required `# Session Compact` and section headings. - Unit tests cover unavailable behavior, classification mapping, secret redaction/exclusion, rendering, option validation, and command wiring. @@ -62,7 +64,7 @@ The feature serves developers and agent harnesses handing work to another agent, ## Constraints & Assumptions - Node.js 20+ and Commander conventions remain unchanged. -- Historical-session discovery and provider parsing remain owned by `@ai-devkit/agent-manager`; the CLI resolves an ID, then calls the adapter's `getConversation(..., { verbose: true })`. +- Historical-session discovery and provider parsing remain owned by `@ai-devkit/agent-manager`; its adapter contract optionally supports exact-ID lookup, and the CLI then calls the selected adapter's `getConversation(..., { verbose: true })`. - `@typesafe-ai/sdk` is the narrow Jev integration dependency. `TypeSafeClient` reads `TYPESAFE_API_KEY`; production code still performs its own presence check first. - Jev is classification-only. A deterministic builder groups retained source content into the output schema and builds the resume prompt. - Secret redaction is defense in depth, not a complete data-loss-prevention system; users remain responsible for the transcript they send to the hosted service. @@ -70,4 +72,4 @@ The feature serves developers and agent harnesses handing work to another agent, ## Questions & Open Items -All MVP decisions are resolved. Potential follow-ups are `--current`, agent-name resolution, arbitrary normalized input, `--out`, bounded concurrency/batching after measurement, and optional non-Jev compaction under an explicitly different mode. +All current decisions are resolved. Potential follow-ups are `--current`, agent-name resolution, arbitrary normalized input, `--out`, provider-native indexes for stores without ID-addressable paths, adaptive concurrency/rate-limit handling, and optional non-Jev compaction under an explicitly different mode. diff --git a/docs/ai/testing/2026-09-22-feature-session-compact-jev.md b/docs/ai/testing/2026-09-22-feature-session-compact-jev.md index 35b5ed05..a371a5ac 100644 --- a/docs/ai/testing/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/testing/2026-09-22-feature-session-compact-jev.md @@ -24,6 +24,7 @@ description: Coverage for explicit availability, typed classification, rendering - [x] Builds current state and resume prompt deterministically. - [x] Renders every required Markdown heading. - [x] Redacts bearer tokens, private-key blocks, and common secret assignments. +- [x] Bounds concurrent classifications and preserves source order when requests complete out of order. ### Jev classifier @@ -38,7 +39,10 @@ description: Coverage for explicit availability, typed classification, rendering - [x] Missing key prints exact Markdown unavailable text and does not create/list/read a session. - [x] Missing key JSON emits the explicit unavailable object. - [x] Present key resolves ID/type, reads verbose normalized conversation, invokes compaction, and emits selected format. -- [x] Unknown and ambiguous sessions reuse the tested `findSessionById` resolution path and add explicit compact errors. +- [x] Present-key compact wiring uses manager exact-ID lookup rather than full session enumeration. +- [x] Manager exact lookup preserves ambiguity across providers, honors `--type`, and falls back for external adapters without the optional method. +- [x] All seven built-in adapters resolve exact IDs through their provider storage implementation. +- [x] Filesystem-backed direct lookup rejects empty, nested, current-directory, parent-directory, and NUL-containing session IDs through one shared path-segment validator. ### Built-in skill @@ -83,8 +87,12 @@ description: Coverage for explicit availability, typed classification, rendering ## Performance Testing -- Unit-test sequential call count equals the number of normalized messages. -- Record a follow-up if representative long sessions make sequential classification impractical; do not add batching without measured need. +- Representative Codex session `01a0c95e-39c5-7701-a406-cb2578f7b75f`: 750-session enumeration took 3.40–4.88 seconds, while targeted file lookup took 8.88–14.02 milliseconds. +- Its 8.73 MB / 1,659-line transcript normalized to 70 verbose messages; parsing took 19.57–23.06 milliseconds and deterministic assembly/rendering took less than 0.4 milliseconds. +- Default concurrency eight changes the classification critical path from roughly 70 sequential round trips to nine waves while retaining one request per message. +- A deterministic 70-message / 100-millisecond-delay benchmark took 7,078 milliseconds at concurrency one and 911 milliseconds at concurrency eight (7.8x faster). +- The manager-level Codex exact lookup, including summary parsing, took 28.68–38.35 milliseconds across ten runs. +- No live Jev latency was measured because `TYPESAFE_API_KEY` was unavailable; projected post-change time is approximately CLI startup plus direct lookup/parsing plus `ceil(70/8)` Jev latency waves. ## Bug Tracking @@ -93,11 +101,12 @@ description: Coverage for explicit availability, typed classification, rendering ## Validation Results -- CLI package: 98 test files and 1171 tests passed. +- CLI package: 98 test files and 1178 tests passed. - Focused feature coverage: 97.1% statements, 96.77% branches, 90.9% functions, and 98.36% lines across the session-compaction service files. - Workspace: all six lint, build, and test targets completed successfully. Lint retains four unrelated pre-existing warnings in channel/preview code. - Compiled CLI: help lists `agent session compact`; missing-key Markdown and JSON outputs match the required contracts and exit successfully. - Skill: `quick_validate.py skills/session-compact` passed; built-in manifest/fallback tests passed. - Formatting: all nine changed TypeScript files pass `oxfmt --check`. The repository-wide formatter still reports two unrelated pre-existing agent-manager files, which were not modified. - Lifecycle: base and `session-compact-jev` feature lint passed; `git diff --check` passed. +- Performance follow-up focused suites: 8 agent-manager files / 254 tests and 2 CLI files / 100 tests passed; both package typechecks and lints passed (only unrelated existing CLI warnings remain). - Live Jev call: not run because no paid credential is required for automated validation. SDK request construction and response handling are covered with mocked boundary tests. diff --git a/packages/agent-manager/src/AgentManager.ts b/packages/agent-manager/src/AgentManager.ts index 7d5c8ae6..3bbcb072 100644 --- a/packages/agent-manager/src/AgentManager.ts +++ b/packages/agent-manager/src/AgentManager.ts @@ -347,6 +347,40 @@ export class AgentManager { return merged; } + /** Resolve an exact historical session ID across the selected providers. */ + async findSessionsById( + sessionId: string, + opts?: Pick, + ): Promise { + const targetAdapters = Array.from(this.adapters.values()).filter( + (adapter) => opts?.type === undefined || adapter.type === opts.type, + ); + const errors: Array<{ type: string; error: Error }> = []; + + const results = await Promise.all( + targetAdapters.map(async (adapter) => { + try { + if (adapter.findSessionsById) return await adapter.findSessionsById(sessionId); + const sessions = await adapter.listSessions(opts); + return sessions.filter((session) => session.sessionId === sessionId); + } catch (error) { + const err = error instanceof Error ? error : new Error(String(error)); + errors.push({ type: adapter.type, error: err }); + return []; + } + }), + ); + + if (errors.length > 0) { + console.error(`Warning: ${errors.length} adapter(s) failed to find session by ID:`); + for (const { type, error } of errors) { + console.error(` - ${type}: ${error.message}`); + } + } + + return results.flat(); + } + /** * Get count of registered adapters * diff --git a/packages/agent-manager/src/__tests__/AgentManager.test.ts b/packages/agent-manager/src/__tests__/AgentManager.test.ts index dfe72514..c7e4d36d 100644 --- a/packages/agent-manager/src/__tests__/AgentManager.test.ts +++ b/packages/agent-manager/src/__tests__/AgentManager.test.ts @@ -21,6 +21,7 @@ import type { HerdrAgentPane } from "../runtime/herdr/HerdrAgentDiscovery.js"; // Mock adapter for testing class MockAdapter implements AgentAdapter { public lastListSessionsOpts: unknown = undefined; + public findSessionsByIdCalls: string[] = []; constructor( public readonly type: AgentType, @@ -53,6 +54,11 @@ class MockAdapter implements AgentAdapter { return this.mockSessions; } + async findSessionsById(sessionId: string): Promise { + this.findSessionsByIdCalls.push(sessionId); + return this.mockSessions.filter((session) => session.sessionId === sessionId); + } + setAgents(agents: AgentInfo[]): void { this.mockAgents = agents; } @@ -1119,6 +1125,60 @@ describe("AgentManager", () => { }); }); + describe("findSessionsById", () => { + const session = (type: AgentType, sessionId: string): SessionSummary => ({ + type, + sessionId, + cwd: "/repo", + firstUserMessage: "hello", + lastActive: new Date("2025-01-01T00:00:00Z"), + startedAt: new Date("2025-01-01T00:00:00Z"), + sessionFilePath: `/tmp/${sessionId}`, + }); + + it("uses each built-in adapter direct lookup and preserves cross-provider ambiguity", async () => { + const claude = new MockAdapter("claude", [], false, [session("claude", "shared")]); + const codex = new MockAdapter("codex", [], false, [session("codex", "shared")]); + manager.registerAdapter(claude); + manager.registerAdapter(codex); + + const result = await manager.findSessionsById("shared"); + + expect(result).toHaveLength(2); + expect(claude.findSessionsByIdCalls).toEqual(["shared"]); + expect(codex.findSessionsByIdCalls).toEqual(["shared"]); + expect(claude.lastListSessionsOpts).toBeUndefined(); + expect(codex.lastListSessionsOpts).toBeUndefined(); + }); + + it("skips non-matching providers when type is supplied", async () => { + const claude = new MockAdapter("claude", [], false, [session("claude", "target")]); + const codex = new MockAdapter("codex", [], false, [session("codex", "target")]); + manager.registerAdapter(claude); + manager.registerAdapter(codex); + + const result = await manager.findSessionsById("target", { type: "codex" }); + + expect(result).toEqual([expect.objectContaining({ type: "codex", sessionId: "target" })]); + expect(claude.findSessionsByIdCalls).toEqual([]); + expect(codex.findSessionsByIdCalls).toEqual(["target"]); + }); + + it("falls back to filtered listing for external adapters without direct lookup", async () => { + const external = new MockAdapter("other", [], false, [ + session("other", "target"), + session("other", "different"), + ]); + (external as Partial).findSessionsById = undefined; + manager.registerAdapter(external); + + const result = await manager.findSessionsById("target"); + + expect(result).toEqual([expect.objectContaining({ sessionId: "target" })]); + expect(external.lastListSessionsOpts).toEqual(undefined); + }); + }); + describe("resolveAgent", () => { it("should return null for empty input or empty agents list", () => { const agent = createMockAgent({ name: "test-agent" }); diff --git a/packages/agent-manager/src/__tests__/adapters/ClaudeCodeAdapter.test.ts b/packages/agent-manager/src/__tests__/adapters/ClaudeCodeAdapter.test.ts index de18bf57..fcf61abd 100644 --- a/packages/agent-manager/src/__tests__/adapters/ClaudeCodeAdapter.test.ts +++ b/packages/agent-manager/src/__tests__/adapters/ClaudeCodeAdapter.test.ts @@ -1624,6 +1624,29 @@ describe("ClaudeCodeAdapter", () => { expect(cwds).toEqual([cwdA, cwdB]); }); + it("finds an exact session ID without returning other project sessions", async () => { + const wantedPath = writeSession(path.join(projectsDir, "-repo-a"), "wanted", [ + { + type: "user", + timestamp: "2025-01-01T00:00:00Z", + cwd: "/repo-a", + message: { content: "wanted prompt" }, + }, + ]); + writeSession(path.join(projectsDir, "-repo-b"), "other", [ + { + type: "user", + timestamp: "2025-01-01T00:00:00Z", + cwd: "/repo-b", + message: { content: "other prompt" }, + }, + ]); + + await expect(adapter.findSessionsById("wanted")).resolves.toEqual([ + expect.objectContaining({ sessionId: "wanted", sessionFilePath: wantedPath }), + ]); + }); + it("drops sessions whose recorded cwd does not match opts.cwd (strict equality)", async () => { const cwdReal = "/Users/test/foo"; const cwdRequested = "/Users/test/foo/sub"; diff --git a/packages/agent-manager/src/__tests__/adapters/CodexAdapter.test.ts b/packages/agent-manager/src/__tests__/adapters/CodexAdapter.test.ts index 037e5ff1..d6593359 100644 --- a/packages/agent-manager/src/__tests__/adapters/CodexAdapter.test.ts +++ b/packages/agent-manager/src/__tests__/adapters/CodexAdapter.test.ts @@ -339,6 +339,13 @@ describe("CodexAdapter", () => { sessionFilePath: sessionFile, }, ]); + await expect(adapter.findSessionsById("listed")).resolves.toMatchObject([ + { + type: "codex", + sessionId: "listed", + sessionFilePath: sessionFile, + }, + ]); }); function writeSession( diff --git a/packages/agent-manager/src/__tests__/adapters/GrokCliAdapter.test.ts b/packages/agent-manager/src/__tests__/adapters/GrokCliAdapter.test.ts index 8001eba5..ed134550 100644 --- a/packages/agent-manager/src/__tests__/adapters/GrokCliAdapter.test.ts +++ b/packages/agent-manager/src/__tests__/adapters/GrokCliAdapter.test.ts @@ -335,6 +335,15 @@ describe("GrokCliAdapter", () => { ); }); + it("finds an exact session ID across project groups", async () => { + writeSession({}); + writeSession({ sessionCwd: "/Users/dev/other", id: "other-session" }); + + await expect(adapter.findSessionsById(SESSION_ID)).resolves.toEqual([ + expect.objectContaining({ sessionId: SESSION_ID, cwd }), + ]); + }); + it("applies the cwd filter against the decoded cwd", async () => { writeSession({ sessionCwd: "/Users/dev/project-a", diff --git a/packages/agent-manager/src/__tests__/providers/copilot/CopilotAdapter.test.ts b/packages/agent-manager/src/__tests__/providers/copilot/CopilotAdapter.test.ts index c543f8ea..059fa3dd 100644 --- a/packages/agent-manager/src/__tests__/providers/copilot/CopilotAdapter.test.ts +++ b/packages/agent-manager/src/__tests__/providers/copilot/CopilotAdapter.test.ts @@ -552,6 +552,18 @@ describe("CopilotAdapter", () => { }); }); + it("finds an exact session ID from its session directory", async () => { + writeSession("wanted", { events: [sessionStart("wanted", "/repo")] }); + writeSession("other", { events: [sessionStart("other", "/other")] }); + + await expect(adapter.findSessionsById("wanted")).resolves.toEqual([ + expect.objectContaining({ + sessionId: "wanted", + sessionFilePath: path.join(sessionStateDir, "wanted", "events.jsonl"), + }), + ]); + }); + it("applies strict cwd filter", async () => { writeSession("keep", { events: [sessionStart("keep", "/repo")] }); writeSession("drop", { events: [sessionStart("drop", "/other")] }); diff --git a/packages/agent-manager/src/__tests__/providers/gemini/GeminiCliAdapter.test.ts b/packages/agent-manager/src/__tests__/providers/gemini/GeminiCliAdapter.test.ts index 7e1e1d4f..cc2221e1 100644 --- a/packages/agent-manager/src/__tests__/providers/gemini/GeminiCliAdapter.test.ts +++ b/packages/agent-manager/src/__tests__/providers/gemini/GeminiCliAdapter.test.ts @@ -1185,6 +1185,27 @@ describe("GeminiCliAdapter", () => { }); }); + it("finds an exact session ID from Gemini metadata", async () => { + const wantedPath = writeSession(tmpHome, "aaa", "session-one", { + sessionId: "wanted", + projectHash: hashProjectRoot("/repo"), + startTime: "2025-01-01T00:00:00Z", + directories: ["/repo"], + messages: [{ type: "user", content: "wanted prompt" }], + }); + writeSession(tmpHome, "bbb", "session-two", { + sessionId: "other", + projectHash: hashProjectRoot("/other"), + startTime: "2025-01-01T00:00:00Z", + directories: ["/other"], + messages: [{ type: "user", content: "other prompt" }], + }); + + await expect(adapter.findSessionsById("wanted")).resolves.toEqual([ + expect.objectContaining({ sessionId: "wanted", sessionFilePath: wantedPath }), + ]); + }); + it("applies strict-equality cwd filter against directories[0]", async () => { writeSession(tmpHome, "aaa", "session-keep", { sessionId: "keep", diff --git a/packages/agent-manager/src/__tests__/providers/opencode/OpenCodeAdapter.test.ts b/packages/agent-manager/src/__tests__/providers/opencode/OpenCodeAdapter.test.ts index 2300aeef..5af99754 100644 --- a/packages/agent-manager/src/__tests__/providers/opencode/OpenCodeAdapter.test.ts +++ b/packages/agent-manager/src/__tests__/providers/opencode/OpenCodeAdapter.test.ts @@ -121,6 +121,24 @@ describe("OpenCodeAdapter", () => { ]); expect(adapter.getConversation("/not-an-opencode-ref")).toEqual([]); }); + + it("finds an exact session ID with a targeted database query", async () => { + const now = Date.now(); + writeDatabase(dbPath, { + sessions: [ + { id: "wanted", directory: "/repo", timeCreated: now }, + { id: "other", directory: "/other", timeCreated: now - 1_000 }, + ], + }); + + await expect(adapter.findSessionsById("wanted")).resolves.toEqual([ + expect.objectContaining({ + type: "opencode", + sessionId: "wanted", + sessionFilePath: `${dbPath}::wanted`, + }), + ]); + }); }); function writeDatabase( diff --git a/packages/agent-manager/src/__tests__/providers/pi/PiAdapter.test.ts b/packages/agent-manager/src/__tests__/providers/pi/PiAdapter.test.ts index 318b4d47..ce94d76f 100644 --- a/packages/agent-manager/src/__tests__/providers/pi/PiAdapter.test.ts +++ b/packages/agent-manager/src/__tests__/providers/pi/PiAdapter.test.ts @@ -260,6 +260,21 @@ describe("PiAdapter", () => { ]); }); + it("finds an exact session ID from the filename suffix", async () => { + const wanted = writePiSession("/repo/wanted", [ + { timestamp: "2026-06-10T08:58:20.754Z", sessionId: "sess-wanted", cwd: "/repo/wanted" }, + { role: "user", content: "wanted" }, + ]); + writePiSession("/repo/other", [ + { timestamp: "2026-06-10T08:58:20.754Z", sessionId: "sess-other", cwd: "/repo/other" }, + { role: "user", content: "other" }, + ]); + + await expect(adapter.findSessionsById("sess-wanted")).resolves.toEqual([ + expect.objectContaining({ sessionId: "sess-wanted", sessionFilePath: wanted }), + ]); + }); + function makeProcess(overrides: Partial): ProcessInfo { return { pid: 1, diff --git a/packages/agent-manager/src/__tests__/utils/session.test.ts b/packages/agent-manager/src/__tests__/utils/session.test.ts index ab628086..856e224e 100644 --- a/packages/agent-manager/src/__tests__/utils/session.test.ts +++ b/packages/agent-manager/src/__tests__/utils/session.test.ts @@ -5,7 +5,7 @@ import type { MockedFunction } from "vitest"; import * as fs from "fs"; -import { batchGetSessionFileBirthtimes } from "../../utils/session.js"; +import { batchGetSessionFileBirthtimes, isSafePathSegment } from "../../utils/session.js"; vi.mock("fs", () => ({ readdirSync: vi.fn(), @@ -15,6 +15,16 @@ vi.mock("fs", () => ({ const mockedReaddirSync = fs.readdirSync as MockedFunction; const mockedStatSync = fs.statSync as MockedFunction; +describe("isSafePathSegment", () => { + it.each(["", ".", "..", "nested/session", "bad\0id"])("rejects unsafe segment %j", (value) => { + expect(isSafePathSegment(value)).toBe(false); + }); + + it.each(["session-id", ".hidden-session"])("accepts safe segment %j", (value) => { + expect(isSafePathSegment(value)).toBe(true); + }); +}); + describe("batchGetSessionFileBirthtimes", () => { beforeEach(() => { mockedReaddirSync.mockReset(); diff --git a/packages/agent-manager/src/adapters/AgentAdapter.ts b/packages/agent-manager/src/adapters/AgentAdapter.ts index 62796141..5973c3d7 100644 --- a/packages/agent-manager/src/adapters/AgentAdapter.ts +++ b/packages/agent-manager/src/adapters/AgentAdapter.ts @@ -208,4 +208,13 @@ export interface AgentAdapter { * @returns Array of sessions discovered on disk */ listSessions(opts?: ListSessionsOptions): Promise; + + /** + * Resolve exact historical session matches without enumerating every summary. + * + * Built-in adapters implement this using their provider-native storage shape. + * The method remains optional so external adapters can fall back to + * {@link listSessions} until they adopt direct lookup. + */ + findSessionsById?(sessionId: string): Promise; } diff --git a/packages/agent-manager/src/adapters/GrokCliAdapter.ts b/packages/agent-manager/src/adapters/GrokCliAdapter.ts index 1abf0b1e..b47a2a85 100644 --- a/packages/agent-manager/src/adapters/GrokCliAdapter.ts +++ b/packages/agent-manager/src/adapters/GrokCliAdapter.ts @@ -14,7 +14,13 @@ import { executableBasename, filterByProcessNames, } from "../utils/process.js"; -import { isDirectory, safeReadFile, safeReaddir, safeStat } from "../utils/session.js"; +import { + isDirectory, + isSafePathSegment, + safeReadFile, + safeReaddir, + safeStat, +} from "../utils/session.js"; import { generateAgentName } from "../utils/matching.js"; /** @@ -240,21 +246,46 @@ export class GrokCliAdapter implements AgentAdapter { const cwd = session.projectPath || decodedCwd; if (filterCwd !== undefined && cwd !== filterCwd) continue; - summaries.push({ - type: this.type, - sessionId: session.sessionId, - cwd, - firstUserMessage: session.firstUserMessage || "", - lastActive: session.lastActive, - startedAt: session.sessionStart, - sessionFilePath: path.join(sessionDir, CHAT_HISTORY_FILE), - }); + summaries.push(this.toSessionSummary(session, sessionDir, cwd)); } } return summaries; } + async findSessionsById(sessionId: string): Promise { + if (!isSafePathSegment(sessionId) || !isDirectory(this.sessionsDir)) { + return []; + } + + const summaries: SessionSummary[] = []; + for (const groupName of safeReaddir(this.sessionsDir)) { + const groupDir = path.join(this.sessionsDir, groupName); + if (!isDirectory(groupDir)) continue; + + const sessionDir = path.join(groupDir, sessionId); + if (!isDirectory(sessionDir)) continue; + const decodedCwd = this.decodeGroupCwd(groupName, groupDir); + const session = this.readSession(sessionDir, decodedCwd); + if (!session || session.sessionId !== sessionId) continue; + + summaries.push(this.toSessionSummary(session, sessionDir, session.projectPath || decodedCwd)); + } + return summaries; + } + + private toSessionSummary(session: GrokSession, sessionDir: string, cwd: string): SessionSummary { + return { + type: this.type, + sessionId: session.sessionId, + cwd, + firstUserMessage: session.firstUserMessage || "", + lastActive: session.lastActive, + startedAt: session.sessionStart, + sessionFilePath: path.join(sessionDir, CHAT_HISTORY_FILE), + }; + } + // --- Session parsing (chat_history.jsonl) --- /** diff --git a/packages/agent-manager/src/providers/claude/ClaudeCodeAdapter.ts b/packages/agent-manager/src/providers/claude/ClaudeCodeAdapter.ts index a318bf63..903026d8 100644 --- a/packages/agent-manager/src/providers/claude/ClaudeCodeAdapter.ts +++ b/packages/agent-manager/src/providers/claude/ClaudeCodeAdapter.ts @@ -138,31 +138,36 @@ export class ClaudeCodeAdapter implements AgentAdapter { const summaries: SessionSummary[] = []; for (const { filePath, defaultCwd } of candidates) { - const session = this.parser.readSession(filePath, defaultCwd); - if (!session) continue; - - // Drop sessions whose JSONL had no parseable conversation entries. - // readSession is permissive (returns a shell record even when every - // line fails to parse); listSessions needs at least one real entry - // so we don't surface garbage files. - if (!session.lastEntryType) continue; - - const recordedCwd = session.lastCwd || defaultCwd; - if (filterCwd !== undefined && recordedCwd !== filterCwd) continue; - - const stat = safeStat(filePath); - - summaries.push({ - type: "claude", - sessionId: session.sessionId, - cwd: recordedCwd, - firstUserMessage: session.firstUserMessage || "", - lastActive: session.lastActive ?? stat?.mtime ?? new Date(), - startedAt: session.sessionStart ?? stat?.birthtime ?? stat?.mtime ?? new Date(), - sessionFilePath: filePath, - }); + const summary = this.toSessionSummary(filePath, defaultCwd); + if (!summary) continue; + if (filterCwd !== undefined && summary.cwd !== filterCwd) continue; + summaries.push(summary); } return summaries; } + + async findSessionsById(sessionId: string): Promise { + return this.createLocator() + .findHistoricalSessionFilesById(sessionId) + .map(({ filePath, defaultCwd }) => this.toSessionSummary(filePath, defaultCwd)) + .filter((summary): summary is SessionSummary => summary?.sessionId === sessionId); + } + + private toSessionSummary(filePath: string, defaultCwd: string): SessionSummary | null { + const session = this.parser.readSession(filePath, defaultCwd); + if (!session?.lastEntryType) return null; + + const recordedCwd = session.lastCwd || defaultCwd; + const stat = safeStat(filePath); + return { + type: "claude", + sessionId: session.sessionId, + cwd: recordedCwd, + firstUserMessage: session.firstUserMessage || "", + lastActive: session.lastActive ?? stat?.mtime ?? new Date(), + startedAt: session.sessionStart ?? stat?.birthtime ?? stat?.mtime ?? new Date(), + sessionFilePath: filePath, + }; + } } diff --git a/packages/agent-manager/src/providers/claude/ClaudeSessionLocator.ts b/packages/agent-manager/src/providers/claude/ClaudeSessionLocator.ts index dabce236..5e4b791c 100644 --- a/packages/agent-manager/src/providers/claude/ClaudeSessionLocator.ts +++ b/packages/agent-manager/src/providers/claude/ClaudeSessionLocator.ts @@ -6,6 +6,7 @@ import { matchProcessesToSessions, type MatchResult } from "../../utils/matching import { batchGetSessionFileBirthtimes, isDirectory, + isSafePathSegment, listJsonl, safeReaddir, safeStat, @@ -130,6 +131,25 @@ export class ClaudeSessionLocator { return out; } + findHistoricalSessionFilesById( + sessionId: string, + ): Array<{ filePath: string; defaultCwd: string }> { + if (!isSafePathSegment(sessionId) || !isDirectory(this.projectsDir)) { + return []; + } + + const matches: Array<{ filePath: string; defaultCwd: string }> = []; + for (const dirName of safeReaddir(this.projectsDir)) { + const projectDir = path.join(this.projectsDir, dirName); + if (!isDirectory(projectDir)) continue; + + const filePath = path.join(projectDir, `${sessionId}.jsonl`); + if (!safeStat(filePath)?.isFile()) continue; + matches.push({ filePath, defaultCwd: dirName.replace(/-/g, "/") }); + } + return matches; + } + /** * Derive the Claude Code project directory for a given CWD. * diff --git a/packages/agent-manager/src/providers/codex/CodexAdapter.ts b/packages/agent-manager/src/providers/codex/CodexAdapter.ts index ae1cdcbe..ebdacd0c 100644 --- a/packages/agent-manager/src/providers/codex/CodexAdapter.ts +++ b/packages/agent-manager/src/providers/codex/CodexAdapter.ts @@ -95,6 +95,14 @@ export class CodexAdapter implements AgentAdapter { return summaries; } + async findSessionsById(sessionId: string): Promise { + const sessionFile = this.createLocator().findSessionFileById(sessionId); + if (!sessionFile) return []; + + const summary = this.parser.fileToSessionSummary(sessionFile.filePath); + return summary?.sessionId === sessionId ? [summary] : []; + } + private async getCodexProcesses(context?: AgentDetectionContext): Promise { const snapshot = context?.processes ?? (await captureProcessSnapshot(this.processNames)); const relevant = filterByProcessNames(snapshot, this.processNames); diff --git a/packages/agent-manager/src/providers/copilot/CopilotAdapter.ts b/packages/agent-manager/src/providers/copilot/CopilotAdapter.ts index 98962c98..4dd9426e 100644 --- a/packages/agent-manager/src/providers/copilot/CopilotAdapter.ts +++ b/packages/agent-manager/src/providers/copilot/CopilotAdapter.ts @@ -29,7 +29,7 @@ import { import { AgentRegistry, type RegistryEntry } from "../../utils/AgentRegistry.js"; import { CopilotAgentMapper } from "./CopilotAgentMapper.js"; import { CopilotSessionLocator } from "./CopilotSessionLocator.js"; -import { CopilotSessionParser } from "./CopilotSessionParser.js"; +import { CopilotSessionParser, type CopilotSession } from "./CopilotSessionParser.js"; export interface CopilotAdapterOptions { sessionStateDir?: string; @@ -111,20 +111,36 @@ export class CopilotAdapter implements AgentAdapter { if (!session) continue; if (opts?.cwd !== undefined && session.projectPath !== opts.cwd) continue; - summaries.push({ - type: this.type, - sessionId: session.sessionId, - cwd: session.projectPath, - firstUserMessage: session.firstUserMessage, - lastActive: session.lastActive, - startedAt: session.sessionStart, - sessionFilePath: session.eventsFilePath, - }); + summaries.push(this.toSessionSummary(session)); } return summaries; } + async findSessionsById(sessionId: string): Promise { + const match = this.locator.findSessionDirById(sessionId); + if (match) { + const session = this.parser.readSessionDir(match.sessionDir, match.sessionId); + if (session?.sessionId === sessionId) { + return [this.toSessionSummary(session)]; + } + } + + return (await this.listSessions()).filter((session) => session.sessionId === sessionId); + } + + private toSessionSummary(session: CopilotSession): SessionSummary { + return { + type: this.type, + sessionId: session.sessionId, + cwd: session.projectPath, + firstUserMessage: session.firstUserMessage, + lastActive: session.lastActive, + startedAt: session.sessionStart, + sessionFilePath: session.eventsFilePath, + }; + } + private applyWrapperRegistryName( agent: AgentInfo, processInfo: ProcessInfo, diff --git a/packages/agent-manager/src/providers/copilot/CopilotSessionLocator.ts b/packages/agent-manager/src/providers/copilot/CopilotSessionLocator.ts index 02987d44..72d28feb 100644 --- a/packages/agent-manager/src/providers/copilot/CopilotSessionLocator.ts +++ b/packages/agent-manager/src/providers/copilot/CopilotSessionLocator.ts @@ -1,5 +1,5 @@ import * as path from "path"; -import { isDirectory, safeReaddir } from "../../utils/session.js"; +import { isDirectory, isSafePathSegment, safeReaddir } from "../../utils/session.js"; export interface CopilotLock { sessionDir: string; @@ -39,6 +39,12 @@ export class CopilotSessionLocator { return sessionDirs; } + findSessionDirById(sessionId: string): CopilotSessionDir | null { + if (!isSafePathSegment(sessionId)) return null; + const sessionDir = path.join(this.sessionStateDir, sessionId); + return isDirectory(sessionDir) ? { sessionDir, sessionId } : null; + } + discoverActiveLocks(): CopilotLock[] { const locks: CopilotLock[] = []; for (const { sessionDir, sessionId } of this.listSessionDirs()) { diff --git a/packages/agent-manager/src/providers/gemini/GeminiCliAdapter.ts b/packages/agent-manager/src/providers/gemini/GeminiCliAdapter.ts index 37e07899..d5b3b257 100644 --- a/packages/agent-manager/src/providers/gemini/GeminiCliAdapter.ts +++ b/packages/agent-manager/src/providers/gemini/GeminiCliAdapter.ts @@ -108,6 +108,14 @@ export class GeminiCliAdapter implements AgentAdapter { return summaries; } + async findSessionsById(sessionId: string): Promise { + for (const filePath of this.locator.discoverHistoricalSessionFiles()) { + const summary = this.parser.fileToSessionSummary(filePath); + if (summary?.sessionId === sessionId) return [summary]; + } + return []; + } + private async getGeminiProcesses(context?: AgentDetectionContext): Promise { const snapshot = context?.processes ?? (await captureProcessSnapshot(this.processNames)); const relevant = filterByProcessNames(snapshot, this.processNames); diff --git a/packages/agent-manager/src/providers/opencode/OpenCodeAdapter.ts b/packages/agent-manager/src/providers/opencode/OpenCodeAdapter.ts index ed4dfd36..e1a981c5 100644 --- a/packages/agent-manager/src/providers/opencode/OpenCodeAdapter.ts +++ b/packages/agent-manager/src/providers/opencode/OpenCodeAdapter.ts @@ -109,4 +109,8 @@ export class OpenCodeAdapter implements AgentAdapter { async listSessions(opts?: ListSessionsOptions): Promise { return this.locator.listSessions(opts); } + + async findSessionsById(sessionId: string): Promise { + return this.locator.findSessionsById(sessionId); + } } diff --git a/packages/agent-manager/src/providers/opencode/OpenCodeSessionLocator.ts b/packages/agent-manager/src/providers/opencode/OpenCodeSessionLocator.ts index bda123e1..d73a8704 100644 --- a/packages/agent-manager/src/providers/opencode/OpenCodeSessionLocator.ts +++ b/packages/agent-manager/src/providers/opencode/OpenCodeSessionLocator.ts @@ -16,6 +16,12 @@ export interface OpenCodeSession { timeCreated: number; } +interface OpenCodeSessionRow { + id: string; + directory: string; + timeCreated: number; +} + export class OpenCodeSessionLocator { private readonly dbPath: string; private db: Database.Database | null = null; @@ -82,7 +88,7 @@ export class OpenCodeSessionLocator { try { const rows = db - .prepare<[], { id: string; directory: string; timeCreated: number }>(` + .prepare<[], OpenCodeSessionRow>(` SELECT id, directory, time_created AS timeCreated FROM session ORDER BY time_created DESC @@ -94,20 +100,7 @@ export class OpenCodeSessionLocator { for (const row of rows) { if (opts?.cwd !== undefined && row.directory !== opts.cwd) continue; - const stats = this.parser.getSessionStats(db, row.id); - const lastActive = - stats.lastTimeUpdated > 0 ? new Date(stats.lastTimeUpdated) : new Date(row.timeCreated); - const startedAt = new Date(row.timeCreated); - - summaries.push({ - type: "opencode", - sessionId: row.id, - cwd: row.directory, - firstUserMessage: stats.summary, - lastActive, - startedAt, - sessionFilePath: encodeOpenCodeSessionRef(this.dbPath, row.id), - }); + summaries.push(this.toSessionSummary(db, row)); } return summaries; @@ -117,6 +110,41 @@ export class OpenCodeSessionLocator { } } + findSessionsById(sessionId: string): SessionSummary[] { + const db = this.openDb(); + if (!db) return []; + + try { + const row = db + .prepare<[string], OpenCodeSessionRow>(` + SELECT id, directory, time_created AS timeCreated + FROM session + WHERE id = ? + `) + .get(sessionId); + return row ? [this.toSessionSummary(db, row)] : []; + } catch { + this.close(); + return []; + } + } + + private toSessionSummary(db: Database.Database, row: OpenCodeSessionRow): SessionSummary { + const stats = this.parser.getSessionStats(db, row.id); + const lastActive = + stats.lastTimeUpdated > 0 ? new Date(stats.lastTimeUpdated) : new Date(row.timeCreated); + + return { + type: "opencode", + sessionId: row.id, + cwd: row.directory, + firstUserMessage: stats.summary, + lastActive, + startedAt: new Date(row.timeCreated), + sessionFilePath: encodeOpenCodeSessionRef(this.dbPath, row.id), + }; + } + get dbFilePath(): string { return this.dbPath; } diff --git a/packages/agent-manager/src/providers/pi/PiAdapter.ts b/packages/agent-manager/src/providers/pi/PiAdapter.ts index 47bb665d..813c4a92 100644 --- a/packages/agent-manager/src/providers/pi/PiAdapter.ts +++ b/packages/agent-manager/src/providers/pi/PiAdapter.ts @@ -92,6 +92,18 @@ export class PiAdapter implements AgentAdapter { return summaries; } + async findSessionsById(sessionId: string): Promise { + const summaries: SessionSummary[] = []; + const filePaths = this.createLocator().findHistoricalSessionFilesById(sessionId); + for (const filePath of filePaths) { + const summary = this.parser.fileToSessionSummary(filePath); + if (summary?.sessionId === sessionId) summaries.push(summary); + } + + if (summaries.length > 0) return summaries; + return (await this.listSessions()).filter((summary) => summary.sessionId === sessionId); + } + private async getPiProcesses(context?: AgentDetectionContext): Promise { const snapshot = context?.processes ?? (await captureProcessSnapshot(this.processNames)); const relevant = filterByProcessNames(snapshot, this.processNames); diff --git a/packages/agent-manager/src/providers/pi/PiSessionLocator.ts b/packages/agent-manager/src/providers/pi/PiSessionLocator.ts index 8da9e4e7..3049b117 100644 --- a/packages/agent-manager/src/providers/pi/PiSessionLocator.ts +++ b/packages/agent-manager/src/providers/pi/PiSessionLocator.ts @@ -1,7 +1,13 @@ import * as path from "path"; import type { ProcessInfo } from "../../adapters/AgentAdapter.js"; import { matchProcessesToSessions, type MatchResult } from "../../utils/matching.js"; -import { isDirectory, safeReaddir, safeStat, type SessionFile } from "../../utils/session.js"; +import { + isDirectory, + isSafePathSegment, + safeReaddir, + safeStat, + type SessionFile, +} from "../../utils/session.js"; import { PiSessionParser } from "./PiSessionParser.js"; export interface PiSessionLocatorOptions { @@ -64,6 +70,15 @@ export class PiSessionLocator { return this.collectJsonlFiles(this.sessionsDir); } + findHistoricalSessionFilesById(sessionId: string): string[] { + if (!isSafePathSegment(sessionId)) return []; + + return this.collectJsonlFiles(this.sessionsDir).filter((filePath) => { + const basename = path.basename(filePath, ".jsonl"); + return basename === sessionId || basename.endsWith(`_${sessionId}`); + }); + } + private buildProjectDirCwdMap(processes: ProcessInfo[]): Map { const map = new Map(); for (const proc of processes) { diff --git a/packages/agent-manager/src/utils/session.ts b/packages/agent-manager/src/utils/session.ts index 4d65a5df..9ec1634c 100644 --- a/packages/agent-manager/src/utils/session.ts +++ b/packages/agent-manager/src/utils/session.ts @@ -36,6 +36,17 @@ export function isDirectory(p: string): boolean { return safeStat(p)?.isDirectory() ?? false; } +/** Return whether a value can be safely appended as one filesystem path segment. */ +export function isSafePathSegment(value: string): boolean { + return ( + value.length > 0 && + value !== "." && + value !== ".." && + !value.includes("\0") && + path.basename(value) === value + ); +} + /** * `fs.statSync` that swallows errors and returns `undefined` on failure. * Callers can pull whichever fields they need (mtime, birthtime, ...). diff --git a/packages/cli/src/__tests__/commands/agent.test.ts b/packages/cli/src/__tests__/commands/agent.test.ts index 25d98ded..a7f028d9 100644 --- a/packages/cli/src/__tests__/commands/agent.test.ts +++ b/packages/cli/src/__tests__/commands/agent.test.ts @@ -16,6 +16,7 @@ const mockManager: any = { registerAdapter: vi.fn(), listAgents: vi.fn(), listSessions: vi.fn(), + findSessionsById: vi.fn(), resolveAgent: vi.fn(), getAdapter: vi.fn(), }; @@ -391,6 +392,7 @@ describe("agent command", () => { mockManager.registerAdapter.mockReset(); mockManager.listAgents.mockReset(); mockManager.listSessions.mockReset(); + mockManager.findSessionsById.mockReset(); mockManager.resolveAgent.mockReset(); mockManager.getAdapter.mockReset(); mockAgentAdapter.getConversation.mockReset(); @@ -2822,6 +2824,7 @@ Waiting on user input`, "Jev is unavailable because TYPESAFE_API_KEY is not set.", ); expect(mockManager.listSessions).not.toHaveBeenCalled(); + expect(mockManager.findSessionsById).not.toHaveBeenCalled(); expect(mockCreateJevClassifier).not.toHaveBeenCalled(); expect(process.exit).not.toHaveBeenCalled(); }); @@ -2847,6 +2850,7 @@ Waiting on user input`, jev: { available: false, reason: "TYPESAFE_API_KEY is not set" }, }); expect(mockManager.listSessions).not.toHaveBeenCalled(); + expect(mockManager.findSessionsById).not.toHaveBeenCalled(); }); it("resolves the historical session and renders a Jev compact", async () => { @@ -2866,7 +2870,7 @@ Waiting on user input`, resumePrompt: "review", jev: { available: true, model: "jev-test", classifiedEvents: 1 }, }; - mockManager.listSessions.mockResolvedValue([session]); + mockManager.findSessionsById.mockResolvedValue([session]); mockManager.getAdapter.mockReturnValue(mockAgentAdapter); mockAgentAdapter.getConversation.mockReturnValue(messages); mockCreateJevClassifier.mockReturnValue(classifier); @@ -2887,7 +2891,9 @@ Waiting on user input`, "codex", ]); - expect(mockManager.listSessions).toHaveBeenCalledWith({ cwd: undefined, type: "codex" }); + expect(mockManager.findSessionsById).toHaveBeenCalledWith("sess-compact", { + type: "codex", + }); expect(mockAgentAdapter.getConversation).toHaveBeenCalledWith("/tmp/sess-compact.jsonl", { verbose: true, }); @@ -2912,7 +2918,7 @@ Waiting on user input`, resumePrompt: "review", jev: { available: true, model: "jev-test", classifiedEvents: 1 }, }; - mockManager.listSessions.mockResolvedValue([session]); + mockManager.findSessionsById.mockResolvedValue([session]); mockManager.getAdapter.mockReturnValue(mockAgentAdapter); mockAgentAdapter.getConversation.mockReturnValue([]); mockCreateJevClassifier.mockReturnValue(classifier); @@ -2957,6 +2963,7 @@ Waiting on user input`, "Failed to compact session: Invalid --format. Expected markdown or json.", ); expect(mockManager.listSessions).not.toHaveBeenCalled(); + expect(mockManager.findSessionsById).not.toHaveBeenCalled(); expect(process.exit).toHaveBeenCalledWith(1); }); }); diff --git a/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts b/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts index cc25c5a4..a79db8c5 100644 --- a/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts +++ b/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts @@ -105,6 +105,54 @@ describe("session compaction", () => { expect(result.currentState).toBe(""); }); + it("classifies with bounded concurrency while preserving source order", async () => { + const messages: ConversationMessage[] = Array.from({ length: 6 }, (_, index) => ({ + role: "assistant", + content: `message-${index}`, + })); + let active = 0; + let maxActive = 0; + const boundedClassifier: SessionEventClassifier = { + model: "jev-test", + classify: vi.fn(async (message) => { + active += 1; + maxActive = Math.max(maxActive, active); + const index = Number(message.content.split("-").at(-1)); + await new Promise((resolve) => setTimeout(resolve, (6 - index) * 2)); + active -= 1; + return event("decision", message.content); + }), + }; + + const result = await compactSession(messages, boundedClassifier, { concurrency: 2 }); + + expect(maxActive).toBe(2); + expect(result.decisions).toEqual(messages.map((message) => message.content)); + }); + + it("defaults to eight concurrent classifications", async () => { + const messages: ConversationMessage[] = Array.from({ length: 10 }, (_, index) => ({ + role: "assistant", + content: `message-${index}`, + })); + let active = 0; + let maxActive = 0; + const boundedClassifier: SessionEventClassifier = { + model: "jev-test", + classify: vi.fn(async (message) => { + active += 1; + maxActive = Math.max(maxActive, active); + await new Promise((resolve) => setTimeout(resolve, 2)); + active -= 1; + return event("decision", message.content); + }), + }; + + await compactSession(messages, boundedClassifier); + + expect(maxActive).toBe(8); + }); + it("redacts common credentials before classification", () => { const content = [ "Authorization: Bearer abc.def.ghi", diff --git a/packages/cli/src/commands/agent.ts b/packages/cli/src/commands/agent.ts index 498aee3a..4c8f21ea 100644 --- a/packages/cli/src/commands/agent.ts +++ b/packages/cli/src/commands/agent.ts @@ -648,21 +648,16 @@ export function registerAgentCommand(program: Command): void { } const manager = createAgentManager(); - const listOptions = resolveListSessionsOptions({ - all: true, - type: options.type, - }).adapterOptions; - const sessions = await manager.listSessions(listOptions); - const resolved = findSessionById(sessions, options.id); - - if (!resolved) { + const matches = await manager.findSessionsById(options.id, { type: options.type }); + if (matches.length === 0) { throw new Error(`No session found matching "${options.id}".`); } - if (Array.isArray(resolved)) { + if (matches.length > 1) { throw new Error( `Multiple sessions match "${options.id}". Use --type to choose the intended session source.`, ); } + const resolved = matches[0]; const adapter = manager.getAdapter(resolved.type); if (!adapter) throw new Error(`Unsupported agent type: ${resolved.type}`); diff --git a/packages/cli/src/services/session-compact/session-compact.service.ts b/packages/cli/src/services/session-compact/session-compact.service.ts index 39fb721c..aa5de1c3 100644 --- a/packages/cli/src/services/session-compact/session-compact.service.ts +++ b/packages/cli/src/services/session-compact/session-compact.service.ts @@ -6,6 +6,12 @@ export interface SessionEventClassifier { classify(message: ConversationMessage): Promise; } +export interface CompactSessionOptions { + concurrency?: number; +} + +export const DEFAULT_SESSION_COMPACT_CONCURRENCY = 8; + const PRIVATE_KEY_PATTERN = /-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g; const BEARER_PATTERN = /(Authorization\s*:\s*Bearer\s+)[^\s]+/gi; @@ -46,14 +52,29 @@ function buildResumePrompt(intent: string, currentState: string, nextStep: strin export async function compactSession( messages: ConversationMessage[], classifier: SessionEventClassifier, + options: CompactSessionOptions = {}, ): Promise { - const classified: ClassifiedSessionEvent[] = []; - for (const message of messages) { - classified.push( - await classifier.classify({ ...message, content: redactSensitiveText(message.content) }), - ); + const concurrency = options.concurrency ?? DEFAULT_SESSION_COMPACT_CONCURRENCY; + if (!Number.isInteger(concurrency) || concurrency < 1) { + throw new Error("Session compact concurrency must be a positive integer."); } + const classified: ClassifiedSessionEvent[] = []; + let nextIndex = 0; + const classifyNext = async (): Promise => { + while (nextIndex < messages.length) { + const index = nextIndex; + nextIndex += 1; + const message = messages[index]; + classified[index] = await classifier.classify({ + ...message, + content: redactSensitiveText(message.content), + }); + } + }; + const workerCount = Math.min(concurrency, messages.length); + await Promise.all(Array.from({ length: workerCount }, () => classifyNext())); + const retained = classified.filter(includeEvent); const intent = list(retained, "user_instruction")[0] ?? ""; const decisions = list(retained, "decision"); From fe595e763ce747249cf85be75c99c68852eb5852 Mon Sep 17 00:00:00 2001 From: Hoang Nguyen Date: Wed, 23 Sep 2026 20:44:52 +0200 Subject: [PATCH 3/3] refactor(session): simplify compact classification --- .../2026-09-22-feature-session-compact-jev.md | 2 - .../2026-09-22-feature-session-compact-jev.md | 2 +- .../2026-09-22-feature-session-compact-jev.md | 4 +- .../session-compact/jev-classifier.test.ts | 7 ++- .../session-compact.service.test.ts | 50 ++++++------------- .../session-compact/jev-classifier.ts | 7 ++- .../session-compact.service.ts | 16 ++---- 7 files changed, 27 insertions(+), 61 deletions(-) diff --git a/docs/ai/design/2026-09-22-feature-session-compact-jev.md b/docs/ai/design/2026-09-22-feature-session-compact-jev.md index b979f491..07e8431d 100644 --- a/docs/ai/design/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/design/2026-09-22-feature-session-compact-jev.md @@ -96,10 +96,8 @@ interface SessionEventClassifier { async function compactSession( messages: ConversationMessage[], classifier: SessionEventClassifier, - options?: { concurrency?: number }, ): Promise; -function redactSensitiveText(content: string): string; function renderSessionCompactMarkdown(result: SessionCompact): string; ``` diff --git a/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md b/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md index c7576c04..ca1cdb19 100644 --- a/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/implementation/2026-09-22-feature-session-compact-jev.md @@ -65,7 +65,7 @@ description: Technical implementation notes, patterns, and code guidelines **How do we keep it fast?** -- Classification defaults to eight concurrent requests and accepts an internal concurrency override for deterministic tests or future tuning. +- Classification uses eight concurrent requests; the fixed internal limit keeps the public service contract small. - Each source message produces exactly one Jev request. ## Security Notes diff --git a/docs/ai/testing/2026-09-22-feature-session-compact-jev.md b/docs/ai/testing/2026-09-22-feature-session-compact-jev.md index a371a5ac..3e26a40d 100644 --- a/docs/ai/testing/2026-09-22-feature-session-compact-jev.md +++ b/docs/ai/testing/2026-09-22-feature-session-compact-jev.md @@ -101,12 +101,12 @@ description: Coverage for explicit availability, typed classification, rendering ## Validation Results -- CLI package: 98 test files and 1178 tests passed. +- CLI package: 98 test files and 1177 tests passed. - Focused feature coverage: 97.1% statements, 96.77% branches, 90.9% functions, and 98.36% lines across the session-compaction service files. - Workspace: all six lint, build, and test targets completed successfully. Lint retains four unrelated pre-existing warnings in channel/preview code. - Compiled CLI: help lists `agent session compact`; missing-key Markdown and JSON outputs match the required contracts and exit successfully. - Skill: `quick_validate.py skills/session-compact` passed; built-in manifest/fallback tests passed. - Formatting: all nine changed TypeScript files pass `oxfmt --check`. The repository-wide formatter still reports two unrelated pre-existing agent-manager files, which were not modified. - Lifecycle: base and `session-compact-jev` feature lint passed; `git diff --check` passed. -- Performance follow-up focused suites: 8 agent-manager files / 254 tests and 2 CLI files / 100 tests passed; both package typechecks and lints passed (only unrelated existing CLI warnings remain). +- Performance follow-up focused suites: 8 agent-manager files / 254 tests and 2 CLI files / 99 tests passed; both package typechecks and lints passed (only unrelated existing CLI warnings remain). - Live Jev call: not run because no paid credential is required for automated validation. SDK request construction and response handling are covered with mocked boundary tests. diff --git a/packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts b/packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts index d80c0e5a..193d56b4 100644 --- a/packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts +++ b/packages/cli/src/__tests__/services/session-compact/jev-classifier.test.ts @@ -31,7 +31,7 @@ describe("JevSessionEventClassifier", () => { const classifier = new JevSessionEventClassifier({ systemOne }, "jev-latest"); const message: ConversationMessage = { role: "assistant", - content: "Authorization: Bearer secret-token\nChanged agent.ts", + content: "Changed agent.ts", timestamp: "2026-09-22T00:00:00.000Z", }; @@ -39,7 +39,7 @@ describe("JevSessionEventClassifier", () => { expect(result).toEqual({ role: "assistant", - content: "Authorization: Bearer [REDACTED]\nChanged agent.ts", + content: "Changed agent.ts", timestamp: "2026-09-22T00:00:00.000Z", category: "code_change", importance: "critical", @@ -51,7 +51,7 @@ describe("JevSessionEventClassifier", () => { expect(request.model).toBe("jev-latest"); expect(request.state).toEqual({ role: "assistant", - content: "Authorization: Bearer [REDACTED]\nChanged agent.ts", + content: "Changed agent.ts", timestamp: "2026-09-22T00:00:00.000Z", }); expect(Object.keys(request.questions)).toEqual(["category", "importance", "keep", "sensitive"]); @@ -59,7 +59,6 @@ describe("JevSessionEventClassifier", () => { expect(request.questions.importance.type).toBe("choice"); expect(request.questions.keep.type).toBe("noul"); expect(request.questions.sensitive.type).toBe("noul"); - expect(JSON.stringify(request)).not.toContain("secret-token"); }); it.each([ diff --git a/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts b/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts index a79db8c5..bee4db2d 100644 --- a/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts +++ b/packages/cli/src/__tests__/services/session-compact/session-compact.service.test.ts @@ -1,7 +1,6 @@ import type { ConversationMessage } from "@ai-devkit/agent-manager"; import { compactSession, - redactSensitiveText, renderSessionCompactMarkdown, type SessionEventClassifier, } from "../../../services/session-compact/session-compact.service.js"; @@ -105,32 +104,7 @@ describe("session compaction", () => { expect(result.currentState).toBe(""); }); - it("classifies with bounded concurrency while preserving source order", async () => { - const messages: ConversationMessage[] = Array.from({ length: 6 }, (_, index) => ({ - role: "assistant", - content: `message-${index}`, - })); - let active = 0; - let maxActive = 0; - const boundedClassifier: SessionEventClassifier = { - model: "jev-test", - classify: vi.fn(async (message) => { - active += 1; - maxActive = Math.max(maxActive, active); - const index = Number(message.content.split("-").at(-1)); - await new Promise((resolve) => setTimeout(resolve, (6 - index) * 2)); - active -= 1; - return event("decision", message.content); - }), - }; - - const result = await compactSession(messages, boundedClassifier, { concurrency: 2 }); - - expect(maxActive).toBe(2); - expect(result.decisions).toEqual(messages.map((message) => message.content)); - }); - - it("defaults to eight concurrent classifications", async () => { + it("classifies eight messages concurrently while preserving source order", async () => { const messages: ConversationMessage[] = Array.from({ length: 10 }, (_, index) => ({ role: "assistant", content: `message-${index}`, @@ -142,30 +116,36 @@ describe("session compaction", () => { classify: vi.fn(async (message) => { active += 1; maxActive = Math.max(maxActive, active); - await new Promise((resolve) => setTimeout(resolve, 2)); + const index = Number(message.content.split("-").at(-1)); + await new Promise((resolve) => setTimeout(resolve, (10 - index) * 2)); active -= 1; return event("decision", message.content); }), }; - await compactSession(messages, boundedClassifier); + const result = await compactSession(messages, boundedClassifier); expect(maxActive).toBe(8); + expect(result.decisions).toEqual(messages.map((message) => message.content)); }); - it("redacts common credentials before classification", () => { + it("redacts common credentials before classification", async () => { const content = [ "Authorization: Bearer abc.def.ghi", "TYPESAFE_API_KEY=jv_live_example", "-----BEGIN PRIVATE KEY-----\nprivate-material\n-----END PRIVATE KEY-----", ].join("\n"); + const classify = vi.fn(async (message: ConversationMessage) => + event("decision", message.content), + ); - const redacted = redactSensitiveText(content); + await compactSession([{ role: "assistant", content }], { model: "jev-test", classify }); - expect(redacted).not.toContain("abc.def.ghi"); - expect(redacted).not.toContain("jv_live_example"); - expect(redacted).not.toContain("private-material"); - expect(redacted.match(/\[REDACTED\]/g)?.length).toBeGreaterThanOrEqual(3); + const classifiedContent = classify.mock.calls[0][0].content; + expect(classifiedContent).not.toContain("abc.def.ghi"); + expect(classifiedContent).not.toContain("jv_live_example"); + expect(classifiedContent).not.toContain("private-material"); + expect(classifiedContent.match(/\[REDACTED\]/g)?.length).toBeGreaterThanOrEqual(3); }); it("renders all required Markdown sections", async () => { diff --git a/packages/cli/src/services/session-compact/jev-classifier.ts b/packages/cli/src/services/session-compact/jev-classifier.ts index b4220a16..cebd3bc7 100644 --- a/packages/cli/src/services/session-compact/jev-classifier.ts +++ b/packages/cli/src/services/session-compact/jev-classifier.ts @@ -1,6 +1,6 @@ import type { ConversationMessage } from "@ai-devkit/agent-manager"; import { choice, noul, TypeSafeClient, type SystemOneResult } from "@typesafe-ai/sdk"; -import { redactSensitiveText, type SessionEventClassifier } from "./session-compact.service.js"; +import type { SessionEventClassifier } from "./session-compact.service.js"; import { COMPACT_CATEGORIES, COMPACT_IMPORTANCE, @@ -60,10 +60,9 @@ export class JevSessionEventClassifier implements SessionEventClassifier { ) {} async classify(message: ConversationMessage): Promise { - const redactedMessage = { ...message, content: redactSensitiveText(message.content) }; const response = await this.client.systemOne({ model: this.model, - state: redactedMessage, + state: message, questions: classificationQuestions, }); @@ -77,7 +76,7 @@ export class JevSessionEventClassifier implements SessionEventClassifier { const sensitive = requireProbability("sensitive", response.answers.sensitive.noul); return { - ...redactedMessage, + ...message, category, importance, keep: keep >= 0.5, diff --git a/packages/cli/src/services/session-compact/session-compact.service.ts b/packages/cli/src/services/session-compact/session-compact.service.ts index aa5de1c3..6392428e 100644 --- a/packages/cli/src/services/session-compact/session-compact.service.ts +++ b/packages/cli/src/services/session-compact/session-compact.service.ts @@ -6,11 +6,7 @@ export interface SessionEventClassifier { classify(message: ConversationMessage): Promise; } -export interface CompactSessionOptions { - concurrency?: number; -} - -export const DEFAULT_SESSION_COMPACT_CONCURRENCY = 8; +const SESSION_COMPACT_CONCURRENCY = 8; const PRIVATE_KEY_PATTERN = /-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g; @@ -18,7 +14,7 @@ const BEARER_PATTERN = /(Authorization\s*:\s*Bearer\s+)[^\s]+/gi; const SECRET_ASSIGNMENT_PATTERN = /\b([A-Z][A-Z0-9_]*(?:API_KEY|TOKEN|SECRET|PASSWORD|PRIVATE_KEY))\s*=\s*([^\s]+)/g; -export function redactSensitiveText(content: string): string { +function redactSensitiveText(content: string): string { return content .replace(PRIVATE_KEY_PATTERN, "[REDACTED]") .replace(BEARER_PATTERN, "$1[REDACTED]") @@ -52,13 +48,7 @@ function buildResumePrompt(intent: string, currentState: string, nextStep: strin export async function compactSession( messages: ConversationMessage[], classifier: SessionEventClassifier, - options: CompactSessionOptions = {}, ): Promise { - const concurrency = options.concurrency ?? DEFAULT_SESSION_COMPACT_CONCURRENCY; - if (!Number.isInteger(concurrency) || concurrency < 1) { - throw new Error("Session compact concurrency must be a positive integer."); - } - const classified: ClassifiedSessionEvent[] = []; let nextIndex = 0; const classifyNext = async (): Promise => { @@ -72,7 +62,7 @@ export async function compactSession( }); } }; - const workerCount = Math.min(concurrency, messages.length); + const workerCount = Math.min(SESSION_COMPACT_CONCURRENCY, messages.length); await Promise.all(Array.from({ length: workerCount }, () => classifyNext())); const retained = classified.filter(includeEvent);