mirror of
https://github.com/dtzp555-max/ocp.git
synced 2026-07-21 21:15:09 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bdb6662c7c | ||
|
|
4f9e2ff281 | ||
|
|
45c5717aea | ||
|
|
47e324b68f | ||
|
|
788cbbcd99 | ||
|
|
ac81badda1 | ||
|
|
fe12419386 | ||
|
|
17038b56e6 | ||
|
|
73314e6698 | ||
|
|
e6f1a6aac1 | ||
|
|
f14f4ec754 | ||
|
|
bafad077ff | ||
|
|
b038d3ceac | ||
|
|
faea02d951 | ||
|
|
6f18613f9d | ||
|
|
0c3e42b2e4 | ||
|
|
0fc8d6973b | ||
|
|
27216646c8 | ||
|
|
d501e786b8 | ||
|
|
b7463a63f5 | ||
|
|
eeec2bf83d | ||
|
|
63c2de7128 | ||
|
|
1d65bc309e | ||
|
|
88d8bed2e3 | ||
|
|
a90f830b5d | ||
|
|
1b324968f4 | ||
|
|
9f5bc3264a | ||
|
|
e7ce9899f3 | ||
|
|
5258d5d395 | ||
|
|
6854075c01 | ||
|
|
45152d58b0 | ||
|
|
d96da46fa0 | ||
|
|
2538233059 | ||
|
|
31e5a44099 | ||
|
|
2922d68842 | ||
|
|
38da104b97 | ||
|
|
5aaab5ea28 | ||
|
|
3bd19956ff | ||
|
|
fe615cb0d3 | ||
|
|
60930f0ba4 | ||
|
|
c86e3d014f | ||
|
|
3322d7bdae | ||
|
|
79c1d61e1d | ||
|
|
a37ff713d9 | ||
|
|
6d4751f983 | ||
|
|
0dced52215 | ||
|
|
d291331998 | ||
|
|
9568411bcb | ||
|
|
1f577c075f | ||
|
|
6dff36959a | ||
|
|
1b02f181fa | ||
|
|
0000926358 | ||
|
|
aa1c65beb1 | ||
|
|
879b40fe93 | ||
|
|
68d58e7df4 | ||
|
|
4a7d79c330 | ||
|
|
c3b1f32c86 | ||
|
|
4458490caa | ||
|
|
36be723198 | ||
|
|
7b065600aa | ||
|
|
1b5a742711 | ||
|
|
05a984df89 | ||
|
|
a30b20978c | ||
|
|
cd98b51b96 | ||
|
|
74260d7f6f | ||
|
|
885f62addf | ||
|
|
1dd6fb9440 | ||
|
|
9e25160527 | ||
|
|
49c6d32e3b | ||
|
|
7766fa0868 | ||
|
|
70faeff067 | ||
|
|
7a69d72886 | ||
|
|
a8601a6d30 | ||
|
|
cd6ec2a212 | ||
|
|
ab03c13332 | ||
|
|
55c576bbb1 | ||
|
|
750b25ba77 | ||
|
|
fd6e875bd7 | ||
|
|
8c0b97f3ae | ||
|
|
68acf15373 | ||
|
|
a71c939bf8 | ||
|
|
d245c62df7 | ||
|
|
047750e642 | ||
|
|
3bdeb50ed5 | ||
|
|
fbbf3b6c7c | ||
|
|
d760d7fcce | ||
|
|
5e2effd05b | ||
|
|
fb2d1d3feb | ||
|
|
12b09c236e | ||
|
|
c0f2d3ab20 | ||
|
|
0d61da5153 | ||
|
|
49baffe2da | ||
|
|
4b01d4e768 | ||
|
|
36fa81d1e6 | ||
|
|
cce0110253 | ||
|
|
40391791a1 | ||
|
|
342a0a44f5 | ||
|
|
9494fd6c69 | ||
|
|
5ff30ac9b6 | ||
|
|
16eeb66557 | ||
|
|
c998d21a4f | ||
|
|
5be369ed68 | ||
|
|
3d52ffc152 | ||
|
|
68b0838074 | ||
|
|
313cb13a78 | ||
|
|
e4b010af5e | ||
|
|
0752f666fb | ||
|
|
ae4a829904 | ||
|
|
51e908e145 | ||
|
|
d99534dc35 | ||
|
|
39ca20536e | ||
|
|
8b3f50912e | ||
|
|
780462c763 | ||
|
|
733a2ed4c2 | ||
|
|
ca2d23d230 | ||
|
|
b871b72b6b | ||
|
|
cff06439fa | ||
|
|
1c29f4867f | ||
|
|
cdd6b41261 | ||
|
|
9facd8307a | ||
|
|
2ecc4945ad | ||
|
|
497ff1dcd2 | ||
|
|
2e634f708b | ||
|
|
820abc4b89 | ||
|
|
ab9e0c656b | ||
|
|
5ef163aa95 | ||
|
|
c6f7850e89 | ||
|
|
43cd7712e6 | ||
|
|
ba273aaf06 | ||
|
|
7cff33cc18 | ||
|
|
d4f4aaf33a | ||
|
|
22806bffb5 | ||
|
|
6bfffd2cba | ||
|
|
2853088261 | ||
|
|
fd7973addb | ||
|
|
b908fec7b4 | ||
|
|
97ca91341c | ||
|
|
f3745fa8fe | ||
|
|
d71e0455e3 | ||
|
|
38070fbabb | ||
|
|
2adea6368d | ||
|
|
7860f71943 | ||
|
|
e326cee9dd | ||
|
|
f8ca9b85b0 | ||
|
|
c22b0dd074 | ||
|
|
7f46405152 | ||
|
|
cb6c2a8b5f | ||
|
|
02c6758a61 | ||
|
|
6d0e43ec37 | ||
|
|
b87992fa3b | ||
|
|
69b20815fa | ||
|
|
3eecca35ce | ||
|
|
9f1a21f7ad | ||
|
|
a0f9268af5 | ||
|
|
087e26346f | ||
|
|
ed53c5e0c5 | ||
|
|
18cb759359 | ||
|
|
cc745aa5d9 | ||
|
|
1983f9372d | ||
|
|
471bda2c40 | ||
|
|
566a01a6bd | ||
|
|
43daf8162c | ||
|
|
aa0adab784 | ||
|
|
4f4e1edf16 | ||
|
|
ea57db6ceb | ||
|
|
5fbeaed568 | ||
|
|
2b18884d2a | ||
|
|
4f72f4844e | ||
|
|
bc17b27f2b | ||
|
|
3e8ff7a509 | ||
|
|
b6b33d2623 | ||
|
|
8796a5fc19 | ||
|
|
609ceb28b0 | ||
|
|
3843ec8fe6 | ||
|
|
afe0c6e1be | ||
|
|
a6007393a3 | ||
|
|
8592150f7a | ||
|
|
eb76971ffc | ||
|
|
d54e73ef89 | ||
|
|
3f10b459f5 | ||
|
|
b72e449337 | ||
|
|
2b05558b8b | ||
|
|
d9b7c0d905 | ||
|
|
27cd8b1735 | ||
|
|
9dd35c90ae | ||
|
|
c005a8fd27 | ||
|
|
77a3f48cfd | ||
|
|
7390a19ec6 | ||
|
|
04643217ae | ||
|
|
3de1868313 | ||
|
|
75cff38f4c | ||
|
|
a8a0af43c9 | ||
|
|
77a6f8ed59 | ||
|
|
b5be1c04f4 | ||
|
|
1f1b3871da | ||
|
|
f6f4f68ecd | ||
|
|
47e39d7820 | ||
|
|
7434f6adc0 | ||
|
|
8a7951134e | ||
|
|
47434a8ba7 | ||
|
|
1d95dbeebc | ||
|
|
9e6bef729b | ||
|
|
256b2bfb77 | ||
|
|
384e66bfa6 | ||
|
|
76a8c56c88 | ||
|
|
cedd2acecc | ||
|
|
9a22372e28 |
@@ -0,0 +1,259 @@
|
||||
---
|
||||
name: speckit-analyze
|
||||
description: Perform a non-destructive cross-artifact consistency and quality analysis across spec.md, plan.md, and tasks.md after task generation.
|
||||
argument-hint: "Optional focus areas for analysis"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/analyze.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before analysis)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_analyze` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Goal.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Goal
|
||||
|
||||
Identify inconsistencies, duplications, ambiguities, and underspecified items across the three core artifacts (`spec.md`, `plan.md`, `tasks.md`) before implementation. This command MUST run only after `/speckit.tasks` has successfully produced a complete `tasks.md`.
|
||||
|
||||
## Operating Constraints
|
||||
|
||||
**STRICTLY READ-ONLY**: Do **not** modify any files. Output a structured analysis report. Offer an optional remediation plan (user must explicitly approve before any follow-up editing commands would be invoked manually).
|
||||
|
||||
**Constitution Authority**: The project constitution (`/memory/constitution.md`) is **non-negotiable** within this analysis scope. Constitution conflicts are automatically CRITICAL and require adjustment of the spec, plan, or tasks—not dilution, reinterpretation, or silent ignoring of the principle. If a principle itself needs to change, that must occur in a separate, explicit constitution update outside `/speckit.analyze`.
|
||||
|
||||
## Execution Steps
|
||||
|
||||
### 1. Initialize Analysis Context
|
||||
|
||||
Run `{SCRIPT}` once from repo root and parse JSON for FEATURE_DIR and AVAILABLE_DOCS. Derive absolute paths:
|
||||
|
||||
- SPEC = FEATURE_DIR/spec.md
|
||||
- PLAN = FEATURE_DIR/plan.md
|
||||
- TASKS = FEATURE_DIR/tasks.md
|
||||
|
||||
Abort with an error message if any required file is missing (instruct the user to run missing prerequisite command).
|
||||
For single quotes in args like "I'm Groot", use escape syntax: e.g 'I'\''m Groot' (or double-quote if possible: "I'm Groot").
|
||||
|
||||
### 2. Load Artifacts (Progressive Disclosure)
|
||||
|
||||
Load only the minimal necessary context from each artifact:
|
||||
|
||||
**From spec.md:**
|
||||
|
||||
- Overview/Context
|
||||
- Functional Requirements
|
||||
- Success Criteria (measurable outcomes — e.g., performance, security, availability, user success, business impact)
|
||||
- User Stories
|
||||
- Edge Cases (if present)
|
||||
|
||||
**From plan.md:**
|
||||
|
||||
- Architecture/stack choices
|
||||
- Data Model references
|
||||
- Phases
|
||||
- Technical constraints
|
||||
|
||||
**From tasks.md:**
|
||||
|
||||
- Task IDs
|
||||
- Descriptions
|
||||
- Phase grouping
|
||||
- Parallel markers [P]
|
||||
- Referenced file paths
|
||||
|
||||
**From constitution:**
|
||||
|
||||
- Load `/memory/constitution.md` for principle validation
|
||||
|
||||
### 3. Build Semantic Models
|
||||
|
||||
Create internal representations (do not include raw artifacts in output):
|
||||
|
||||
- **Requirements inventory**: For each Functional Requirement (FR-###) and Success Criterion (SC-###), record a stable key. Use the explicit FR-/SC- identifier as the primary key when present, and optionally also derive an imperative-phrase slug for readability (e.g., "User can upload file" → `user-can-upload-file`). Include only Success Criteria items that require buildable work (e.g., load-testing infrastructure, security audit tooling), and exclude post-launch outcome metrics and business KPIs (e.g., "Reduce support tickets by 50%").
|
||||
- **User story/action inventory**: Discrete user actions with acceptance criteria
|
||||
- **Task coverage mapping**: Map each task to one or more requirements or stories (inference by keyword / explicit reference patterns like IDs or key phrases)
|
||||
- **Constitution rule set**: Extract principle names and MUST/SHOULD normative statements
|
||||
|
||||
### 4. Detection Passes (Token-Efficient Analysis)
|
||||
|
||||
Focus on high-signal findings. Limit to 50 findings total; aggregate remainder in overflow summary.
|
||||
|
||||
#### A. Duplication Detection
|
||||
|
||||
- Identify near-duplicate requirements
|
||||
- Mark lower-quality phrasing for consolidation
|
||||
|
||||
#### B. Ambiguity Detection
|
||||
|
||||
- Flag vague adjectives (fast, scalable, secure, intuitive, robust) lacking measurable criteria
|
||||
- Flag unresolved placeholders (TODO, TKTK, ???, `<placeholder>`, etc.)
|
||||
|
||||
#### C. Underspecification
|
||||
|
||||
- Requirements with verbs but missing object or measurable outcome
|
||||
- User stories missing acceptance criteria alignment
|
||||
- Tasks referencing files or components not defined in spec/plan
|
||||
|
||||
#### D. Constitution Alignment
|
||||
|
||||
- Any requirement or plan element conflicting with a MUST principle
|
||||
- Missing mandated sections or quality gates from constitution
|
||||
|
||||
#### E. Coverage Gaps
|
||||
|
||||
- Requirements with zero associated tasks
|
||||
- Tasks with no mapped requirement/story
|
||||
- Success Criteria requiring buildable work (performance, security, availability) not reflected in tasks
|
||||
|
||||
#### F. Inconsistency
|
||||
|
||||
- Terminology drift (same concept named differently across files)
|
||||
- Data entities referenced in plan but absent in spec (or vice versa)
|
||||
- Task ordering contradictions (e.g., integration tasks before foundational setup tasks without dependency note)
|
||||
- Conflicting requirements (e.g., one requires Next.js while other specifies Vue)
|
||||
|
||||
### 5. Severity Assignment
|
||||
|
||||
Use this heuristic to prioritize findings:
|
||||
|
||||
- **CRITICAL**: Violates constitution MUST, missing core spec artifact, or requirement with zero coverage that blocks baseline functionality
|
||||
- **HIGH**: Duplicate or conflicting requirement, ambiguous security/performance attribute, untestable acceptance criterion
|
||||
- **MEDIUM**: Terminology drift, missing non-functional task coverage, underspecified edge case
|
||||
- **LOW**: Style/wording improvements, minor redundancy not affecting execution order
|
||||
|
||||
### 6. Produce Compact Analysis Report
|
||||
|
||||
Output a Markdown report (no file writes) with the following structure:
|
||||
|
||||
## Specification Analysis Report
|
||||
|
||||
| ID | Category | Severity | Location(s) | Summary | Recommendation |
|
||||
|----|----------|----------|-------------|---------|----------------|
|
||||
| A1 | Duplication | HIGH | spec.md:L120-134 | Two similar requirements ... | Merge phrasing; keep clearer version |
|
||||
|
||||
(Add one row per finding; generate stable IDs prefixed by category initial.)
|
||||
|
||||
**Coverage Summary Table:**
|
||||
|
||||
| Requirement Key | Has Task? | Task IDs | Notes |
|
||||
|-----------------|-----------|----------|-------|
|
||||
|
||||
**Constitution Alignment Issues:** (if any)
|
||||
|
||||
**Unmapped Tasks:** (if any)
|
||||
|
||||
**Metrics:**
|
||||
|
||||
- Total Requirements
|
||||
- Total Tasks
|
||||
- Coverage % (requirements with >=1 task)
|
||||
- Ambiguity Count
|
||||
- Duplication Count
|
||||
- Critical Issues Count
|
||||
|
||||
### 7. Provide Next Actions
|
||||
|
||||
At end of report, output a concise Next Actions block:
|
||||
|
||||
- If CRITICAL issues exist: Recommend resolving before `/speckit.implement`
|
||||
- If only LOW/MEDIUM: User may proceed, but provide improvement suggestions
|
||||
- Provide explicit command suggestions: e.g., "Run /speckit.specify with refinement", "Run /speckit.plan to adjust architecture", "Manually edit tasks.md to add coverage for 'performance-metrics'"
|
||||
|
||||
### 8. Offer Remediation
|
||||
|
||||
Ask the user: "Would you like me to suggest concrete remediation edits for the top N issues?" (Do NOT apply them automatically.)
|
||||
|
||||
### 9. Check for extension hooks
|
||||
|
||||
After reporting, check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_analyze` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Operating Principles
|
||||
|
||||
### Context Efficiency
|
||||
|
||||
- **Minimal high-signal tokens**: Focus on actionable findings, not exhaustive documentation
|
||||
- **Progressive disclosure**: Load artifacts incrementally; don't dump all content into analysis
|
||||
- **Token-efficient output**: Limit findings table to 50 rows; summarize overflow
|
||||
- **Deterministic results**: Rerunning without changes should produce consistent IDs and counts
|
||||
|
||||
### Analysis Guidelines
|
||||
|
||||
- **NEVER modify files** (this is read-only analysis)
|
||||
- **NEVER hallucinate missing sections** (if absent, report them accurately)
|
||||
- **Prioritize constitution violations** (these are always CRITICAL)
|
||||
- **Use examples over exhaustive rules** (cite specific instances, not generic patterns)
|
||||
- **Report zero issues gracefully** (emit success report with coverage statistics)
|
||||
|
||||
## Context
|
||||
|
||||
{ARGS}
|
||||
@@ -0,0 +1,371 @@
|
||||
---
|
||||
name: speckit-checklist
|
||||
description: Generate a custom checklist for the current feature based on user requirements.
|
||||
argument-hint: "Domain or focus area for the checklist"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/checklist.md
|
||||
---
|
||||
|
||||
## Checklist Purpose: "Unit Tests for English"
|
||||
|
||||
**CRITICAL CONCEPT**: Checklists are **UNIT TESTS FOR REQUIREMENTS WRITING** - they validate the quality, clarity, and completeness of requirements in a given domain.
|
||||
|
||||
**NOT for verification/testing**:
|
||||
|
||||
- ❌ NOT "Verify the button clicks correctly"
|
||||
- ❌ NOT "Test error handling works"
|
||||
- ❌ NOT "Confirm the API returns 200"
|
||||
- ❌ NOT checking if code/implementation matches the spec
|
||||
|
||||
**FOR requirements quality validation**:
|
||||
|
||||
- ✅ "Are visual hierarchy requirements defined for all card types?" (completeness)
|
||||
- ✅ "Is 'prominent display' quantified with specific sizing/positioning?" (clarity)
|
||||
- ✅ "Are hover state requirements consistent across all interactive elements?" (consistency)
|
||||
- ✅ "Are accessibility requirements defined for keyboard navigation?" (coverage)
|
||||
- ✅ "Does the spec define what happens when logo image fails to load?" (edge cases)
|
||||
|
||||
**Metaphor**: If your spec is code written in English, the checklist is its unit test suite. You're testing whether the requirements are well-written, complete, unambiguous, and ready for implementation - NOT whether the implementation works.
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before checklist generation)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_checklist` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Execution Steps.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Execution Steps
|
||||
|
||||
1. **Setup**: Run `{SCRIPT}` from repo root and parse JSON for FEATURE_DIR and AVAILABLE_DOCS list.
|
||||
- All file paths must be absolute.
|
||||
- For single quotes in args like "I'm Groot", use escape syntax: e.g 'I'\''m Groot' (or double-quote if possible: "I'm Groot").
|
||||
|
||||
2. **Clarify intent (dynamic)**: Derive up to THREE initial contextual clarifying questions (no pre-baked catalog). They MUST:
|
||||
- Be generated from the user's phrasing + extracted signals from spec/plan/tasks
|
||||
- Only ask about information that materially changes checklist content
|
||||
- Be skipped individually if already unambiguous in `$ARGUMENTS`
|
||||
- Prefer precision over breadth
|
||||
|
||||
Generation algorithm:
|
||||
1. Extract signals: feature domain keywords (e.g., auth, latency, UX, API), risk indicators ("critical", "must", "compliance"), stakeholder hints ("QA", "review", "security team"), and explicit deliverables ("a11y", "rollback", "contracts").
|
||||
2. Cluster signals into candidate focus areas (max 4) ranked by relevance.
|
||||
3. Identify probable audience & timing (author, reviewer, QA, release) if not explicit.
|
||||
4. Detect missing dimensions: scope breadth, depth/rigor, risk emphasis, exclusion boundaries, measurable acceptance criteria.
|
||||
5. Formulate questions chosen from these archetypes:
|
||||
- Scope refinement (e.g., "Should this include integration touchpoints with X and Y or stay limited to local module correctness?")
|
||||
- Risk prioritization (e.g., "Which of these potential risk areas should receive mandatory gating checks?")
|
||||
- Depth calibration (e.g., "Is this a lightweight pre-commit sanity list or a formal release gate?")
|
||||
- Audience framing (e.g., "Will this be used by the author only or peers during PR review?")
|
||||
- Boundary exclusion (e.g., "Should we explicitly exclude performance tuning items this round?")
|
||||
- Scenario class gap (e.g., "No recovery flows detected—are rollback / partial failure paths in scope?")
|
||||
|
||||
Question formatting rules:
|
||||
- If presenting options, generate a compact table with columns: Option | Candidate | Why It Matters
|
||||
- Limit to A–E options maximum; omit table if a free-form answer is clearer
|
||||
- Never ask the user to restate what they already said
|
||||
- Avoid speculative categories (no hallucination). If uncertain, ask explicitly: "Confirm whether X belongs in scope."
|
||||
|
||||
Defaults when interaction impossible:
|
||||
- Depth: Standard
|
||||
- Audience: Reviewer (PR) if code-related; Author otherwise
|
||||
- Focus: Top 2 relevance clusters
|
||||
|
||||
Output the questions (label Q1/Q2/Q3). After answers: if ≥2 scenario classes (Alternate / Exception / Recovery / Non-Functional domain) remain unclear, you MAY ask up to TWO more targeted follow‑ups (Q4/Q5) with a one-line justification each (e.g., "Unresolved recovery path risk"). Do not exceed five total questions. Skip escalation if user explicitly declines more.
|
||||
|
||||
3. **Understand user request**: Combine `$ARGUMENTS` + clarifying answers:
|
||||
- Derive checklist theme (e.g., security, review, deploy, ux)
|
||||
- Consolidate explicit must-have items mentioned by user
|
||||
- Map focus selections to category scaffolding
|
||||
- Infer any missing context from spec/plan/tasks (do NOT hallucinate)
|
||||
|
||||
4. **Load feature context**: Read from FEATURE_DIR:
|
||||
- spec.md: Feature requirements and scope
|
||||
- plan.md (if exists): Technical details, dependencies
|
||||
- tasks.md (if exists): Implementation tasks
|
||||
|
||||
**Context Loading Strategy**:
|
||||
- Load only necessary portions relevant to active focus areas (avoid full-file dumping)
|
||||
- Prefer summarizing long sections into concise scenario/requirement bullets
|
||||
- Use progressive disclosure: add follow-on retrieval only if gaps detected
|
||||
- If source docs are large, generate interim summary items instead of embedding raw text
|
||||
|
||||
5. **Generate checklist** - Create "Unit Tests for Requirements":
|
||||
- Create `FEATURE_DIR/checklists/` directory if it doesn't exist
|
||||
- Generate unique checklist filename:
|
||||
- Use short, descriptive name based on domain (e.g., `ux.md`, `api.md`, `security.md`)
|
||||
- Format: `[domain].md`
|
||||
- File handling behavior:
|
||||
- If file does NOT exist: Create new file and number items starting from CHK001
|
||||
- If file exists: Append new items to existing file, continuing from the last CHK ID (e.g., if last item is CHK015, start new items at CHK016)
|
||||
- Never delete or replace existing checklist content - always preserve and append
|
||||
|
||||
**CORE PRINCIPLE - Test the Requirements, Not the Implementation**:
|
||||
Every checklist item MUST evaluate the REQUIREMENTS THEMSELVES for:
|
||||
- **Completeness**: Are all necessary requirements present?
|
||||
- **Clarity**: Are requirements unambiguous and specific?
|
||||
- **Consistency**: Do requirements align with each other?
|
||||
- **Measurability**: Can requirements be objectively verified?
|
||||
- **Coverage**: Are all scenarios/edge cases addressed?
|
||||
|
||||
**Category Structure** - Group items by requirement quality dimensions:
|
||||
- **Requirement Completeness** (Are all necessary requirements documented?)
|
||||
- **Requirement Clarity** (Are requirements specific and unambiguous?)
|
||||
- **Requirement Consistency** (Do requirements align without conflicts?)
|
||||
- **Acceptance Criteria Quality** (Are success criteria measurable?)
|
||||
- **Scenario Coverage** (Are all flows/cases addressed?)
|
||||
- **Edge Case Coverage** (Are boundary conditions defined?)
|
||||
- **Non-Functional Requirements** (Performance, Security, Accessibility, etc. - are they specified?)
|
||||
- **Dependencies & Assumptions** (Are they documented and validated?)
|
||||
- **Ambiguities & Conflicts** (What needs clarification?)
|
||||
|
||||
**HOW TO WRITE CHECKLIST ITEMS - "Unit Tests for English"**:
|
||||
|
||||
❌ **WRONG** (Testing implementation):
|
||||
- "Verify landing page displays 3 episode cards"
|
||||
- "Test hover states work on desktop"
|
||||
- "Confirm logo click navigates home"
|
||||
|
||||
✅ **CORRECT** (Testing requirements quality):
|
||||
- "Are the exact number and layout of featured episodes specified?" [Completeness]
|
||||
- "Is 'prominent display' quantified with specific sizing/positioning?" [Clarity]
|
||||
- "Are hover state requirements consistent across all interactive elements?" [Consistency]
|
||||
- "Are keyboard navigation requirements defined for all interactive UI?" [Coverage]
|
||||
- "Is the fallback behavior specified when logo image fails to load?" [Edge Cases]
|
||||
- "Are loading states defined for asynchronous episode data?" [Completeness]
|
||||
- "Does the spec define visual hierarchy for competing UI elements?" [Clarity]
|
||||
|
||||
**ITEM STRUCTURE**:
|
||||
Each item should follow this pattern:
|
||||
- Question format asking about requirement quality
|
||||
- Focus on what's WRITTEN (or not written) in the spec/plan
|
||||
- Include quality dimension in brackets [Completeness/Clarity/Consistency/etc.]
|
||||
- Reference spec section `[Spec §X.Y]` when checking existing requirements
|
||||
- Use `[Gap]` marker when checking for missing requirements
|
||||
|
||||
**EXAMPLES BY QUALITY DIMENSION**:
|
||||
|
||||
Completeness:
|
||||
- "Are error handling requirements defined for all API failure modes? [Gap]"
|
||||
- "Are accessibility requirements specified for all interactive elements? [Completeness]"
|
||||
- "Are mobile breakpoint requirements defined for responsive layouts? [Gap]"
|
||||
|
||||
Clarity:
|
||||
- "Is 'fast loading' quantified with specific timing thresholds? [Clarity, Spec §NFR-2]"
|
||||
- "Are 'related episodes' selection criteria explicitly defined? [Clarity, Spec §FR-5]"
|
||||
- "Is 'prominent' defined with measurable visual properties? [Ambiguity, Spec §FR-4]"
|
||||
|
||||
Consistency:
|
||||
- "Do navigation requirements align across all pages? [Consistency, Spec §FR-10]"
|
||||
- "Are card component requirements consistent between landing and detail pages? [Consistency]"
|
||||
|
||||
Coverage:
|
||||
- "Are requirements defined for zero-state scenarios (no episodes)? [Coverage, Edge Case]"
|
||||
- "Are concurrent user interaction scenarios addressed? [Coverage, Gap]"
|
||||
- "Are requirements specified for partial data loading failures? [Coverage, Exception Flow]"
|
||||
|
||||
Measurability:
|
||||
- "Are visual hierarchy requirements measurable/testable? [Acceptance Criteria, Spec §FR-1]"
|
||||
- "Can 'balanced visual weight' be objectively verified? [Measurability, Spec §FR-2]"
|
||||
|
||||
**Scenario Classification & Coverage** (Requirements Quality Focus):
|
||||
- Check if requirements exist for: Primary, Alternate, Exception/Error, Recovery, Non-Functional scenarios
|
||||
- For each scenario class, ask: "Are [scenario type] requirements complete, clear, and consistent?"
|
||||
- If scenario class missing: "Are [scenario type] requirements intentionally excluded or missing? [Gap]"
|
||||
- Include resilience/rollback when state mutation occurs: "Are rollback requirements defined for migration failures? [Gap]"
|
||||
|
||||
**Traceability Requirements**:
|
||||
- MINIMUM: ≥80% of items MUST include at least one traceability reference
|
||||
- Each item should reference: spec section `[Spec §X.Y]`, or use markers: `[Gap]`, `[Ambiguity]`, `[Conflict]`, `[Assumption]`
|
||||
- If no ID system exists: "Is a requirement & acceptance criteria ID scheme established? [Traceability]"
|
||||
|
||||
**Surface & Resolve Issues** (Requirements Quality Problems):
|
||||
Ask questions about the requirements themselves:
|
||||
- Ambiguities: "Is the term 'fast' quantified with specific metrics? [Ambiguity, Spec §NFR-1]"
|
||||
- Conflicts: "Do navigation requirements conflict between §FR-10 and §FR-10a? [Conflict]"
|
||||
- Assumptions: "Is the assumption of 'always available podcast API' validated? [Assumption]"
|
||||
- Dependencies: "Are external podcast API requirements documented? [Dependency, Gap]"
|
||||
- Missing definitions: "Is 'visual hierarchy' defined with measurable criteria? [Gap]"
|
||||
|
||||
**Content Consolidation**:
|
||||
- Soft cap: If raw candidate items > 40, prioritize by risk/impact
|
||||
- Merge near-duplicates checking the same requirement aspect
|
||||
- If >5 low-impact edge cases, create one item: "Are edge cases X, Y, Z addressed in requirements? [Coverage]"
|
||||
|
||||
**🚫 ABSOLUTELY PROHIBITED** - These make it an implementation test, not a requirements test:
|
||||
- ❌ Any item starting with "Verify", "Test", "Confirm", "Check" + implementation behavior
|
||||
- ❌ References to code execution, user actions, system behavior
|
||||
- ❌ "Displays correctly", "works properly", "functions as expected"
|
||||
- ❌ "Click", "navigate", "render", "load", "execute"
|
||||
- ❌ Test cases, test plans, QA procedures
|
||||
- ❌ Implementation details (frameworks, APIs, algorithms)
|
||||
|
||||
**✅ REQUIRED PATTERNS** - These test requirements quality:
|
||||
- ✅ "Are [requirement type] defined/specified/documented for [scenario]?"
|
||||
- ✅ "Is [vague term] quantified/clarified with specific criteria?"
|
||||
- ✅ "Are requirements consistent between [section A] and [section B]?"
|
||||
- ✅ "Can [requirement] be objectively measured/verified?"
|
||||
- ✅ "Are [edge cases/scenarios] addressed in requirements?"
|
||||
- ✅ "Does the spec define [missing aspect]?"
|
||||
|
||||
6. **Structure Reference**: Generate the checklist following the canonical template in `templates/checklist-template.md` for title, meta section, category headings, and ID formatting. If template is unavailable, use: H1 title, purpose/created meta lines, `##` category sections containing `- [ ] CHK### <requirement item>` lines with globally incrementing IDs starting at CHK001.
|
||||
|
||||
7. **Report**: Output full path to checklist file, item count, and summarize whether the run created a new file or appended to an existing one. Summarize:
|
||||
- Focus areas selected
|
||||
- Depth level
|
||||
- Actor/timing
|
||||
- Any explicit user-specified must-have items incorporated
|
||||
|
||||
**Important**: Each `/speckit.checklist` command invocation uses a short, descriptive checklist filename and either creates a new file or appends to an existing one. This allows:
|
||||
|
||||
- Multiple checklists of different types (e.g., `ux.md`, `test.md`, `security.md`)
|
||||
- Simple, memorable filenames that indicate checklist purpose
|
||||
- Easy identification and navigation in the `checklists/` folder
|
||||
|
||||
To avoid clutter, use descriptive types and clean up obsolete checklists when done.
|
||||
|
||||
## Example Checklist Types & Sample Items
|
||||
|
||||
**UX Requirements Quality:** `ux.md`
|
||||
|
||||
Sample items (testing the requirements, NOT the implementation):
|
||||
|
||||
- "Are visual hierarchy requirements defined with measurable criteria? [Clarity, Spec §FR-1]"
|
||||
- "Is the number and positioning of UI elements explicitly specified? [Completeness, Spec §FR-1]"
|
||||
- "Are interaction state requirements (hover, focus, active) consistently defined? [Consistency]"
|
||||
- "Are accessibility requirements specified for all interactive elements? [Coverage, Gap]"
|
||||
- "Is fallback behavior defined when images fail to load? [Edge Case, Gap]"
|
||||
- "Can 'prominent display' be objectively measured? [Measurability, Spec §FR-4]"
|
||||
|
||||
**API Requirements Quality:** `api.md`
|
||||
|
||||
Sample items:
|
||||
|
||||
- "Are error response formats specified for all failure scenarios? [Completeness]"
|
||||
- "Are rate limiting requirements quantified with specific thresholds? [Clarity]"
|
||||
- "Are authentication requirements consistent across all endpoints? [Consistency]"
|
||||
- "Are retry/timeout requirements defined for external dependencies? [Coverage, Gap]"
|
||||
- "Is versioning strategy documented in requirements? [Gap]"
|
||||
|
||||
**Performance Requirements Quality:** `performance.md`
|
||||
|
||||
Sample items:
|
||||
|
||||
- "Are performance requirements quantified with specific metrics? [Clarity]"
|
||||
- "Are performance targets defined for all critical user journeys? [Coverage]"
|
||||
- "Are performance requirements under different load conditions specified? [Completeness]"
|
||||
- "Can performance requirements be objectively measured? [Measurability]"
|
||||
- "Are degradation requirements defined for high-load scenarios? [Edge Case, Gap]"
|
||||
|
||||
**Security Requirements Quality:** `security.md`
|
||||
|
||||
Sample items:
|
||||
|
||||
- "Are authentication requirements specified for all protected resources? [Coverage]"
|
||||
- "Are data protection requirements defined for sensitive information? [Completeness]"
|
||||
- "Is the threat model documented and requirements aligned to it? [Traceability]"
|
||||
- "Are security requirements consistent with compliance obligations? [Consistency]"
|
||||
- "Are security failure/breach response requirements defined? [Gap, Exception Flow]"
|
||||
|
||||
## Anti-Examples: What NOT To Do
|
||||
|
||||
**❌ WRONG - These test implementation, not requirements:**
|
||||
|
||||
```markdown
|
||||
- [ ] CHK001 - Verify landing page displays 3 episode cards [Spec §FR-001]
|
||||
- [ ] CHK002 - Test hover states work correctly on desktop [Spec §FR-003]
|
||||
- [ ] CHK003 - Confirm logo click navigates to home page [Spec §FR-010]
|
||||
- [ ] CHK004 - Check that related episodes section shows 3-5 items [Spec §FR-005]
|
||||
```
|
||||
|
||||
**✅ CORRECT - These test requirements quality:**
|
||||
|
||||
```markdown
|
||||
- [ ] CHK001 - Are the number and layout of featured episodes explicitly specified? [Completeness, Spec §FR-001]
|
||||
- [ ] CHK002 - Are hover state requirements consistently defined for all interactive elements? [Consistency, Spec §FR-003]
|
||||
- [ ] CHK003 - Are navigation requirements clear for all clickable brand elements? [Clarity, Spec §FR-010]
|
||||
- [ ] CHK004 - Is the selection criteria for related episodes documented? [Gap, Spec §FR-005]
|
||||
- [ ] CHK005 - Are loading state requirements defined for asynchronous episode data? [Gap]
|
||||
- [ ] CHK006 - Can "visual hierarchy" requirements be objectively measured? [Measurability, Spec §FR-001]
|
||||
```
|
||||
|
||||
**Key Differences:**
|
||||
|
||||
- Wrong: Tests if the system works correctly
|
||||
- Correct: Tests if the requirements are written correctly
|
||||
- Wrong: Verification of behavior
|
||||
- Correct: Validation of requirement quality
|
||||
- Wrong: "Does it do X?"
|
||||
- Correct: "Is X clearly specified?"
|
||||
|
||||
## Post-Execution Checks
|
||||
|
||||
**Check for extension hooks (after checklist generation)**:
|
||||
Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_checklist` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
@@ -0,0 +1,253 @@
|
||||
---
|
||||
name: speckit-clarify
|
||||
description: Identify underspecified areas in the current feature spec by asking up to 5 highly targeted clarification questions and encoding answers back into the spec.
|
||||
argument-hint: "Optional areas to clarify in the spec"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/clarify.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before clarification)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_clarify` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Outline.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Outline
|
||||
|
||||
Goal: Detect and reduce ambiguity or missing decision points in the active feature specification and record the clarifications directly in the spec file.
|
||||
|
||||
Note: This clarification workflow is expected to run (and be completed) BEFORE invoking `/speckit.plan`. If the user explicitly states they are skipping clarification (e.g., exploratory spike), you may proceed, but must warn that downstream rework risk increases.
|
||||
|
||||
Execution steps:
|
||||
|
||||
1. Run `{SCRIPT}` from repo root **once** (combined `--json --paths-only` mode / `-Json -PathsOnly`). Parse minimal JSON payload fields:
|
||||
- `FEATURE_DIR`
|
||||
- `FEATURE_SPEC`
|
||||
- (Optionally capture `IMPL_PLAN`, `TASKS` for future chained flows.)
|
||||
- If JSON parsing fails, abort and instruct user to re-run `/speckit.specify` or verify feature branch environment.
|
||||
- For single quotes in args like "I'm Groot", use escape syntax: e.g 'I'\''m Groot' (or double-quote if possible: "I'm Groot").
|
||||
|
||||
2. Load the current spec file. Perform a structured ambiguity & coverage scan using this taxonomy. For each category, mark status: Clear / Partial / Missing. Produce an internal coverage map used for prioritization (do not output raw map unless no questions will be asked).
|
||||
|
||||
Functional Scope & Behavior:
|
||||
- Core user goals & success criteria
|
||||
- Explicit out-of-scope declarations
|
||||
- User roles / personas differentiation
|
||||
|
||||
Domain & Data Model:
|
||||
- Entities, attributes, relationships
|
||||
- Identity & uniqueness rules
|
||||
- Lifecycle/state transitions
|
||||
- Data volume / scale assumptions
|
||||
|
||||
Interaction & UX Flow:
|
||||
- Critical user journeys / sequences
|
||||
- Error/empty/loading states
|
||||
- Accessibility or localization notes
|
||||
|
||||
Non-Functional Quality Attributes:
|
||||
- Performance (latency, throughput targets)
|
||||
- Scalability (horizontal/vertical, limits)
|
||||
- Reliability & availability (uptime, recovery expectations)
|
||||
- Observability (logging, metrics, tracing signals)
|
||||
- Security & privacy (authN/Z, data protection, threat assumptions)
|
||||
- Compliance / regulatory constraints (if any)
|
||||
|
||||
Integration & External Dependencies:
|
||||
- External services/APIs and failure modes
|
||||
- Data import/export formats
|
||||
- Protocol/versioning assumptions
|
||||
|
||||
Edge Cases & Failure Handling:
|
||||
- Negative scenarios
|
||||
- Rate limiting / throttling
|
||||
- Conflict resolution (e.g., concurrent edits)
|
||||
|
||||
Constraints & Tradeoffs:
|
||||
- Technical constraints (language, storage, hosting)
|
||||
- Explicit tradeoffs or rejected alternatives
|
||||
|
||||
Terminology & Consistency:
|
||||
- Canonical glossary terms
|
||||
- Avoided synonyms / deprecated terms
|
||||
|
||||
Completion Signals:
|
||||
- Acceptance criteria testability
|
||||
- Measurable Definition of Done style indicators
|
||||
|
||||
Misc / Placeholders:
|
||||
- TODO markers / unresolved decisions
|
||||
- Ambiguous adjectives ("robust", "intuitive") lacking quantification
|
||||
|
||||
For each category with Partial or Missing status, add a candidate question opportunity unless:
|
||||
- Clarification would not materially change implementation or validation strategy
|
||||
- Information is better deferred to planning phase (note internally)
|
||||
|
||||
3. Generate (internally) a prioritized queue of candidate clarification questions (maximum 5). Do NOT output them all at once. Apply these constraints:
|
||||
- Maximum of 5 total questions across the whole session.
|
||||
- Each question must be answerable with EITHER:
|
||||
- A short multiple‑choice selection (2–5 distinct, mutually exclusive options), OR
|
||||
- A one-word / short‑phrase answer (explicitly constrain: "Answer in <=5 words").
|
||||
- Only include questions whose answers materially impact architecture, data modeling, task decomposition, test design, UX behavior, operational readiness, or compliance validation.
|
||||
- Ensure category coverage balance: attempt to cover the highest impact unresolved categories first; avoid asking two low-impact questions when a single high-impact area (e.g., security posture) is unresolved.
|
||||
- Exclude questions already answered, trivial stylistic preferences, or plan-level execution details (unless blocking correctness).
|
||||
- Favor clarifications that reduce downstream rework risk or prevent misaligned acceptance tests.
|
||||
- If more than 5 categories remain unresolved, select the top 5 by (Impact * Uncertainty) heuristic.
|
||||
|
||||
4. Sequential questioning loop (interactive):
|
||||
- Present EXACTLY ONE question at a time.
|
||||
- For multiple‑choice questions:
|
||||
- **Analyze all options** and determine the **most suitable option** based on:
|
||||
- Best practices for the project type
|
||||
- Common patterns in similar implementations
|
||||
- Risk reduction (security, performance, maintainability)
|
||||
- Alignment with any explicit project goals or constraints visible in the spec
|
||||
- Present your **recommended option prominently** at the top with clear reasoning (1-2 sentences explaining why this is the best choice).
|
||||
- Format as: `**Recommended:** Option [X] - <reasoning>`
|
||||
- Then render all options as a Markdown table:
|
||||
|
||||
| Option | Description |
|
||||
|--------|-------------|
|
||||
| A | <Option A description> |
|
||||
| B | <Option B description> |
|
||||
| C | <Option C description> (add D/E as needed up to 5) |
|
||||
| Short | Provide a different short answer (<=5 words) (Include only if free-form alternative is appropriate) |
|
||||
|
||||
- After the table, add: `You can reply with the option letter (e.g., "A"), accept the recommendation by saying "yes" or "recommended", or provide your own short answer.`
|
||||
- For short‑answer style (no meaningful discrete options):
|
||||
- Provide your **suggested answer** based on best practices and context.
|
||||
- Format as: `**Suggested:** <your proposed answer> - <brief reasoning>`
|
||||
- Then output: `Format: Short answer (<=5 words). You can accept the suggestion by saying "yes" or "suggested", or provide your own answer.`
|
||||
- After the user answers:
|
||||
- If the user replies with "yes", "recommended", or "suggested", use your previously stated recommendation/suggestion as the answer.
|
||||
- Otherwise, validate the answer maps to one option or fits the <=5 word constraint.
|
||||
- If ambiguous, ask for a quick disambiguation (count still belongs to same question; do not advance).
|
||||
- Once satisfactory, record it in working memory (do not yet write to disk) and move to the next queued question.
|
||||
- Stop asking further questions when:
|
||||
- All critical ambiguities resolved early (remaining queued items become unnecessary), OR
|
||||
- User signals completion ("done", "good", "no more"), OR
|
||||
- You reach 5 asked questions.
|
||||
- Never reveal future queued questions in advance.
|
||||
- If no valid questions exist at start, immediately report no critical ambiguities.
|
||||
|
||||
5. Integration after EACH accepted answer (incremental update approach):
|
||||
- Maintain in-memory representation of the spec (loaded once at start) plus the raw file contents.
|
||||
- For the first integrated answer in this session:
|
||||
- Ensure a `## Clarifications` section exists (create it just after the highest-level contextual/overview section per the spec template if missing).
|
||||
- Under it, create (if not present) a `### Session YYYY-MM-DD` subheading for today.
|
||||
- Append a bullet line immediately after acceptance: `- Q: <question> → A: <final answer>`.
|
||||
- Then immediately apply the clarification to the most appropriate section(s):
|
||||
- Functional ambiguity → Update or add a bullet in Functional Requirements.
|
||||
- User interaction / actor distinction → Update User Stories or Actors subsection (if present) with clarified role, constraint, or scenario.
|
||||
- Data shape / entities → Update Data Model (add fields, types, relationships) preserving ordering; note added constraints succinctly.
|
||||
- Non-functional constraint → Add/modify measurable criteria in Success Criteria > Measurable Outcomes (convert vague adjective to metric or explicit target).
|
||||
- Edge case / negative flow → Add a new bullet under Edge Cases / Error Handling (or create such subsection if template provides placeholder for it).
|
||||
- Terminology conflict → Normalize term across spec; retain original only if necessary by adding `(formerly referred to as "X")` once.
|
||||
- If the clarification invalidates an earlier ambiguous statement, replace that statement instead of duplicating; leave no obsolete contradictory text.
|
||||
- Save the spec file AFTER each integration to minimize risk of context loss (atomic overwrite).
|
||||
- Preserve formatting: do not reorder unrelated sections; keep heading hierarchy intact.
|
||||
- Keep each inserted clarification minimal and testable (avoid narrative drift).
|
||||
|
||||
6. Validation (performed after EACH write plus final pass):
|
||||
- Clarifications session contains exactly one bullet per accepted answer (no duplicates).
|
||||
- Total asked (accepted) questions ≤ 5.
|
||||
- Updated sections contain no lingering vague placeholders the new answer was meant to resolve.
|
||||
- No contradictory earlier statement remains (scan for now-invalid alternative choices removed).
|
||||
- Markdown structure valid; only allowed new headings: `## Clarifications`, `### Session YYYY-MM-DD`.
|
||||
- Terminology consistency: same canonical term used across all updated sections.
|
||||
|
||||
7. Write the updated spec back to `FEATURE_SPEC`.
|
||||
|
||||
8. Report completion (after questioning loop ends or early termination):
|
||||
- Number of questions asked & answered.
|
||||
- Path to updated spec.
|
||||
- Sections touched (list names).
|
||||
- Coverage summary table listing each taxonomy category with Status: Resolved (was Partial/Missing and addressed), Deferred (exceeds question quota or better suited for planning), Clear (already sufficient), Outstanding (still Partial/Missing but low impact).
|
||||
- If any Outstanding or Deferred remain, recommend whether to proceed to `/speckit.plan` or run `/speckit.clarify` again later post-plan.
|
||||
- Suggested next command.
|
||||
|
||||
Behavior rules:
|
||||
|
||||
- If no meaningful ambiguities found (or all potential questions would be low-impact), respond: "No critical ambiguities detected worth formal clarification." and suggest proceeding.
|
||||
- If spec file missing, instruct user to run `/speckit.specify` first (do not create a new spec here).
|
||||
- Never exceed 5 total asked questions (clarification retries for a single question do not count as new questions).
|
||||
- Avoid speculative tech stack questions unless the absence blocks functional clarity.
|
||||
- Respect user early termination signals ("stop", "done", "proceed").
|
||||
- If no questions asked due to full coverage, output a compact coverage summary (all categories Clear) then suggest advancing.
|
||||
- If quota reached with unresolved high-impact categories remaining, explicitly flag them under Deferred with rationale.
|
||||
|
||||
Context for prioritization: {ARGS}
|
||||
|
||||
## Post-Execution Checks
|
||||
|
||||
**Check for extension hooks (after clarification)**:
|
||||
Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_clarify` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
@@ -0,0 +1,156 @@
|
||||
---
|
||||
name: speckit-constitution
|
||||
description: Create or update the project constitution from interactive or provided principle inputs, ensuring all dependent templates stay in sync.
|
||||
argument-hint: "Principles or values for the project constitution"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/constitution.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before constitution update)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_constitution` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Outline.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Outline
|
||||
|
||||
You are updating the project constitution at `.specify/memory/constitution.md`. This file is a TEMPLATE containing placeholder tokens in square brackets (e.g. `[PROJECT_NAME]`, `[PRINCIPLE_1_NAME]`). Your job is to (a) collect/derive concrete values, (b) fill the template precisely, and (c) propagate any amendments across dependent artifacts.
|
||||
|
||||
**Note**: If `.specify/memory/constitution.md` does not exist yet, it should have been initialized from `.specify/templates/constitution-template.md` during project setup. If it's missing, copy the template first.
|
||||
|
||||
Follow this execution flow:
|
||||
|
||||
1. Load the existing constitution at `.specify/memory/constitution.md`.
|
||||
- Identify every placeholder token of the form `[ALL_CAPS_IDENTIFIER]`.
|
||||
**IMPORTANT**: The user might require less or more principles than the ones used in the template. If a number is specified, respect that - follow the general template. You will update the doc accordingly.
|
||||
|
||||
2. Collect/derive values for placeholders:
|
||||
- If user input (conversation) supplies a value, use it.
|
||||
- Otherwise infer from existing repo context (README, docs, prior constitution versions if embedded).
|
||||
- For governance dates: `RATIFICATION_DATE` is the original adoption date (if unknown ask or mark TODO), `LAST_AMENDED_DATE` is today if changes are made, otherwise keep previous.
|
||||
- `CONSTITUTION_VERSION` must increment according to semantic versioning rules:
|
||||
- MAJOR: Backward incompatible governance/principle removals or redefinitions.
|
||||
- MINOR: New principle/section added or materially expanded guidance.
|
||||
- PATCH: Clarifications, wording, typo fixes, non-semantic refinements.
|
||||
- If version bump type ambiguous, propose reasoning before finalizing.
|
||||
|
||||
3. Draft the updated constitution content:
|
||||
- Replace every placeholder with concrete text (no bracketed tokens left except intentionally retained template slots that the project has chosen not to define yet—explicitly justify any left).
|
||||
- Preserve heading hierarchy and comments can be removed once replaced unless they still add clarifying guidance.
|
||||
- Ensure each Principle section: succinct name line, paragraph (or bullet list) capturing non‑negotiable rules, explicit rationale if not obvious.
|
||||
- Ensure Governance section lists amendment procedure, versioning policy, and compliance review expectations.
|
||||
|
||||
4. Consistency propagation checklist (convert prior checklist into active validations):
|
||||
- Read `.specify/templates/plan-template.md` and ensure any "Constitution Check" or rules align with updated principles.
|
||||
- Read `.specify/templates/spec-template.md` for scope/requirements alignment—update if constitution adds/removes mandatory sections or constraints.
|
||||
- Read `.specify/templates/tasks-template.md` and ensure task categorization reflects new or removed principle-driven task types (e.g., observability, versioning, testing discipline).
|
||||
- Read each command file in `.specify/templates/commands/*.md` (including this one) to verify no outdated references (agent-specific names like CLAUDE only) remain when generic guidance is required.
|
||||
- Read any runtime guidance docs (e.g., `README.md`, `docs/quickstart.md`, or agent-specific guidance files if present). Update references to principles changed.
|
||||
|
||||
5. Produce a Sync Impact Report (prepend as an HTML comment at top of the constitution file after update):
|
||||
- Version change: old → new
|
||||
- List of modified principles (old title → new title if renamed)
|
||||
- Added sections
|
||||
- Removed sections
|
||||
- Templates requiring updates (✅ updated / ⚠ pending) with file paths
|
||||
- Follow-up TODOs if any placeholders intentionally deferred.
|
||||
|
||||
6. Validation before final output:
|
||||
- No remaining unexplained bracket tokens.
|
||||
- Version line matches report.
|
||||
- Dates ISO format YYYY-MM-DD.
|
||||
- Principles are declarative, testable, and free of vague language ("should" → replace with MUST/SHOULD rationale where appropriate).
|
||||
|
||||
7. Write the completed constitution back to `.specify/memory/constitution.md` (overwrite).
|
||||
|
||||
8. Output a final summary to the user with:
|
||||
- New version and bump rationale.
|
||||
- Any files flagged for manual follow-up.
|
||||
- Suggested commit message (e.g., `docs: amend constitution to vX.Y.Z (principle additions + governance update)`).
|
||||
|
||||
Formatting & Style Requirements:
|
||||
|
||||
- Use Markdown headings exactly as in the template (do not demote/promote levels).
|
||||
- Wrap long rationale lines to keep readability (<100 chars ideally) but do not hard enforce with awkward breaks.
|
||||
- Keep a single blank line between sections.
|
||||
- Avoid trailing whitespace.
|
||||
|
||||
If the user supplies partial updates (e.g., only one principle revision), still perform validation and version decision steps.
|
||||
|
||||
If critical info missing (e.g., ratification date truly unknown), insert `TODO(<FIELD_NAME>): explanation` and include in the Sync Impact Report under deferred items.
|
||||
|
||||
Do not create a new template; always operate on the existing `.specify/memory/constitution.md` file.
|
||||
|
||||
## Post-Execution Checks
|
||||
|
||||
**Check for extension hooks (after constitution update)**:
|
||||
Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_constitution` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
@@ -0,0 +1,208 @@
|
||||
---
|
||||
name: speckit-implement
|
||||
description: Execute the implementation plan by processing and executing all tasks defined in tasks.md
|
||||
argument-hint: "Optional implementation guidance or task filter"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/implement.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before implementation)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_implement` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Outline.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Outline
|
||||
|
||||
1. Run `{SCRIPT}` from repo root and parse FEATURE_DIR and AVAILABLE_DOCS list. All paths must be absolute. For single quotes in args like "I'm Groot", use escape syntax: e.g 'I'\''m Groot' (or double-quote if possible: "I'm Groot").
|
||||
|
||||
2. **Check checklists status** (if FEATURE_DIR/checklists/ exists):
|
||||
- Scan all checklist files in the checklists/ directory
|
||||
- For each checklist, count:
|
||||
- Total items: All lines matching `- [ ]` or `- [X]` or `- [x]`
|
||||
- Completed items: Lines matching `- [X]` or `- [x]`
|
||||
- Incomplete items: Lines matching `- [ ]`
|
||||
- Create a status table:
|
||||
|
||||
```text
|
||||
| Checklist | Total | Completed | Incomplete | Status |
|
||||
|-----------|-------|-----------|------------|--------|
|
||||
| ux.md | 12 | 12 | 0 | ✓ PASS |
|
||||
| test.md | 8 | 5 | 3 | ✗ FAIL |
|
||||
| security.md | 6 | 6 | 0 | ✓ PASS |
|
||||
```
|
||||
|
||||
- Calculate overall status:
|
||||
- **PASS**: All checklists have 0 incomplete items
|
||||
- **FAIL**: One or more checklists have incomplete items
|
||||
|
||||
- **If any checklist is incomplete**:
|
||||
- Display the table with incomplete item counts
|
||||
- **STOP** and ask: "Some checklists are incomplete. Do you want to proceed with implementation anyway? (yes/no)"
|
||||
- Wait for user response before continuing
|
||||
- If user says "no" or "wait" or "stop", halt execution
|
||||
- If user says "yes" or "proceed" or "continue", proceed to step 3
|
||||
|
||||
- **If all checklists are complete**:
|
||||
- Display the table showing all checklists passed
|
||||
- Automatically proceed to step 3
|
||||
|
||||
3. Load and analyze the implementation context:
|
||||
- **REQUIRED**: Read tasks.md for the complete task list and execution plan
|
||||
- **REQUIRED**: Read plan.md for tech stack, architecture, and file structure
|
||||
- **IF EXISTS**: Read data-model.md for entities and relationships
|
||||
- **IF EXISTS**: Read contracts/ for API specifications and test requirements
|
||||
- **IF EXISTS**: Read research.md for technical decisions and constraints
|
||||
- **IF EXISTS**: Read quickstart.md for integration scenarios
|
||||
|
||||
4. **Project Setup Verification**:
|
||||
- **REQUIRED**: Create/verify ignore files based on actual project setup:
|
||||
|
||||
**Detection & Creation Logic**:
|
||||
- Check if the following command succeeds to determine if the repository is a git repo (create/verify .gitignore if so):
|
||||
|
||||
```sh
|
||||
git rev-parse --git-dir 2>/dev/null
|
||||
```
|
||||
|
||||
- Check if Dockerfile* exists or Docker in plan.md → create/verify .dockerignore
|
||||
- Check if .eslintrc* exists → create/verify .eslintignore
|
||||
- Check if eslint.config.* exists → ensure the config's `ignores` entries cover required patterns
|
||||
- Check if .prettierrc* exists → create/verify .prettierignore
|
||||
- Check if .npmrc or package.json exists → create/verify .npmignore (if publishing)
|
||||
- Check if terraform files (*.tf) exist → create/verify .terraformignore
|
||||
- Check if .helmignore needed (helm charts present) → create/verify .helmignore
|
||||
|
||||
**If ignore file already exists**: Verify it contains essential patterns, append missing critical patterns only
|
||||
**If ignore file missing**: Create with full pattern set for detected technology
|
||||
|
||||
**Common Patterns by Technology** (from plan.md tech stack):
|
||||
- **Node.js/JavaScript/TypeScript**: `node_modules/`, `dist/`, `build/`, `*.log`, `.env*`
|
||||
- **Python**: `__pycache__/`, `*.pyc`, `.venv/`, `venv/`, `dist/`, `*.egg-info/`
|
||||
- **Java**: `target/`, `*.class`, `*.jar`, `.gradle/`, `build/`
|
||||
- **C#/.NET**: `bin/`, `obj/`, `*.user`, `*.suo`, `packages/`
|
||||
- **Go**: `*.exe`, `*.test`, `vendor/`, `*.out`
|
||||
- **Ruby**: `.bundle/`, `log/`, `tmp/`, `*.gem`, `vendor/bundle/`
|
||||
- **PHP**: `vendor/`, `*.log`, `*.cache`, `*.env`
|
||||
- **Rust**: `target/`, `debug/`, `release/`, `*.rs.bk`, `*.rlib`, `*.prof*`, `.idea/`, `*.log`, `.env*`
|
||||
- **Kotlin**: `build/`, `out/`, `.gradle/`, `.idea/`, `*.class`, `*.jar`, `*.iml`, `*.log`, `.env*`
|
||||
- **C++**: `build/`, `bin/`, `obj/`, `out/`, `*.o`, `*.so`, `*.a`, `*.exe`, `*.dll`, `.idea/`, `*.log`, `.env*`
|
||||
- **C**: `build/`, `bin/`, `obj/`, `out/`, `*.o`, `*.a`, `*.so`, `*.exe`, `*.dll`, `autom4te.cache/`, `config.status`, `config.log`, `.idea/`, `*.log`, `.env*`
|
||||
- **Swift**: `.build/`, `DerivedData/`, `*.swiftpm/`, `Packages/`
|
||||
- **R**: `.Rproj.user/`, `.Rhistory`, `.RData`, `.Ruserdata`, `*.Rproj`, `packrat/`, `renv/`
|
||||
- **Universal**: `.DS_Store`, `Thumbs.db`, `*.tmp`, `*.swp`, `.vscode/`, `.idea/`
|
||||
|
||||
**Tool-Specific Patterns**:
|
||||
- **Docker**: `node_modules/`, `.git/`, `Dockerfile*`, `.dockerignore`, `*.log*`, `.env*`, `coverage/`
|
||||
- **ESLint**: `node_modules/`, `dist/`, `build/`, `coverage/`, `*.min.js`
|
||||
- **Prettier**: `node_modules/`, `dist/`, `build/`, `coverage/`, `package-lock.json`, `yarn.lock`, `pnpm-lock.yaml`
|
||||
- **Terraform**: `.terraform/`, `*.tfstate*`, `*.tfvars`, `.terraform.lock.hcl`
|
||||
- **Kubernetes/k8s**: `*.secret.yaml`, `secrets/`, `.kube/`, `kubeconfig*`, `*.key`, `*.crt`
|
||||
|
||||
5. Parse tasks.md structure and extract:
|
||||
- **Task phases**: Setup, Tests, Core, Integration, Polish
|
||||
- **Task dependencies**: Sequential vs parallel execution rules
|
||||
- **Task details**: ID, description, file paths, parallel markers [P]
|
||||
- **Execution flow**: Order and dependency requirements
|
||||
|
||||
6. Execute implementation following the task plan:
|
||||
- **Phase-by-phase execution**: Complete each phase before moving to the next
|
||||
- **Respect dependencies**: Run sequential tasks in order, parallel tasks [P] can run together
|
||||
- **Follow TDD approach**: Execute test tasks before their corresponding implementation tasks
|
||||
- **File-based coordination**: Tasks affecting the same files must run sequentially
|
||||
- **Validation checkpoints**: Verify each phase completion before proceeding
|
||||
|
||||
7. Implementation execution rules:
|
||||
- **Setup first**: Initialize project structure, dependencies, configuration
|
||||
- **Tests before code**: If you need to write tests for contracts, entities, and integration scenarios
|
||||
- **Core development**: Implement models, services, CLI commands, endpoints
|
||||
- **Integration work**: Database connections, middleware, logging, external services
|
||||
- **Polish and validation**: Unit tests, performance optimization, documentation
|
||||
|
||||
8. Progress tracking and error handling:
|
||||
- Report progress after each completed task
|
||||
- Halt execution if any non-parallel task fails
|
||||
- For parallel tasks [P], continue with successful tasks, report failed ones
|
||||
- Provide clear error messages with context for debugging
|
||||
- Suggest next steps if implementation cannot proceed
|
||||
- **IMPORTANT** For completed tasks, make sure to mark the task off as [X] in the tasks file.
|
||||
|
||||
9. Completion validation:
|
||||
- Verify all required tasks are completed
|
||||
- Check that implemented features match the original specification
|
||||
- Validate that tests pass and coverage meets requirements
|
||||
- Confirm the implementation follows the technical plan
|
||||
- Report final status with summary of completed work
|
||||
|
||||
Note: This command assumes a complete task breakdown exists in tasks.md. If tasks are incomplete or missing, suggest running `/speckit.tasks` first to regenerate the task list.
|
||||
|
||||
10. **Check for extension hooks**: After completion validation, check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_implement` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
@@ -0,0 +1,151 @@
|
||||
---
|
||||
name: speckit-plan
|
||||
description: Execute the implementation planning workflow using the plan template to generate design artifacts.
|
||||
argument-hint: "Optional guidance for the planning phase"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/plan.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before planning)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_plan` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Outline.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Outline
|
||||
|
||||
1. **Setup**: Run `{SCRIPT}` from repo root and parse JSON for FEATURE_SPEC, IMPL_PLAN, SPECS_DIR, BRANCH. For single quotes in args like "I'm Groot", use escape syntax: e.g 'I'\''m Groot' (or double-quote if possible: "I'm Groot").
|
||||
|
||||
2. **Load context**: Read FEATURE_SPEC and `/memory/constitution.md`. Load IMPL_PLAN template (already copied).
|
||||
|
||||
3. **Execute plan workflow**: Follow the structure in IMPL_PLAN template to:
|
||||
- Fill Technical Context (mark unknowns as "NEEDS CLARIFICATION")
|
||||
- Fill Constitution Check section from constitution
|
||||
- Evaluate gates (ERROR if violations unjustified)
|
||||
- Phase 0: Generate research.md (resolve all NEEDS CLARIFICATION)
|
||||
- Phase 1: Generate data-model.md, contracts/, quickstart.md
|
||||
- Phase 1: Update agent context by running the agent script
|
||||
- Re-evaluate Constitution Check post-design
|
||||
|
||||
4. **Stop and report**: Command ends after Phase 2 planning. Report branch, IMPL_PLAN path, and generated artifacts.
|
||||
|
||||
5. **Check for extension hooks**: After reporting, check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_plan` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Phases
|
||||
|
||||
### Phase 0: Outline & Research
|
||||
|
||||
1. **Extract unknowns from Technical Context** above:
|
||||
- For each NEEDS CLARIFICATION → research task
|
||||
- For each dependency → best practices task
|
||||
- For each integration → patterns task
|
||||
|
||||
2. **Generate and dispatch research agents**:
|
||||
|
||||
```text
|
||||
For each unknown in Technical Context:
|
||||
Task: "Research {unknown} for {feature context}"
|
||||
For each technology choice:
|
||||
Task: "Find best practices for {tech} in {domain}"
|
||||
```
|
||||
|
||||
3. **Consolidate findings** in `research.md` using format:
|
||||
- Decision: [what was chosen]
|
||||
- Rationale: [why chosen]
|
||||
- Alternatives considered: [what else evaluated]
|
||||
|
||||
**Output**: research.md with all NEEDS CLARIFICATION resolved
|
||||
|
||||
### Phase 1: Design & Contracts
|
||||
|
||||
**Prerequisites:** `research.md` complete
|
||||
|
||||
1. **Extract entities from feature spec** → `data-model.md`:
|
||||
- Entity name, fields, relationships
|
||||
- Validation rules from requirements
|
||||
- State transitions if applicable
|
||||
|
||||
2. **Define interface contracts** (if project has external interfaces) → `/contracts/`:
|
||||
- Identify what interfaces the project exposes to users or other systems
|
||||
- Document the contract format appropriate for the project type
|
||||
- Examples: public APIs for libraries, command schemas for CLI tools, endpoints for web services, grammars for parsers, UI contracts for applications
|
||||
- Skip if project is purely internal (build scripts, one-off tools, etc.)
|
||||
|
||||
3. **Agent context update**:
|
||||
- Update the plan reference between the `<!-- SPECKIT START -->` and `<!-- SPECKIT END -->` markers in `__CONTEXT_FILE__` to point to the plan file created in step 1 (the IMPL_PLAN path)
|
||||
|
||||
**Output**: data-model.md, /contracts/*, quickstart.md, updated agent context file
|
||||
|
||||
## Key rules
|
||||
|
||||
- Use absolute paths for filesystem operations; use project-relative paths for references in documentation and agent context files
|
||||
- ERROR on gate failures or unresolved clarifications
|
||||
@@ -0,0 +1,329 @@
|
||||
---
|
||||
name: speckit-specify
|
||||
description: Create or update the feature specification from a natural language feature description.
|
||||
argument-hint: "Describe the feature you want to specify"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/specify.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before specification)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_specify` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Outline.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Outline
|
||||
|
||||
The text the user typed after `/speckit.specify` in the triggering message **is** the feature description. Assume you always have it available in this conversation even if `{ARGS}` appears literally below. Do not ask the user to repeat it unless they provided an empty command.
|
||||
|
||||
Given that feature description, do this:
|
||||
|
||||
1. **Generate a concise short name** (2-4 words) for the feature:
|
||||
- Analyze the feature description and extract the most meaningful keywords
|
||||
- Create a 2-4 word short name that captures the essence of the feature
|
||||
- Use action-noun format when possible (e.g., "add-user-auth", "fix-payment-bug")
|
||||
- Preserve technical terms and acronyms (OAuth2, API, JWT, etc.)
|
||||
- Keep it concise but descriptive enough to understand the feature at a glance
|
||||
- Examples:
|
||||
- "I want to add user authentication" → "user-auth"
|
||||
- "Implement OAuth2 integration for the API" → "oauth2-api-integration"
|
||||
- "Create a dashboard for analytics" → "analytics-dashboard"
|
||||
- "Fix payment processing timeout bug" → "fix-payment-timeout"
|
||||
|
||||
2. **Branch creation** (optional, via hook):
|
||||
|
||||
If a `before_specify` hook ran successfully in the Pre-Execution Checks above, it will have created/switched to a git branch and output JSON containing `BRANCH_NAME` and `FEATURE_NUM`. Note these values for reference, but the branch name does **not** dictate the spec directory name.
|
||||
|
||||
If the user explicitly provided `GIT_BRANCH_NAME`, pass it through to the hook so the branch script uses the exact value as the branch name (bypassing all prefix/suffix generation).
|
||||
|
||||
3. **Create the spec feature directory**:
|
||||
|
||||
Specs live under the default `specs/` directory unless the user explicitly provides `SPECIFY_FEATURE_DIRECTORY`.
|
||||
|
||||
**Resolution order for `SPECIFY_FEATURE_DIRECTORY`**:
|
||||
1. If the user explicitly provided `SPECIFY_FEATURE_DIRECTORY` (e.g., via environment variable, argument, or configuration), use it as-is
|
||||
2. Otherwise, auto-generate it under `specs/`:
|
||||
- Check `.specify/init-options.json` for `branch_numbering`
|
||||
- If `"timestamp"`: prefix is `YYYYMMDD-HHMMSS` (current timestamp)
|
||||
- If `"sequential"` or absent: prefix is `NNN` (next available 3-digit number after scanning existing directories in `specs/`)
|
||||
- Construct the directory name: `<prefix>-<short-name>` (e.g., `003-user-auth` or `20260319-143022-user-auth`)
|
||||
- Set `SPECIFY_FEATURE_DIRECTORY` to `specs/<directory-name>`
|
||||
|
||||
**Create the directory and spec file**:
|
||||
- `mkdir -p SPECIFY_FEATURE_DIRECTORY`
|
||||
- Copy `templates/spec-template.md` to `SPECIFY_FEATURE_DIRECTORY/spec.md` as the starting point
|
||||
- Set `SPEC_FILE` to `SPECIFY_FEATURE_DIRECTORY/spec.md`
|
||||
- Persist the resolved path to `.specify/feature.json`:
|
||||
```json
|
||||
{
|
||||
"feature_directory": "<resolved feature dir>"
|
||||
}
|
||||
```
|
||||
Write the actual resolved directory path value (for example, `specs/003-user-auth`), not the literal string `SPECIFY_FEATURE_DIRECTORY`.
|
||||
This allows downstream commands (`/speckit.plan`, `/speckit.tasks`, etc.) to locate the feature directory without relying on git branch name conventions.
|
||||
|
||||
**IMPORTANT**:
|
||||
- You must only create one feature per `/speckit.specify` invocation
|
||||
- The spec directory name and the git branch name are independent — they may be the same but that is the user's choice
|
||||
- The spec directory and file are always created by this command, never by the hook
|
||||
|
||||
4. Load `templates/spec-template.md` to understand required sections.
|
||||
|
||||
5. Follow this execution flow:
|
||||
1. Parse user description from arguments
|
||||
If empty: ERROR "No feature description provided"
|
||||
2. Extract key concepts from description
|
||||
Identify: actors, actions, data, constraints
|
||||
3. For unclear aspects:
|
||||
- Make informed guesses based on context and industry standards
|
||||
- Only mark with [NEEDS CLARIFICATION: specific question] if:
|
||||
- The choice significantly impacts feature scope or user experience
|
||||
- Multiple reasonable interpretations exist with different implications
|
||||
- No reasonable default exists
|
||||
- **LIMIT: Maximum 3 [NEEDS CLARIFICATION] markers total**
|
||||
- Prioritize clarifications by impact: scope > security/privacy > user experience > technical details
|
||||
4. Fill User Scenarios & Testing section
|
||||
If no clear user flow: ERROR "Cannot determine user scenarios"
|
||||
5. Generate Functional Requirements
|
||||
Each requirement must be testable
|
||||
Use reasonable defaults for unspecified details (document assumptions in Assumptions section)
|
||||
6. Define Success Criteria
|
||||
Create measurable, technology-agnostic outcomes
|
||||
Include both quantitative metrics (time, performance, volume) and qualitative measures (user satisfaction, task completion)
|
||||
Each criterion must be verifiable without implementation details
|
||||
7. Identify Key Entities (if data involved)
|
||||
8. Return: SUCCESS (spec ready for planning)
|
||||
|
||||
6. Write the specification to SPEC_FILE using the template structure, replacing placeholders with concrete details derived from the feature description (arguments) while preserving section order and headings.
|
||||
|
||||
7. **Specification Quality Validation**: After writing the initial spec, validate it against quality criteria:
|
||||
|
||||
a. **Create Spec Quality Checklist**: Generate a checklist file at `SPECIFY_FEATURE_DIRECTORY/checklists/requirements.md` using the checklist template structure with these validation items:
|
||||
|
||||
```markdown
|
||||
# Specification Quality Checklist: [FEATURE NAME]
|
||||
|
||||
**Purpose**: Validate specification completeness and quality before proceeding to planning
|
||||
**Created**: [DATE]
|
||||
**Feature**: [Link to spec.md]
|
||||
|
||||
## Content Quality
|
||||
|
||||
- [ ] No implementation details (languages, frameworks, APIs)
|
||||
- [ ] Focused on user value and business needs
|
||||
- [ ] Written for non-technical stakeholders
|
||||
- [ ] All mandatory sections completed
|
||||
|
||||
## Requirement Completeness
|
||||
|
||||
- [ ] No [NEEDS CLARIFICATION] markers remain
|
||||
- [ ] Requirements are testable and unambiguous
|
||||
- [ ] Success criteria are measurable
|
||||
- [ ] Success criteria are technology-agnostic (no implementation details)
|
||||
- [ ] All acceptance scenarios are defined
|
||||
- [ ] Edge cases are identified
|
||||
- [ ] Scope is clearly bounded
|
||||
- [ ] Dependencies and assumptions identified
|
||||
|
||||
## Feature Readiness
|
||||
|
||||
- [ ] All functional requirements have clear acceptance criteria
|
||||
- [ ] User scenarios cover primary flows
|
||||
- [ ] Feature meets measurable outcomes defined in Success Criteria
|
||||
- [ ] No implementation details leak into specification
|
||||
|
||||
## Notes
|
||||
|
||||
- Items marked incomplete require spec updates before `/speckit.clarify` or `/speckit.plan`
|
||||
```
|
||||
|
||||
b. **Run Validation Check**: Review the spec against each checklist item:
|
||||
- For each item, determine if it passes or fails
|
||||
- Document specific issues found (quote relevant spec sections)
|
||||
|
||||
c. **Handle Validation Results**:
|
||||
|
||||
- **If all items pass**: Mark checklist complete and proceed to step 7
|
||||
|
||||
- **If items fail (excluding [NEEDS CLARIFICATION])**:
|
||||
1. List the failing items and specific issues
|
||||
2. Update the spec to address each issue
|
||||
3. Re-run validation until all items pass (max 3 iterations)
|
||||
4. If still failing after 3 iterations, document remaining issues in checklist notes and warn user
|
||||
|
||||
- **If [NEEDS CLARIFICATION] markers remain**:
|
||||
1. Extract all [NEEDS CLARIFICATION: ...] markers from the spec
|
||||
2. **LIMIT CHECK**: If more than 3 markers exist, keep only the 3 most critical (by scope/security/UX impact) and make informed guesses for the rest
|
||||
3. For each clarification needed (max 3), present options to user in this format:
|
||||
|
||||
```markdown
|
||||
## Question [N]: [Topic]
|
||||
|
||||
**Context**: [Quote relevant spec section]
|
||||
|
||||
**What we need to know**: [Specific question from NEEDS CLARIFICATION marker]
|
||||
|
||||
**Suggested Answers**:
|
||||
|
||||
| Option | Answer | Implications |
|
||||
|--------|--------|--------------|
|
||||
| A | [First suggested answer] | [What this means for the feature] |
|
||||
| B | [Second suggested answer] | [What this means for the feature] |
|
||||
| C | [Third suggested answer] | [What this means for the feature] |
|
||||
| Custom | Provide your own answer | [Explain how to provide custom input] |
|
||||
|
||||
**Your choice**: _[Wait for user response]_
|
||||
```
|
||||
|
||||
4. **CRITICAL - Table Formatting**: Ensure markdown tables are properly formatted:
|
||||
- Use consistent spacing with pipes aligned
|
||||
- Each cell should have spaces around content: `| Content |` not `|Content|`
|
||||
- Header separator must have at least 3 dashes: `|--------|`
|
||||
- Test that the table renders correctly in markdown preview
|
||||
5. Number questions sequentially (Q1, Q2, Q3 - max 3 total)
|
||||
6. Present all questions together before waiting for responses
|
||||
7. Wait for user to respond with their choices for all questions (e.g., "Q1: A, Q2: Custom - [details], Q3: B")
|
||||
8. Update the spec by replacing each [NEEDS CLARIFICATION] marker with the user's selected or provided answer
|
||||
9. Re-run validation after all clarifications are resolved
|
||||
|
||||
d. **Update Checklist**: After each validation iteration, update the checklist file with current pass/fail status
|
||||
|
||||
8. **Report completion** to the user with:
|
||||
- `SPECIFY_FEATURE_DIRECTORY` — the feature directory path
|
||||
- `SPEC_FILE` — the spec file path
|
||||
- Checklist results summary
|
||||
- Readiness for the next phase (`/speckit.clarify` or `/speckit.plan`)
|
||||
|
||||
9. **Check for extension hooks**: After reporting completion, check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_specify` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
**NOTE:** Branch creation is handled by the `before_specify` hook (git extension). Spec directory and file creation are always handled by this core command.
|
||||
|
||||
## Quick Guidelines
|
||||
|
||||
- Focus on **WHAT** users need and **WHY**.
|
||||
- Avoid HOW to implement (no tech stack, APIs, code structure).
|
||||
- Written for business stakeholders, not developers.
|
||||
- DO NOT create any checklists that are embedded in the spec. That will be a separate command.
|
||||
|
||||
### Section Requirements
|
||||
|
||||
- **Mandatory sections**: Must be completed for every feature
|
||||
- **Optional sections**: Include only when relevant to the feature
|
||||
- When a section doesn't apply, remove it entirely (don't leave as "N/A")
|
||||
|
||||
### For AI Generation
|
||||
|
||||
When creating this spec from a user prompt:
|
||||
|
||||
1. **Make informed guesses**: Use context, industry standards, and common patterns to fill gaps
|
||||
2. **Document assumptions**: Record reasonable defaults in the Assumptions section
|
||||
3. **Limit clarifications**: Maximum 3 [NEEDS CLARIFICATION] markers - use only for critical decisions that:
|
||||
- Significantly impact feature scope or user experience
|
||||
- Have multiple reasonable interpretations with different implications
|
||||
- Lack any reasonable default
|
||||
4. **Prioritize clarifications**: scope > security/privacy > user experience > technical details
|
||||
5. **Think like a tester**: Every vague requirement should fail the "testable and unambiguous" checklist item
|
||||
6. **Common areas needing clarification** (only if no reasonable default exists):
|
||||
- Feature scope and boundaries (include/exclude specific use cases)
|
||||
- User types and permissions (if multiple conflicting interpretations possible)
|
||||
- Security/compliance requirements (when legally/financially significant)
|
||||
|
||||
**Examples of reasonable defaults** (don't ask about these):
|
||||
|
||||
- Data retention: Industry-standard practices for the domain
|
||||
- Performance targets: Standard web/mobile app expectations unless specified
|
||||
- Error handling: User-friendly messages with appropriate fallbacks
|
||||
- Authentication method: Standard session-based or OAuth2 for web apps
|
||||
- Integration patterns: Use project-appropriate patterns (REST/GraphQL for web services, function calls for libraries, CLI args for tools, etc.)
|
||||
|
||||
### Success Criteria Guidelines
|
||||
|
||||
Success criteria must be:
|
||||
|
||||
1. **Measurable**: Include specific metrics (time, percentage, count, rate)
|
||||
2. **Technology-agnostic**: No mention of frameworks, languages, databases, or tools
|
||||
3. **User-focused**: Describe outcomes from user/business perspective, not system internals
|
||||
4. **Verifiable**: Can be tested/validated without knowing implementation details
|
||||
|
||||
**Good examples**:
|
||||
|
||||
- "Users can complete checkout in under 3 minutes"
|
||||
- "System supports 10,000 concurrent users"
|
||||
- "95% of searches return results in under 1 second"
|
||||
- "Task completion rate improves by 40%"
|
||||
|
||||
**Bad examples** (implementation-focused):
|
||||
|
||||
- "API response time is under 200ms" (too technical, use "Users see results instantly")
|
||||
- "Database can handle 1000 TPS" (implementation detail, use user-facing metric)
|
||||
- "React components render efficiently" (framework-specific)
|
||||
- "Redis cache hit rate above 80%" (technology-specific)
|
||||
@@ -0,0 +1,201 @@
|
||||
---
|
||||
name: speckit-tasks
|
||||
description: Generate an actionable, dependency-ordered tasks.md for the feature based on available design artifacts.
|
||||
argument-hint: "Optional task generation constraints"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/tasks.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before tasks generation)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_tasks` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Outline.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Outline
|
||||
|
||||
1. **Setup**: Run `{SCRIPT}` from repo root and parse FEATURE_DIR and AVAILABLE_DOCS list. All paths must be absolute. For single quotes in args like "I'm Groot", use escape syntax: e.g 'I'\''m Groot' (or double-quote if possible: "I'm Groot").
|
||||
|
||||
2. **Load design documents**: Read from FEATURE_DIR:
|
||||
- **Required**: plan.md (tech stack, libraries, structure), spec.md (user stories with priorities)
|
||||
- **Optional**: data-model.md (entities), contracts/ (interface contracts), research.md (decisions), quickstart.md (test scenarios)
|
||||
- Note: Not all projects have all documents. Generate tasks based on what's available.
|
||||
|
||||
3. **Execute task generation workflow**:
|
||||
- Load plan.md and extract tech stack, libraries, project structure
|
||||
- Load spec.md and extract user stories with their priorities (P1, P2, P3, etc.)
|
||||
- If data-model.md exists: Extract entities and map to user stories
|
||||
- If contracts/ exists: Map interface contracts to user stories
|
||||
- If research.md exists: Extract decisions for setup tasks
|
||||
- Generate tasks organized by user story (see Task Generation Rules below)
|
||||
- Generate dependency graph showing user story completion order
|
||||
- Create parallel execution examples per user story
|
||||
- Validate task completeness (each user story has all needed tasks, independently testable)
|
||||
|
||||
4. **Generate tasks.md**: Use `templates/tasks-template.md` as structure, fill with:
|
||||
- Correct feature name from plan.md
|
||||
- Phase 1: Setup tasks (project initialization)
|
||||
- Phase 2: Foundational tasks (blocking prerequisites for all user stories)
|
||||
- Phase 3+: One phase per user story (in priority order from spec.md)
|
||||
- Each phase includes: story goal, independent test criteria, tests (if requested), implementation tasks
|
||||
- Final Phase: Polish & cross-cutting concerns
|
||||
- All tasks must follow the strict checklist format (see Task Generation Rules below)
|
||||
- Clear file paths for each task
|
||||
- Dependencies section showing story completion order
|
||||
- Parallel execution examples per story
|
||||
- Implementation strategy section (MVP first, incremental delivery)
|
||||
|
||||
5. **Report**: Output path to generated tasks.md and summary:
|
||||
- Total task count
|
||||
- Task count per user story
|
||||
- Parallel opportunities identified
|
||||
- Independent test criteria for each story
|
||||
- Suggested MVP scope (typically just User Story 1)
|
||||
- Format validation: Confirm ALL tasks follow the checklist format (checkbox, ID, labels, file paths)
|
||||
|
||||
6. **Check for extension hooks**: After tasks.md is generated, check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_tasks` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
Context for task generation: {ARGS}
|
||||
|
||||
The tasks.md should be immediately executable - each task must be specific enough that an LLM can complete it without additional context.
|
||||
|
||||
## Task Generation Rules
|
||||
|
||||
**CRITICAL**: Tasks MUST be organized by user story to enable independent implementation and testing.
|
||||
|
||||
**Tests are OPTIONAL**: Only generate test tasks if explicitly requested in the feature specification or if user requests TDD approach.
|
||||
|
||||
### Checklist Format (REQUIRED)
|
||||
|
||||
Every task MUST strictly follow this format:
|
||||
|
||||
```text
|
||||
- [ ] [TaskID] [P?] [Story?] Description with file path
|
||||
```
|
||||
|
||||
**Format Components**:
|
||||
|
||||
1. **Checkbox**: ALWAYS start with `- [ ]` (markdown checkbox)
|
||||
2. **Task ID**: Sequential number (T001, T002, T003...) in execution order
|
||||
3. **[P] marker**: Include ONLY if task is parallelizable (different files, no dependencies on incomplete tasks)
|
||||
4. **[Story] label**: REQUIRED for user story phase tasks only
|
||||
- Format: [US1], [US2], [US3], etc. (maps to user stories from spec.md)
|
||||
- Setup phase: NO story label
|
||||
- Foundational phase: NO story label
|
||||
- User Story phases: MUST have story label
|
||||
- Polish phase: NO story label
|
||||
5. **Description**: Clear action with exact file path
|
||||
|
||||
**Examples**:
|
||||
|
||||
- ✅ CORRECT: `- [ ] T001 Create project structure per implementation plan`
|
||||
- ✅ CORRECT: `- [ ] T005 [P] Implement authentication middleware in src/middleware/auth.py`
|
||||
- ✅ CORRECT: `- [ ] T012 [P] [US1] Create User model in src/models/user.py`
|
||||
- ✅ CORRECT: `- [ ] T014 [US1] Implement UserService in src/services/user_service.py`
|
||||
- ❌ WRONG: `- [ ] Create User model` (missing ID and Story label)
|
||||
- ❌ WRONG: `T001 [US1] Create model` (missing checkbox)
|
||||
- ❌ WRONG: `- [ ] [US1] Create User model` (missing Task ID)
|
||||
- ❌ WRONG: `- [ ] T001 [US1] Create model` (missing file path)
|
||||
|
||||
### Task Organization
|
||||
|
||||
1. **From User Stories (spec.md)** - PRIMARY ORGANIZATION:
|
||||
- Each user story (P1, P2, P3...) gets its own phase
|
||||
- Map all related components to their story:
|
||||
- Models needed for that story
|
||||
- Services needed for that story
|
||||
- Interfaces/UI needed for that story
|
||||
- If tests requested: Tests specific to that story
|
||||
- Mark story dependencies (most stories should be independent)
|
||||
|
||||
2. **From Contracts**:
|
||||
- Map each interface contract → to the user story it serves
|
||||
- If tests requested: Each interface contract → contract test task [P] before implementation in that story's phase
|
||||
|
||||
3. **From Data Model**:
|
||||
- Map each entity to the user story(ies) that need it
|
||||
- If entity serves multiple stories: Put in earliest story or Setup phase
|
||||
- Relationships → service layer tasks in appropriate story phase
|
||||
|
||||
4. **From Setup/Infrastructure**:
|
||||
- Shared infrastructure → Setup phase (Phase 1)
|
||||
- Foundational/blocking tasks → Foundational phase (Phase 2)
|
||||
- Story-specific setup → within that story's phase
|
||||
|
||||
### Phase Structure
|
||||
|
||||
- **Phase 1**: Setup (project initialization)
|
||||
- **Phase 2**: Foundational (blocking prerequisites - MUST complete before user stories)
|
||||
- **Phase 3+**: User Stories in priority order (P1, P2, P3...)
|
||||
- Within each story: Tests (if requested) → Models → Services → Endpoints → Integration
|
||||
- Each phase should be a complete, independently testable increment
|
||||
- **Final Phase**: Polish & Cross-Cutting Concerns
|
||||
@@ -0,0 +1,105 @@
|
||||
---
|
||||
name: speckit-taskstoissues
|
||||
description: Convert existing tasks into actionable, dependency-ordered GitHub issues for the feature based on available design artifacts.
|
||||
argument-hint: "Optional filter or label for GitHub issues"
|
||||
user-invocable: true
|
||||
disable-model-invocation: false
|
||||
compatibility: "Requires spec-kit project structure with .specify/ directory"
|
||||
metadata:
|
||||
author: github-spec-kit
|
||||
source: claude:templates/commands/taskstoissues.md
|
||||
---
|
||||
|
||||
## User Input
|
||||
|
||||
```text
|
||||
$ARGUMENTS
|
||||
```
|
||||
|
||||
You **MUST** consider the user input before proceeding (if not empty).
|
||||
|
||||
## Pre-Execution Checks
|
||||
|
||||
**Check for extension hooks (before tasks-to-issues conversion)**:
|
||||
- Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.before_taskstoissues` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Pre-Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Pre-Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
|
||||
Wait for the result of the hook command before proceeding to the Outline.
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
|
||||
## Outline
|
||||
|
||||
1. Run `{SCRIPT}` from repo root and parse FEATURE_DIR and AVAILABLE_DOCS list. All paths must be absolute. For single quotes in args like "I'm Groot", use escape syntax: e.g 'I'\''m Groot' (or double-quote if possible: "I'm Groot").
|
||||
1. From the executed script, extract the path to **tasks**.
|
||||
1. Get the Git remote by running:
|
||||
|
||||
```bash
|
||||
git config --get remote.origin.url
|
||||
```
|
||||
|
||||
> [!CAUTION]
|
||||
> ONLY PROCEED TO NEXT STEPS IF THE REMOTE IS A GITHUB URL
|
||||
|
||||
1. For each task in the list, use the GitHub MCP server to create a new issue in the repository that is representative of the Git remote.
|
||||
|
||||
> [!CAUTION]
|
||||
> UNDER NO CIRCUMSTANCES EVER CREATE ISSUES IN REPOSITORIES THAT DO NOT MATCH THE REMOTE URL
|
||||
|
||||
## Post-Execution Checks
|
||||
|
||||
**Check for extension hooks (after tasks-to-issues conversion)**:
|
||||
Check if `.specify/extensions.yml` exists in the project root.
|
||||
- If it exists, read it and look for entries under the `hooks.after_taskstoissues` key
|
||||
- If the YAML cannot be parsed or is invalid, skip hook checking silently and continue normally
|
||||
- Filter out hooks where `enabled` is explicitly `false`. Treat hooks without an `enabled` field as enabled by default.
|
||||
- For each remaining hook, do **not** attempt to interpret or evaluate hook `condition` expressions:
|
||||
- If the hook has no `condition` field, or it is null/empty, treat the hook as executable
|
||||
- If the hook defines a non-empty `condition`, skip the hook and leave condition evaluation to the HookExecutor implementation
|
||||
- When constructing slash commands from hook command names, replace dots (`.`) with hyphens (`-`). For example, `speckit.git.commit` -> `/speckit-git-commit`.
|
||||
- For each executable hook, output the following based on its `optional` flag:
|
||||
- **Optional hook** (`optional: true`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Optional Hook**: {extension}
|
||||
Command: `/{command}`
|
||||
Description: {description}
|
||||
|
||||
Prompt: {prompt}
|
||||
To execute: `/{command}`
|
||||
```
|
||||
- **Mandatory hook** (`optional: false`):
|
||||
```
|
||||
## Extension Hooks
|
||||
|
||||
**Automatic Hook**: {extension}
|
||||
Executing: `/{command}`
|
||||
EXECUTE_COMMAND: {command}
|
||||
```
|
||||
- If no hooks are registered or `.specify/extensions.yml` does not exist, skip silently
|
||||
@@ -1,8 +0,0 @@
|
||||
.git
|
||||
.gitignore
|
||||
*.md
|
||||
node_modules
|
||||
.env
|
||||
.env.*
|
||||
scripts/
|
||||
start.sh
|
||||
@@ -0,0 +1,9 @@
|
||||
# GitHub recognizes this file and shows a "Sponsor" button on the repo page.
|
||||
# Add other platforms here as they get set up. Empty / commented-out entries
|
||||
# are skipped silently.
|
||||
|
||||
buy_me_a_coffee: dtzp555
|
||||
|
||||
# github: [dtzp555-max] # uncomment after GitHub Sponsors enrollment is approved
|
||||
# ko_fi: dtzp555
|
||||
# custom: ["https://example.com/donate"]
|
||||
@@ -0,0 +1,82 @@
|
||||
# Pull Request
|
||||
|
||||
## Summary
|
||||
|
||||
<!-- One or two sentences describing the change and why it is in scope for OCP. -->
|
||||
|
||||
## Endpoint Class (REQUIRED)
|
||||
|
||||
Per `ALIGNMENT.md` and ADR 0006, every PR that touches a network-facing endpoint must declare its class. Pick the most specific applicable class (Hybrid covers PRs that touch both A and B):
|
||||
|
||||
- [ ] **Class A** — forwards a `cli.js` operation (e.g., `/v1/messages`, `/api/oauth/*`, or the Anthropic-side wire call inside `/usage`)
|
||||
- [ ] **Class B** — extends an OCP-owned compatibility endpoint (per ADR 0006). Sub-bucket:
|
||||
- [ ] B.1 — OpenAI-compatibility surface (`/v1/chat/completions`, `/v1/models`)
|
||||
- [ ] B.2 — OCP-administrative surface (`/health`, `/dashboard`, `/sessions`, `/logs`, `/status`, `/settings`, `/api/keys*`, `/api/usage`, `/cache*`)
|
||||
- [ ] **Hybrid** — touches both classes (e.g., `/usage` if the PR modifies both the Anthropic wire call AND the local synthesis layer). Both evidence sections below must be filled.
|
||||
- [ ] **Not endpoint-touching** — refactor / docs / tooling that does not modify any request handler. Skip both evidence sections; explain in Summary.
|
||||
|
||||
## Claude Code Alignment Evidence (REQUIRED)
|
||||
|
||||
PRs with the relevant evidence section blank or unchecked will receive a `request changes` review and cannot be merged.
|
||||
|
||||
### If Class A
|
||||
|
||||
- [ ] **Corresponding `cli.js` reference.** I have identified the `cli.js` function and line range that performs the operation this PR forwards. Citation (format `cli.js:NNNN` or `cli.js vE4 <functionName>`):
|
||||
<!-- e.g. cli.js:18423-18467 (function: sendUserMessage) -->
|
||||
|
||||
- [ ] **If `cli.js` does not perform this operation**, I have stated this explicitly below and justified the scope under `ALIGNMENT.md` Rule 2. (Note: in almost all cases this means the PR should be closed, not merged. Proxy layers do not invent endpoints. If the endpoint is in fact Class B, switch the class above and use the Class B section instead.)
|
||||
<!-- Justification, if applicable. Empty is fine when cli.js does perform the operation. -->
|
||||
|
||||
- [ ] **Commit message citations.** Every "Claude Code uses X" or "cli.js uses X" assertion in every commit of this PR is immediately followed by a `cli.js:NNNN` or `cli.js vE4 <functionName>` citation. I have verified this by rereading each commit message.
|
||||
|
||||
### If Class B
|
||||
|
||||
- [ ] **Authorizing ADR.** Cite the ADR number that authorizes the endpoint this PR modifies (e.g., "ADR 0006 — OpenAI shim scope"). For B.1 endpoints (`/v1/chat/completions`, `/v1/models`), this is ADR 0006. For grandfathered B.2 endpoints, this is "ADR 0006 (grandfathered as of v3.16.4)." For new B.2 endpoints, cite the endpoint's own authorizing ADR; if none exists, the PR cannot proceed — the authorizing ADR must be drafted and merged first.
|
||||
<!-- e.g., ADR 0006 -->
|
||||
|
||||
- [ ] **Specification citation.** For B.1 endpoints, link to the relevant section of OpenAI's `/v1/chat/completions` specification (https://platform.openai.com/docs/api-reference/chat/create), including the specific field or behaviour being implemented. For B.2 endpoints with their own ADR, cite the ADR section that specifies the behaviour. For grandfathered B.2 endpoints, the PR must be a behaviour-preserving refactor — link the existing handler code being modified.
|
||||
<!-- B.1 example: OpenAI chat/completions, `response_format` parameter, https://platform.openai.com/docs/api-reference/chat/create#chat-create-response_format -->
|
||||
<!-- B.2 example: ADR 00NN § "Behaviour" -->
|
||||
|
||||
- [ ] **No invention beyond the specification.** I confirm this PR does not introduce any field or behaviour not present in OpenAI's spec for the endpoint (B.1) or beyond the scope of the authorizing ADR (B.2). For grandfathered B.2 endpoints, I confirm the change is behaviour-preserving (no contract drift). If something the user actually wants is not in the spec, the right answer is to close this PR and propose an upstream spec change or a new ADR.
|
||||
|
||||
## Type of change
|
||||
|
||||
- [ ] Bug fix (alignment with existing `cli.js` behavior, or with the cited spec / ADR for Class B)
|
||||
- [ ] Feature (new `cli.js` behavior now surfaced through OCP, or new field already in OpenAI's spec for Class B)
|
||||
- [ ] Refactor (no wire-level behavior change)
|
||||
- [ ] Deletion (unalignable feature removal per `ALIGNMENT.md` Unalignable Policy)
|
||||
- [ ] Documentation / governance
|
||||
|
||||
## Reviewer checklist
|
||||
|
||||
Reviewers: this section is for you, not the author. Do not approve until every box is checked.
|
||||
|
||||
- [ ] If Class A, I opened `cli.js` at the cited line range and confirmed the operation matches. If Class B, I opened the OpenAI spec at the cited section (B.1) or the authorizing ADR (B.2) and confirmed the behaviour described in this PR matches the cited reference.
|
||||
- [ ] I ran (or confirmed CI ran) `.github/workflows/alignment.yml` and it passed.
|
||||
- [ ] I am not the commit author of any commit in this PR (Iron Rule 10).
|
||||
- [ ] If the PR asserts scope without a `cli.js` citation (Class A) or without an ADR (Class B), I confirmed the justification is sound per `ALIGNMENT.md` Rule 2 and ADR 0006.
|
||||
- [ ] If the PR is Class B and adds a new endpoint or new method, I confirmed the authorizing ADR lands in the same merge or before this PR.
|
||||
|
||||
## Related
|
||||
|
||||
- `ALIGNMENT.md` Rule(s) invoked: <!-- e.g. Rule 3, or Rule 3 (Class B mapping) -->
|
||||
- Authorizing ADR (Class B only): <!-- e.g. ADR 0006 -->
|
||||
- Related issue / prior PR: <!-- #NNN -->
|
||||
- Historical lesson reference (if relevant): <!-- e.g. 2026-04-11 drift, b87992f -->
|
||||
|
||||
### User-visible change self-check (铁律第五律 5.3)
|
||||
|
||||
- [ ] This PR has user-visible changes → README has corresponding documentation (paste diff link or line range)
|
||||
- [ ] This PR has no user-visible changes → stated "no user-visible change" in summary above
|
||||
|
||||
Reviewers: if "user-visible" is checked but README diff is empty, block merge (per 5.3 reviewer gate).
|
||||
|
||||
### Privacy self-check (for PUBLIC repos) — Iron Rule adjacent
|
||||
|
||||
- [ ] This PR does not introduce real names, nicknames, or handles that identify specific individuals. All references use role-based terms (`project maintainer`, `contributor`, `user`, `reviewer`).
|
||||
- [ ] This PR does not introduce literal personal paths (`/Users/<username>/`, `/home/<username>/`). Uses `$HOME/` or `~/` instead.
|
||||
- [ ] This PR does not introduce personal machine hostnames. Uses role-based names or generic descriptors.
|
||||
- [ ] This PR does not introduce personal email addresses beyond automated placeholders like `noreply@<vendor>.com`.
|
||||
|
||||
Reviewers: if any of the above is violated and the repo is PUBLIC, block merge and request scrub.
|
||||
@@ -0,0 +1,206 @@
|
||||
name: Alignment Guardrail
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'server.mjs'
|
||||
- 'setup.mjs'
|
||||
- 'scripts/**'
|
||||
- 'lib/**'
|
||||
- 'ocp'
|
||||
- 'ocp-connect'
|
||||
- '.github/workflows/alignment.yml'
|
||||
|
||||
jobs:
|
||||
blacklist:
|
||||
name: server.mjs blacklist (hard fail)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Scan server.mjs for hallucinated tokens
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
if [ ! -f server.mjs ]; then
|
||||
echo "server.mjs not found; nothing to scan."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Blacklisted tokens — two kinds (see ALIGNMENT.md "OAuth token-host verification"):
|
||||
# (1) known LLM hallucinations (e.g. the 2026-04-11 /api/oauth/usage drift), and
|
||||
# (2) pinned wrong-host variants of a VERIFIED Class A endpoint (a hit means a
|
||||
# drift to a known-wrong host, not necessarily a hallucination).
|
||||
# Extend only via an ALIGNMENT.md amendment PR. Matched as fixed strings vs server.mjs.
|
||||
BLACKLIST=(
|
||||
"api.anthropic.com/api/oauth/usage"
|
||||
"console.anthropic.com/v1/oauth/token"
|
||||
)
|
||||
|
||||
FAIL=0
|
||||
for token in "${BLACKLIST[@]}"; do
|
||||
if sed "s|//.*\$||" server.mjs | grep -n -F "$token"; then
|
||||
echo "::error file=server.mjs::Blacklisted token '$token' detected in server.mjs."
|
||||
FAIL=1
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$FAIL" -ne 0 ]; then
|
||||
cat <<'EOF'
|
||||
|
||||
============================================================
|
||||
ALIGNMENT GUARDRAIL FAILURE
|
||||
============================================================
|
||||
server.mjs contains a token on the OCP alignment blacklist.
|
||||
|
||||
These tokens are either LLM hallucinations that never appeared in cli.js,
|
||||
or pinned wrong-host variants of a verified Class A endpoint (a drift).
|
||||
See ALIGNMENT.md -> "Historical Lesson: The 2026-04-11 Drift"
|
||||
(commit b87992f) for the full incident record.
|
||||
|
||||
Required action:
|
||||
1. Remove the token from server.mjs.
|
||||
2. grep the reference cli.js for the operation you
|
||||
intended and cite the real line numbers.
|
||||
3. See ALIGNMENT.md Rules 1, 2, and 5.
|
||||
|
||||
Do not add allowlist entries to this workflow without an
|
||||
amendment PR to ALIGNMENT.md (see Amendment Procedure).
|
||||
============================================================
|
||||
EOF
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Blacklist scan clean."
|
||||
|
||||
port-spot:
|
||||
name: port literal SPOT (hard fail)
|
||||
# Background: from 2026-05-08 (PR #71 dogfood accident) through 2026-05-13
|
||||
# a hardcoded "3478" in scripts/upgrade.mjs + scripts/doctor.mjs cascaded
|
||||
# into wrong baseUrl writes for the OpenClaw "claude-local" provider,
|
||||
# taking out the "大内总管" Telegram agent.
|
||||
#
|
||||
# Rule: the only places allowed to write a literal port number in source
|
||||
# are (a) lib/constants.mjs (the SPOT), (b) bash scripts ocp / ocp-connect
|
||||
# (which can't import .mjs and must keep the literal in sync — flagged
|
||||
# with a `// keep in sync with lib/constants.mjs` style comment), and
|
||||
# (c) test-features.mjs (intentionally pins historical ports for plist /
|
||||
# systemd parser tests). Everything else MUST import from lib/constants.mjs.
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Scan for hardcoded port literals outside SPOT
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# Files/paths exempt from the SPOT requirement.
|
||||
EXEMPT_REGEX='^(lib/constants\.mjs|test-features\.mjs|ocp|ocp-connect|CHANGELOG\.md|README\.md|docs/|\.github/workflows/alignment\.yml)'
|
||||
|
||||
# Hardcoded port literals to forbid in non-exempt source.
|
||||
FORBIDDEN_PORTS=("3478" "3456")
|
||||
|
||||
FAIL=0
|
||||
for port in "${FORBIDDEN_PORTS[@]}"; do
|
||||
HITS="$(git ls-files | grep -E '\.(mjs|js|ts|json)$' \
|
||||
| xargs grep -n -E "[^0-9]${port}[^0-9]" 2>/dev/null \
|
||||
| grep -v -E "${EXEMPT_REGEX}" \
|
||||
|| true)"
|
||||
if [ -n "$HITS" ]; then
|
||||
echo "::error::Hardcoded port literal '${port}' found outside lib/constants.mjs:"
|
||||
echo "$HITS"
|
||||
FAIL=1
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$FAIL" -ne 0 ]; then
|
||||
cat <<'EOF'
|
||||
|
||||
============================================================
|
||||
PORT LITERAL SPOT VIOLATION
|
||||
============================================================
|
||||
A hardcoded TCP port literal was found in a source file
|
||||
that should import from lib/constants.mjs instead.
|
||||
|
||||
Background: this rule exists because between 2026-05-08 and
|
||||
2026-05-13 a stray hardcoded "3478" in scripts/upgrade.mjs
|
||||
and scripts/doctor.mjs cascaded into downstream OpenClaw
|
||||
config writes, taking out the OpenClaw Telegram agent.
|
||||
See v3.16.3 CHANGELOG and lib/constants.mjs header comment.
|
||||
|
||||
Required action:
|
||||
1. Import DEFAULT_PORT (or related constant) from
|
||||
lib/constants.mjs instead of hardcoding the literal.
|
||||
2. If the file genuinely cannot import .mjs (e.g. bash
|
||||
script), add it to EXEMPT_REGEX in this workflow and
|
||||
add a `keep in sync with lib/constants.mjs` comment
|
||||
at the reference.
|
||||
3. For test files that intentionally pin historical ports
|
||||
(test-features.mjs), the regex already exempts them.
|
||||
|
||||
============================================================
|
||||
EOF
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Port SPOT scan clean."
|
||||
|
||||
commit-citation:
|
||||
name: commit message citation (soft check)
|
||||
runs-on: ubuntu-latest
|
||||
# Soft check: reports a warning but does not block merge. Reviewers
|
||||
# are expected to enforce per CLAUDE.md. Escalate to hard-fail via
|
||||
# an ALIGNMENT.md amendment if drift recurs.
|
||||
continue-on-error: true
|
||||
steps:
|
||||
- name: Checkout full history
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Scan PR commits for uncited assertions
|
||||
shell: bash
|
||||
env:
|
||||
BASE_SHA: ${{ github.event.pull_request.base.sha }}
|
||||
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
if [ -z "${BASE_SHA:-}" ] || [ -z "${HEAD_SHA:-}" ]; then
|
||||
echo "No PR context; skipping."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Collect commit messages in range.
|
||||
MSGS="$(git log --format=%B "${BASE_SHA}..${HEAD_SHA}")"
|
||||
|
||||
# Look for assertions of the form "Claude Code uses ..." or
|
||||
# "cli.js uses ..." (case-insensitive). For each hit, require
|
||||
# a citation in the same commit message in one of the forms:
|
||||
# cli.js:NNNN (line-number citation)
|
||||
# cli.js vE4 <name> (version + function-name citation)
|
||||
# Absence of a citation is a soft finding.
|
||||
|
||||
WARN=0
|
||||
# Split log into per-commit blocks for precise matching.
|
||||
git log --format="%H" "${BASE_SHA}..${HEAD_SHA}" | while read -r sha; do
|
||||
BODY="$(git log -1 --format=%B "$sha")"
|
||||
if echo "$BODY" | grep -E -i -q '(claude[[:space:]]+code|cli\.js)[[:space:]]+uses'; then
|
||||
if echo "$BODY" | grep -E -q '(cli\.js:[0-9]+|cli\.js[[:space:]]+v[A-Za-z0-9]+[[:space:]]+[A-Za-z_][A-Za-z0-9_]*)'; then
|
||||
echo "OK $sha: assertion cited."
|
||||
else
|
||||
echo "::warning::Commit $sha asserts 'Claude Code uses ...' or 'cli.js uses ...' but does not cite a cli.js line number or versioned function name. See CLAUDE.md -> Commit message conventions."
|
||||
WARN=1
|
||||
fi
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$WARN" -ne 0 ]; then
|
||||
echo "Soft check raised warnings. Reviewer: please enforce per CLAUDE.md."
|
||||
else
|
||||
echo "Commit citation soft check clean."
|
||||
fi
|
||||
@@ -0,0 +1,32 @@
|
||||
name: gitleaks
|
||||
|
||||
# Secret scanning gate. Runs on every PR (any branch) and every push to main.
|
||||
# Configuration is read from the repo-root `.gitleaks.toml` automatically.
|
||||
# Hard-fails the build on any detected leak — public repo, no tolerance.
|
||||
#
|
||||
# To extend the allowlist (e.g. a new known-safe placeholder), edit
|
||||
# `.gitleaks.toml`. Do not add `continue-on-error` here without an explicit
|
||||
# governance decision.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
scan:
|
||||
name: gitleaks scan (hard fail)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout (full history)
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Run gitleaks
|
||||
uses: gitleaks/gitleaks-action@v2
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
@@ -0,0 +1,53 @@
|
||||
name: Auto Release on Tag
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
jobs:
|
||||
release:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Extract version from tag
|
||||
id: ver
|
||||
run: echo "version=${GITHUB_REF#refs/tags/v}" >> $GITHUB_OUTPUT
|
||||
- name: Extract CHANGELOG section
|
||||
id: notes
|
||||
run: |
|
||||
VERSION="${{ steps.ver.outputs.version }}"
|
||||
# Extract section for this version from CHANGELOG.md
|
||||
# Pattern: "## v${VERSION}" through the next "## " or EOF
|
||||
if [ ! -f CHANGELOG.md ]; then
|
||||
echo "No CHANGELOG.md found; using minimal release notes"
|
||||
echo "notes=Release v${VERSION}" >> $GITHUB_OUTPUT
|
||||
exit 0
|
||||
fi
|
||||
awk -v ver="v${VERSION}" '
|
||||
$0 ~ "^## " ver { found=1; print; next }
|
||||
found && /^## v/ { exit }
|
||||
found { print }
|
||||
' CHANGELOG.md > /tmp/release-notes.md
|
||||
if [ ! -s /tmp/release-notes.md ]; then
|
||||
echo "No matching section in CHANGELOG for v${VERSION}; using minimal notes"
|
||||
echo "Release v${VERSION}" > /tmp/release-notes.md
|
||||
fi
|
||||
echo "notes_file=/tmp/release-notes.md" >> $GITHUB_OUTPUT
|
||||
- name: Create GitHub Release
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
if gh release view "v${{ steps.ver.outputs.version }}" >/dev/null 2>&1; then
|
||||
echo "Release v${{ steps.ver.outputs.version }} already exists — skipping"
|
||||
exit 0
|
||||
fi
|
||||
gh release create "v${{ steps.ver.outputs.version }}" \
|
||||
--title "v${{ steps.ver.outputs.version }}" \
|
||||
--notes-file /tmp/release-notes.md \
|
||||
--latest
|
||||
@@ -0,0 +1,41 @@
|
||||
name: Tests
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
test-features:
|
||||
name: test-features.mjs (smoke)
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
# `test-features.mjs` is self-contained — it runs assertions against the
|
||||
# `keys.mjs` DB layer using a throwaway test database. It does NOT need a
|
||||
# live claude CLI or a running OCP server. So this job runs as a hard
|
||||
# check on every push / PR.
|
||||
#
|
||||
# If a future expansion of the suite adds tests that DO require a live
|
||||
# claude CLI or a running OCP server, mark those steps `continue-on-error:
|
||||
# true` (or split them into a separate job) — CI must not be flaky on
|
||||
# things outside the contributor's machine.
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Setup Node
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
# Node 24 ships `node:sqlite` as stable. The test imports keys.mjs,
|
||||
# which uses `import { DatabaseSync } from "node:sqlite"`.
|
||||
# Node 22 also works with `--experimental-sqlite`, but we run on 24
|
||||
# to keep the CI step simple and to match what released OCP runs on.
|
||||
node-version: '24'
|
||||
|
||||
# OCP has zero runtime npm dependencies (package-lock.json shows only
|
||||
# the project's own package and zero external entries). No install
|
||||
# step needed — `node:*` modules are built into Node 24.
|
||||
|
||||
- name: Run test-features.mjs
|
||||
run: npm test
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
# Runtime artifacts
|
||||
logs/
|
||||
*.log
|
||||
|
||||
# Dependencies
|
||||
node_modules/
|
||||
|
||||
# Environment files
|
||||
.env
|
||||
.env.*
|
||||
|
||||
# Editor / OS scratch
|
||||
.DS_Store
|
||||
*.swp
|
||||
*~
|
||||
@@ -0,0 +1,14 @@
|
||||
title = "OCP gitleaks configuration"
|
||||
|
||||
[allowlist]
|
||||
description = "OCP known-safe allowlist"
|
||||
regexes = [
|
||||
# Public OAuth client ID for Claude Code PKCE flow (constant, not a secret)
|
||||
'''9d1c250a-e61b-44d9-88ed-5944d1962f5e''',
|
||||
# OCP README API key placeholder example
|
||||
'''ocp_(example|XXX+|xxxx+)''',
|
||||
]
|
||||
paths = [
|
||||
# Old plan doc with test-admin-123 placeholders
|
||||
'''docs/superpowers/plans/2026-04-10-lan-mode\.md''',
|
||||
]
|
||||
@@ -0,0 +1,40 @@
|
||||
# [CHECKLIST TYPE] Checklist: [FEATURE NAME]
|
||||
|
||||
**Purpose**: [Brief description of what this checklist covers]
|
||||
**Created**: [DATE]
|
||||
**Feature**: [Link to spec.md or relevant documentation]
|
||||
|
||||
**Note**: This checklist is generated by the `/speckit.checklist` command based on feature context and requirements.
|
||||
|
||||
<!--
|
||||
============================================================================
|
||||
IMPORTANT: The checklist items below are SAMPLE ITEMS for illustration only.
|
||||
|
||||
The /speckit.checklist command MUST replace these with actual items based on:
|
||||
- User's specific checklist request
|
||||
- Feature requirements from spec.md
|
||||
- Technical context from plan.md
|
||||
- Implementation details from tasks.md
|
||||
|
||||
DO NOT keep these sample items in the generated checklist file.
|
||||
============================================================================
|
||||
-->
|
||||
|
||||
## [Category 1]
|
||||
|
||||
- [ ] CHK001 First checklist item with clear action
|
||||
- [ ] CHK002 Second checklist item
|
||||
- [ ] CHK003 Third checklist item
|
||||
|
||||
## [Category 2]
|
||||
|
||||
- [ ] CHK004 Another category item
|
||||
- [ ] CHK005 Item with specific criteria
|
||||
- [ ] CHK006 Final item in this category
|
||||
|
||||
## Notes
|
||||
|
||||
- Check items off as completed: `[x]`
|
||||
- Add comments or findings inline
|
||||
- Link to relevant resources or documentation
|
||||
- Items are numbered sequentially for easy reference
|
||||
@@ -0,0 +1,50 @@
|
||||
# [PROJECT_NAME] Constitution
|
||||
<!-- Example: Spec Constitution, TaskFlow Constitution, etc. -->
|
||||
|
||||
## Core Principles
|
||||
|
||||
### [PRINCIPLE_1_NAME]
|
||||
<!-- Example: I. Library-First -->
|
||||
[PRINCIPLE_1_DESCRIPTION]
|
||||
<!-- Example: Every feature starts as a standalone library; Libraries must be self-contained, independently testable, documented; Clear purpose required - no organizational-only libraries -->
|
||||
|
||||
### [PRINCIPLE_2_NAME]
|
||||
<!-- Example: II. CLI Interface -->
|
||||
[PRINCIPLE_2_DESCRIPTION]
|
||||
<!-- Example: Every library exposes functionality via CLI; Text in/out protocol: stdin/args → stdout, errors → stderr; Support JSON + human-readable formats -->
|
||||
|
||||
### [PRINCIPLE_3_NAME]
|
||||
<!-- Example: III. Test-First (NON-NEGOTIABLE) -->
|
||||
[PRINCIPLE_3_DESCRIPTION]
|
||||
<!-- Example: TDD mandatory: Tests written → User approved → Tests fail → Then implement; Red-Green-Refactor cycle strictly enforced -->
|
||||
|
||||
### [PRINCIPLE_4_NAME]
|
||||
<!-- Example: IV. Integration Testing -->
|
||||
[PRINCIPLE_4_DESCRIPTION]
|
||||
<!-- Example: Focus areas requiring integration tests: New library contract tests, Contract changes, Inter-service communication, Shared schemas -->
|
||||
|
||||
### [PRINCIPLE_5_NAME]
|
||||
<!-- Example: V. Observability, VI. Versioning & Breaking Changes, VII. Simplicity -->
|
||||
[PRINCIPLE_5_DESCRIPTION]
|
||||
<!-- Example: Text I/O ensures debuggability; Structured logging required; Or: MAJOR.MINOR.BUILD format; Or: Start simple, YAGNI principles -->
|
||||
|
||||
## [SECTION_2_NAME]
|
||||
<!-- Example: Additional Constraints, Security Requirements, Performance Standards, etc. -->
|
||||
|
||||
[SECTION_2_CONTENT]
|
||||
<!-- Example: Technology stack requirements, compliance standards, deployment policies, etc. -->
|
||||
|
||||
## [SECTION_3_NAME]
|
||||
<!-- Example: Development Workflow, Review Process, Quality Gates, etc. -->
|
||||
|
||||
[SECTION_3_CONTENT]
|
||||
<!-- Example: Code review requirements, testing gates, deployment approval process, etc. -->
|
||||
|
||||
## Governance
|
||||
<!-- Example: Constitution supersedes all other practices; Amendments require documentation, approval, migration plan -->
|
||||
|
||||
[GOVERNANCE_RULES]
|
||||
<!-- Example: All PRs/reviews must verify compliance; Complexity must be justified; Use [GUIDANCE_FILE] for runtime development guidance -->
|
||||
|
||||
**Version**: [CONSTITUTION_VERSION] | **Ratified**: [RATIFICATION_DATE] | **Last Amended**: [LAST_AMENDED_DATE]
|
||||
<!-- Example: Version: 2.1.1 | Ratified: 2025-06-13 | Last Amended: 2025-07-16 -->
|
||||
@@ -0,0 +1,104 @@
|
||||
# Implementation Plan: [FEATURE]
|
||||
|
||||
**Branch**: `[###-feature-name]` | **Date**: [DATE] | **Spec**: [link]
|
||||
**Input**: Feature specification from `/specs/[###-feature-name]/spec.md`
|
||||
|
||||
**Note**: This template is filled in by the `/speckit.plan` command. See `.specify/templates/plan-template.md` for the execution workflow.
|
||||
|
||||
## Summary
|
||||
|
||||
[Extract from feature spec: primary requirement + technical approach from research]
|
||||
|
||||
## Technical Context
|
||||
|
||||
<!--
|
||||
ACTION REQUIRED: Replace the content in this section with the technical details
|
||||
for the project. The structure here is presented in advisory capacity to guide
|
||||
the iteration process.
|
||||
-->
|
||||
|
||||
**Language/Version**: [e.g., Python 3.11, Swift 5.9, Rust 1.75 or NEEDS CLARIFICATION]
|
||||
**Primary Dependencies**: [e.g., FastAPI, UIKit, LLVM or NEEDS CLARIFICATION]
|
||||
**Storage**: [if applicable, e.g., PostgreSQL, CoreData, files or N/A]
|
||||
**Testing**: [e.g., pytest, XCTest, cargo test or NEEDS CLARIFICATION]
|
||||
**Target Platform**: [e.g., Linux server, iOS 15+, WASM or NEEDS CLARIFICATION]
|
||||
**Project Type**: [e.g., library/cli/web-service/mobile-app/compiler/desktop-app or NEEDS CLARIFICATION]
|
||||
**Performance Goals**: [domain-specific, e.g., 1000 req/s, 10k lines/sec, 60 fps or NEEDS CLARIFICATION]
|
||||
**Constraints**: [domain-specific, e.g., <200ms p95, <100MB memory, offline-capable or NEEDS CLARIFICATION]
|
||||
**Scale/Scope**: [domain-specific, e.g., 10k users, 1M LOC, 50 screens or NEEDS CLARIFICATION]
|
||||
|
||||
## Constitution Check
|
||||
|
||||
*GATE: Must pass before Phase 0 research. Re-check after Phase 1 design.*
|
||||
|
||||
[Gates determined based on constitution file]
|
||||
|
||||
## Project Structure
|
||||
|
||||
### Documentation (this feature)
|
||||
|
||||
```text
|
||||
specs/[###-feature]/
|
||||
├── plan.md # This file (/speckit.plan command output)
|
||||
├── research.md # Phase 0 output (/speckit.plan command)
|
||||
├── data-model.md # Phase 1 output (/speckit.plan command)
|
||||
├── quickstart.md # Phase 1 output (/speckit.plan command)
|
||||
├── contracts/ # Phase 1 output (/speckit.plan command)
|
||||
└── tasks.md # Phase 2 output (/speckit.tasks command - NOT created by /speckit.plan)
|
||||
```
|
||||
|
||||
### Source Code (repository root)
|
||||
<!--
|
||||
ACTION REQUIRED: Replace the placeholder tree below with the concrete layout
|
||||
for this feature. Delete unused options and expand the chosen structure with
|
||||
real paths (e.g., apps/admin, packages/something). The delivered plan must
|
||||
not include Option labels.
|
||||
-->
|
||||
|
||||
```text
|
||||
# [REMOVE IF UNUSED] Option 1: Single project (DEFAULT)
|
||||
src/
|
||||
├── models/
|
||||
├── services/
|
||||
├── cli/
|
||||
└── lib/
|
||||
|
||||
tests/
|
||||
├── contract/
|
||||
├── integration/
|
||||
└── unit/
|
||||
|
||||
# [REMOVE IF UNUSED] Option 2: Web application (when "frontend" + "backend" detected)
|
||||
backend/
|
||||
├── src/
|
||||
│ ├── models/
|
||||
│ ├── services/
|
||||
│ └── api/
|
||||
└── tests/
|
||||
|
||||
frontend/
|
||||
├── src/
|
||||
│ ├── components/
|
||||
│ ├── pages/
|
||||
│ └── services/
|
||||
└── tests/
|
||||
|
||||
# [REMOVE IF UNUSED] Option 3: Mobile + API (when "iOS/Android" detected)
|
||||
api/
|
||||
└── [same as backend above]
|
||||
|
||||
ios/ or android/
|
||||
└── [platform-specific structure: feature modules, UI flows, platform tests]
|
||||
```
|
||||
|
||||
**Structure Decision**: [Document the selected structure and reference the real
|
||||
directories captured above]
|
||||
|
||||
## Complexity Tracking
|
||||
|
||||
> **Fill ONLY if Constitution Check has violations that must be justified**
|
||||
|
||||
| Violation | Why Needed | Simpler Alternative Rejected Because |
|
||||
|-----------|------------|-------------------------------------|
|
||||
| [e.g., 4th project] | [current need] | [why 3 projects insufficient] |
|
||||
| [e.g., Repository pattern] | [specific problem] | [why direct DB access insufficient] |
|
||||
@@ -0,0 +1,128 @@
|
||||
# Feature Specification: [FEATURE NAME]
|
||||
|
||||
**Feature Branch**: `[###-feature-name]`
|
||||
**Created**: [DATE]
|
||||
**Status**: Draft
|
||||
**Input**: User description: "$ARGUMENTS"
|
||||
|
||||
## User Scenarios & Testing *(mandatory)*
|
||||
|
||||
<!--
|
||||
IMPORTANT: User stories should be PRIORITIZED as user journeys ordered by importance.
|
||||
Each user story/journey must be INDEPENDENTLY TESTABLE - meaning if you implement just ONE of them,
|
||||
you should still have a viable MVP (Minimum Viable Product) that delivers value.
|
||||
|
||||
Assign priorities (P1, P2, P3, etc.) to each story, where P1 is the most critical.
|
||||
Think of each story as a standalone slice of functionality that can be:
|
||||
- Developed independently
|
||||
- Tested independently
|
||||
- Deployed independently
|
||||
- Demonstrated to users independently
|
||||
-->
|
||||
|
||||
### User Story 1 - [Brief Title] (Priority: P1)
|
||||
|
||||
[Describe this user journey in plain language]
|
||||
|
||||
**Why this priority**: [Explain the value and why it has this priority level]
|
||||
|
||||
**Independent Test**: [Describe how this can be tested independently - e.g., "Can be fully tested by [specific action] and delivers [specific value]"]
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** [initial state], **When** [action], **Then** [expected outcome]
|
||||
2. **Given** [initial state], **When** [action], **Then** [expected outcome]
|
||||
|
||||
---
|
||||
|
||||
### User Story 2 - [Brief Title] (Priority: P2)
|
||||
|
||||
[Describe this user journey in plain language]
|
||||
|
||||
**Why this priority**: [Explain the value and why it has this priority level]
|
||||
|
||||
**Independent Test**: [Describe how this can be tested independently]
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** [initial state], **When** [action], **Then** [expected outcome]
|
||||
|
||||
---
|
||||
|
||||
### User Story 3 - [Brief Title] (Priority: P3)
|
||||
|
||||
[Describe this user journey in plain language]
|
||||
|
||||
**Why this priority**: [Explain the value and why it has this priority level]
|
||||
|
||||
**Independent Test**: [Describe how this can be tested independently]
|
||||
|
||||
**Acceptance Scenarios**:
|
||||
|
||||
1. **Given** [initial state], **When** [action], **Then** [expected outcome]
|
||||
|
||||
---
|
||||
|
||||
[Add more user stories as needed, each with an assigned priority]
|
||||
|
||||
### Edge Cases
|
||||
|
||||
<!--
|
||||
ACTION REQUIRED: The content in this section represents placeholders.
|
||||
Fill them out with the right edge cases.
|
||||
-->
|
||||
|
||||
- What happens when [boundary condition]?
|
||||
- How does system handle [error scenario]?
|
||||
|
||||
## Requirements *(mandatory)*
|
||||
|
||||
<!--
|
||||
ACTION REQUIRED: The content in this section represents placeholders.
|
||||
Fill them out with the right functional requirements.
|
||||
-->
|
||||
|
||||
### Functional Requirements
|
||||
|
||||
- **FR-001**: System MUST [specific capability, e.g., "allow users to create accounts"]
|
||||
- **FR-002**: System MUST [specific capability, e.g., "validate email addresses"]
|
||||
- **FR-003**: Users MUST be able to [key interaction, e.g., "reset their password"]
|
||||
- **FR-004**: System MUST [data requirement, e.g., "persist user preferences"]
|
||||
- **FR-005**: System MUST [behavior, e.g., "log all security events"]
|
||||
|
||||
*Example of marking unclear requirements:*
|
||||
|
||||
- **FR-006**: System MUST authenticate users via [NEEDS CLARIFICATION: auth method not specified - email/password, SSO, OAuth?]
|
||||
- **FR-007**: System MUST retain user data for [NEEDS CLARIFICATION: retention period not specified]
|
||||
|
||||
### Key Entities *(include if feature involves data)*
|
||||
|
||||
- **[Entity 1]**: [What it represents, key attributes without implementation]
|
||||
- **[Entity 2]**: [What it represents, relationships to other entities]
|
||||
|
||||
## Success Criteria *(mandatory)*
|
||||
|
||||
<!--
|
||||
ACTION REQUIRED: Define measurable success criteria.
|
||||
These must be technology-agnostic and measurable.
|
||||
-->
|
||||
|
||||
### Measurable Outcomes
|
||||
|
||||
- **SC-001**: [Measurable metric, e.g., "Users can complete account creation in under 2 minutes"]
|
||||
- **SC-002**: [Measurable metric, e.g., "System handles 1000 concurrent users without degradation"]
|
||||
- **SC-003**: [User satisfaction metric, e.g., "90% of users successfully complete primary task on first attempt"]
|
||||
- **SC-004**: [Business metric, e.g., "Reduce support tickets related to [X] by 50%"]
|
||||
|
||||
## Assumptions
|
||||
|
||||
<!--
|
||||
ACTION REQUIRED: The content in this section represents placeholders.
|
||||
Fill them out with the right assumptions based on reasonable defaults
|
||||
chosen when the feature description did not specify certain details.
|
||||
-->
|
||||
|
||||
- [Assumption about target users, e.g., "Users have stable internet connectivity"]
|
||||
- [Assumption about scope boundaries, e.g., "Mobile support is out of scope for v1"]
|
||||
- [Assumption about data/environment, e.g., "Existing authentication system will be reused"]
|
||||
- [Dependency on existing system/service, e.g., "Requires access to the existing user profile API"]
|
||||
@@ -0,0 +1,251 @@
|
||||
---
|
||||
|
||||
description: "Task list template for feature implementation"
|
||||
---
|
||||
|
||||
# Tasks: [FEATURE NAME]
|
||||
|
||||
**Input**: Design documents from `/specs/[###-feature-name]/`
|
||||
**Prerequisites**: plan.md (required), spec.md (required for user stories), research.md, data-model.md, contracts/
|
||||
|
||||
**Tests**: The examples below include test tasks. Tests are OPTIONAL - only include them if explicitly requested in the feature specification.
|
||||
|
||||
**Organization**: Tasks are grouped by user story to enable independent implementation and testing of each story.
|
||||
|
||||
## Format: `[ID] [P?] [Story] Description`
|
||||
|
||||
- **[P]**: Can run in parallel (different files, no dependencies)
|
||||
- **[Story]**: Which user story this task belongs to (e.g., US1, US2, US3)
|
||||
- Include exact file paths in descriptions
|
||||
|
||||
## Path Conventions
|
||||
|
||||
- **Single project**: `src/`, `tests/` at repository root
|
||||
- **Web app**: `backend/src/`, `frontend/src/`
|
||||
- **Mobile**: `api/src/`, `ios/src/` or `android/src/`
|
||||
- Paths shown below assume single project - adjust based on plan.md structure
|
||||
|
||||
<!--
|
||||
============================================================================
|
||||
IMPORTANT: The tasks below are SAMPLE TASKS for illustration purposes only.
|
||||
|
||||
The /speckit.tasks command MUST replace these with actual tasks based on:
|
||||
- User stories from spec.md (with their priorities P1, P2, P3...)
|
||||
- Feature requirements from plan.md
|
||||
- Entities from data-model.md
|
||||
- Endpoints from contracts/
|
||||
|
||||
Tasks MUST be organized by user story so each story can be:
|
||||
- Implemented independently
|
||||
- Tested independently
|
||||
- Delivered as an MVP increment
|
||||
|
||||
DO NOT keep these sample tasks in the generated tasks.md file.
|
||||
============================================================================
|
||||
-->
|
||||
|
||||
## Phase 1: Setup (Shared Infrastructure)
|
||||
|
||||
**Purpose**: Project initialization and basic structure
|
||||
|
||||
- [ ] T001 Create project structure per implementation plan
|
||||
- [ ] T002 Initialize [language] project with [framework] dependencies
|
||||
- [ ] T003 [P] Configure linting and formatting tools
|
||||
|
||||
---
|
||||
|
||||
## Phase 2: Foundational (Blocking Prerequisites)
|
||||
|
||||
**Purpose**: Core infrastructure that MUST be complete before ANY user story can be implemented
|
||||
|
||||
**⚠️ CRITICAL**: No user story work can begin until this phase is complete
|
||||
|
||||
Examples of foundational tasks (adjust based on your project):
|
||||
|
||||
- [ ] T004 Setup database schema and migrations framework
|
||||
- [ ] T005 [P] Implement authentication/authorization framework
|
||||
- [ ] T006 [P] Setup API routing and middleware structure
|
||||
- [ ] T007 Create base models/entities that all stories depend on
|
||||
- [ ] T008 Configure error handling and logging infrastructure
|
||||
- [ ] T009 Setup environment configuration management
|
||||
|
||||
**Checkpoint**: Foundation ready - user story implementation can now begin in parallel
|
||||
|
||||
---
|
||||
|
||||
## Phase 3: User Story 1 - [Title] (Priority: P1) 🎯 MVP
|
||||
|
||||
**Goal**: [Brief description of what this story delivers]
|
||||
|
||||
**Independent Test**: [How to verify this story works on its own]
|
||||
|
||||
### Tests for User Story 1 (OPTIONAL - only if tests requested) ⚠️
|
||||
|
||||
> **NOTE: Write these tests FIRST, ensure they FAIL before implementation**
|
||||
|
||||
- [ ] T010 [P] [US1] Contract test for [endpoint] in tests/contract/test_[name].py
|
||||
- [ ] T011 [P] [US1] Integration test for [user journey] in tests/integration/test_[name].py
|
||||
|
||||
### Implementation for User Story 1
|
||||
|
||||
- [ ] T012 [P] [US1] Create [Entity1] model in src/models/[entity1].py
|
||||
- [ ] T013 [P] [US1] Create [Entity2] model in src/models/[entity2].py
|
||||
- [ ] T014 [US1] Implement [Service] in src/services/[service].py (depends on T012, T013)
|
||||
- [ ] T015 [US1] Implement [endpoint/feature] in src/[location]/[file].py
|
||||
- [ ] T016 [US1] Add validation and error handling
|
||||
- [ ] T017 [US1] Add logging for user story 1 operations
|
||||
|
||||
**Checkpoint**: At this point, User Story 1 should be fully functional and testable independently
|
||||
|
||||
---
|
||||
|
||||
## Phase 4: User Story 2 - [Title] (Priority: P2)
|
||||
|
||||
**Goal**: [Brief description of what this story delivers]
|
||||
|
||||
**Independent Test**: [How to verify this story works on its own]
|
||||
|
||||
### Tests for User Story 2 (OPTIONAL - only if tests requested) ⚠️
|
||||
|
||||
- [ ] T018 [P] [US2] Contract test for [endpoint] in tests/contract/test_[name].py
|
||||
- [ ] T019 [P] [US2] Integration test for [user journey] in tests/integration/test_[name].py
|
||||
|
||||
### Implementation for User Story 2
|
||||
|
||||
- [ ] T020 [P] [US2] Create [Entity] model in src/models/[entity].py
|
||||
- [ ] T021 [US2] Implement [Service] in src/services/[service].py
|
||||
- [ ] T022 [US2] Implement [endpoint/feature] in src/[location]/[file].py
|
||||
- [ ] T023 [US2] Integrate with User Story 1 components (if needed)
|
||||
|
||||
**Checkpoint**: At this point, User Stories 1 AND 2 should both work independently
|
||||
|
||||
---
|
||||
|
||||
## Phase 5: User Story 3 - [Title] (Priority: P3)
|
||||
|
||||
**Goal**: [Brief description of what this story delivers]
|
||||
|
||||
**Independent Test**: [How to verify this story works on its own]
|
||||
|
||||
### Tests for User Story 3 (OPTIONAL - only if tests requested) ⚠️
|
||||
|
||||
- [ ] T024 [P] [US3] Contract test for [endpoint] in tests/contract/test_[name].py
|
||||
- [ ] T025 [P] [US3] Integration test for [user journey] in tests/integration/test_[name].py
|
||||
|
||||
### Implementation for User Story 3
|
||||
|
||||
- [ ] T026 [P] [US3] Create [Entity] model in src/models/[entity].py
|
||||
- [ ] T027 [US3] Implement [Service] in src/services/[service].py
|
||||
- [ ] T028 [US3] Implement [endpoint/feature] in src/[location]/[file].py
|
||||
|
||||
**Checkpoint**: All user stories should now be independently functional
|
||||
|
||||
---
|
||||
|
||||
[Add more user story phases as needed, following the same pattern]
|
||||
|
||||
---
|
||||
|
||||
## Phase N: Polish & Cross-Cutting Concerns
|
||||
|
||||
**Purpose**: Improvements that affect multiple user stories
|
||||
|
||||
- [ ] TXXX [P] Documentation updates in docs/
|
||||
- [ ] TXXX Code cleanup and refactoring
|
||||
- [ ] TXXX Performance optimization across all stories
|
||||
- [ ] TXXX [P] Additional unit tests (if requested) in tests/unit/
|
||||
- [ ] TXXX Security hardening
|
||||
- [ ] TXXX Run quickstart.md validation
|
||||
|
||||
---
|
||||
|
||||
## Dependencies & Execution Order
|
||||
|
||||
### Phase Dependencies
|
||||
|
||||
- **Setup (Phase 1)**: No dependencies - can start immediately
|
||||
- **Foundational (Phase 2)**: Depends on Setup completion - BLOCKS all user stories
|
||||
- **User Stories (Phase 3+)**: All depend on Foundational phase completion
|
||||
- User stories can then proceed in parallel (if staffed)
|
||||
- Or sequentially in priority order (P1 → P2 → P3)
|
||||
- **Polish (Final Phase)**: Depends on all desired user stories being complete
|
||||
|
||||
### User Story Dependencies
|
||||
|
||||
- **User Story 1 (P1)**: Can start after Foundational (Phase 2) - No dependencies on other stories
|
||||
- **User Story 2 (P2)**: Can start after Foundational (Phase 2) - May integrate with US1 but should be independently testable
|
||||
- **User Story 3 (P3)**: Can start after Foundational (Phase 2) - May integrate with US1/US2 but should be independently testable
|
||||
|
||||
### Within Each User Story
|
||||
|
||||
- Tests (if included) MUST be written and FAIL before implementation
|
||||
- Models before services
|
||||
- Services before endpoints
|
||||
- Core implementation before integration
|
||||
- Story complete before moving to next priority
|
||||
|
||||
### Parallel Opportunities
|
||||
|
||||
- All Setup tasks marked [P] can run in parallel
|
||||
- All Foundational tasks marked [P] can run in parallel (within Phase 2)
|
||||
- Once Foundational phase completes, all user stories can start in parallel (if team capacity allows)
|
||||
- All tests for a user story marked [P] can run in parallel
|
||||
- Models within a story marked [P] can run in parallel
|
||||
- Different user stories can be worked on in parallel by different team members
|
||||
|
||||
---
|
||||
|
||||
## Parallel Example: User Story 1
|
||||
|
||||
```bash
|
||||
# Launch all tests for User Story 1 together (if tests requested):
|
||||
Task: "Contract test for [endpoint] in tests/contract/test_[name].py"
|
||||
Task: "Integration test for [user journey] in tests/integration/test_[name].py"
|
||||
|
||||
# Launch all models for User Story 1 together:
|
||||
Task: "Create [Entity1] model in src/models/[entity1].py"
|
||||
Task: "Create [Entity2] model in src/models/[entity2].py"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Implementation Strategy
|
||||
|
||||
### MVP First (User Story 1 Only)
|
||||
|
||||
1. Complete Phase 1: Setup
|
||||
2. Complete Phase 2: Foundational (CRITICAL - blocks all stories)
|
||||
3. Complete Phase 3: User Story 1
|
||||
4. **STOP and VALIDATE**: Test User Story 1 independently
|
||||
5. Deploy/demo if ready
|
||||
|
||||
### Incremental Delivery
|
||||
|
||||
1. Complete Setup + Foundational → Foundation ready
|
||||
2. Add User Story 1 → Test independently → Deploy/Demo (MVP!)
|
||||
3. Add User Story 2 → Test independently → Deploy/Demo
|
||||
4. Add User Story 3 → Test independently → Deploy/Demo
|
||||
5. Each story adds value without breaking previous stories
|
||||
|
||||
### Parallel Team Strategy
|
||||
|
||||
With multiple developers:
|
||||
|
||||
1. Team completes Setup + Foundational together
|
||||
2. Once Foundational is done:
|
||||
- Developer A: User Story 1
|
||||
- Developer B: User Story 2
|
||||
- Developer C: User Story 3
|
||||
3. Stories complete and integrate independently
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- [P] tasks = different files, no dependencies
|
||||
- [Story] label maps task to specific user story for traceability
|
||||
- Each user story should be independently completable and testable
|
||||
- Verify tests fail before implementing
|
||||
- Commit after each task or logical group
|
||||
- Stop at any checkpoint to validate story independently
|
||||
- Avoid: vague tasks, same file conflicts, cross-story dependencies that break independence
|
||||
@@ -0,0 +1,74 @@
|
||||
Inherits: @~/.cc-rules/AGENTS.md
|
||||
|
||||
# OCP — Open Claude Proxy — Agent Guidelines
|
||||
|
||||
**Scope**: the `dtzp555-max/ocp` repository.
|
||||
**Audience**: any AI coding agent (Claude Code / Cursor / OpenCode / Copilot / Codex / Gemini) touching OCP source.
|
||||
|
||||
---
|
||||
|
||||
## What this project is
|
||||
|
||||
OCP (Open Claude Proxy) is an open-source HTTP gateway that sits between the Claude Code CLI (`cli.js`) and Anthropic's public API. It forwards, observes, and multiplexes traffic that `cli.js` already emits — it is explicitly **not** an extension layer. A secondary role: registering OCP as a local provider inside OpenClaw (a sibling IDE-agnostic tool), so that users running OpenClaw against OCP see the same model list as native Claude Code.
|
||||
|
||||
Runtime: Node.js (ESM, `.mjs` throughout). No build step. No bundler. `server.mjs` is the single executable entrypoint; `ocp` and `ocp-connect` are CLI wrappers.
|
||||
|
||||
---
|
||||
|
||||
## Stack
|
||||
|
||||
- Node.js >=18, native ESM modules
|
||||
- `http`/`https` built-ins for the proxy core (no Express, no Fastify)
|
||||
- `models.json` as the single source of truth for model metadata
|
||||
- GitHub Actions for CI (`alignment.yml`, `release.yml`)
|
||||
- `gh` CLI assumed for PR creation and release automation
|
||||
- No TypeScript. No test framework beyond `test-features.mjs` (run via `npm test`; CI workflow `.github/workflows/test.yml`). Keep dependencies minimal.
|
||||
|
||||
---
|
||||
|
||||
## Key files to know
|
||||
|
||||
- `server.mjs` — the proxy itself; every request path lives here. Governed by `ALIGNMENT.md`.
|
||||
- `models.json` — single source of truth for model IDs, aliases, and context windows. See ADR 0003.
|
||||
- `setup.mjs` — first-time installer; reads `models.json` to derive bootstrap config.
|
||||
- `scripts/sync-openclaw.mjs` — idempotent OpenClaw registry sync invoked by `ocp update`. See ADR 0004.
|
||||
- `ocp` — user-facing CLI (install, update, start, stop, status, logs, etc.).
|
||||
- `ALIGNMENT.md` — the constitution. Binding for any `server.mjs` change. See ADR 0002.
|
||||
- `.github/workflows/alignment.yml` — CI blacklist grep; fails the build on known-hallucinated tokens.
|
||||
- `CLAUDE.md` — Claude-Code-specific session instructions + release_kit overlay (Iron Rule 5.5).
|
||||
- `docs/adr/` — Architecture Decision Records. Read these before proposing governance or SPOT changes. See `docs/adr/README.md` for the index.
|
||||
- `docs/superpowers/plans/` — active spec-kit plans. `docs/superpowers/plans/shipped/` archives plans that have been delivered (don't propose changes against shipped plans — they're history). `docs/superpowers/specs/` holds long-lived design documents that other code references (e.g., the SSE heartbeat design referenced from `server.mjs`).
|
||||
- `memory/constitution.md` — spec-kit's project constitution (its standard `memory/` location). Distinct from `~/.cc-rules/memory/` (cross-machine personal memory) and from this repo's `ALIGNMENT.md` (the OCP code-level constitution).
|
||||
|
||||
---
|
||||
|
||||
## Project-specific constraints
|
||||
|
||||
- **`ALIGNMENT.md` is binding.** Any PR touching `server.mjs` must cite `cli.js:NNNN` (or `cli.js vE4 <functionName>`) in the commit body and PR description. See `CLAUDE.md` § "Hard requirements for `server.mjs` changes" and ADR 0002.
|
||||
- **Alignment CI is not suppressible.** The `alignment.yml` workflow greps `server.mjs` for known-hallucinated tokens (currently blocking `api.anthropic.com/api/oauth/usage`). Adding new tokens is done via PR amendment to `alignment.yml`; removing entries requires an `ALIGNMENT.md` amendment PR.
|
||||
- **No self-approval.** Implementation author cannot merge their own PR (Iron Rule 10). A fresh-context reviewer must open `cli.js` at the cited lines and confirm in the review comment.
|
||||
- **`models.json` is the only place to add/edit models.** Do not touch `MODEL_MAP` or `MODELS` arrays directly in `server.mjs` or `setup.mjs`. See ADR 0003.
|
||||
- **OpenClaw boundary.** `scripts/sync-openclaw.mjs` only writes `models.providers["claude-local"].models` and `agents.defaults.models["claude-local/*"]` in `~/.openclaw/openclaw.json`. Do not expand scope. See ADR 0004.
|
||||
|
||||
---
|
||||
|
||||
## Release protocol
|
||||
|
||||
OCP follows the machine-readable `release_kit:` overlay in `CLAUDE.md` (Iron Rule 5.5). Before any version bump or tag push, re-read that YAML block and walk every item in `new_feature_doc_expectations` and `bootstrap_quirk_policy`. Tag push triggers `.github/workflows/release.yml`, which creates the GitHub Release automatically — do not create the release manually.
|
||||
|
||||
Version is sourced from `package.json`; changelog from `CHANGELOG.md`; user-facing docs from `README.md`.
|
||||
|
||||
---
|
||||
|
||||
## Handoff expectations
|
||||
|
||||
A fresh session picking up OCP work should read, in order:
|
||||
|
||||
1. This file (`AGENTS.md`).
|
||||
2. `ALIGNMENT.md` — constitution; non-optional.
|
||||
3. `CLAUDE.md` — tool-specific instructions and release_kit overlay.
|
||||
4. `docs/adr/` — most recent ADRs first; they explain why the current structure exists.
|
||||
5. Any active plan under `docs/superpowers/plans/` (excluding `shipped/` which is the archive).
|
||||
6. `~/.cc-rules/memory/auto/MEMORY.md` — cross-machine memory index.
|
||||
|
||||
Only after these should the session touch code.
|
||||
+173
@@ -0,0 +1,173 @@
|
||||
# OCP Alignment Constitution
|
||||
|
||||
**Status:** Active. This document is the supreme source of truth for OCP scope decisions. Conflicts with other documents (README, issues, prior commit messages) resolve in favor of this file.
|
||||
|
||||
---
|
||||
|
||||
## Core Principle
|
||||
|
||||
OCP (Open Claude Proxy) is a **proxy layer** for the Claude Code CLI. It forwards, observes, and multiplexes the traffic that `cli.js` already emits. It is **not** an extension layer. If `cli.js` does not perform a given operation, or performs it differently, OCP does not invent one.
|
||||
|
||||
This Core Principle applies in full to **Class A** endpoints (the `cli.js`-mirror surface). A second class of endpoint — **Class B**, the OCP-owned compatibility surface — has its own scope discipline anchored to its own specification authority. See "Scope Clarification: OCP-Owned Compatibility Endpoints (Class B)" below and ADR 0006.
|
||||
|
||||
---
|
||||
|
||||
## Rules
|
||||
|
||||
The following Rules apply to **Class A operations** (the `cli.js`-mirror surface — the inbound `/v1/messages` forwarding route, the outbound `/v1/messages` wire call used by `handleUsage()` for rate-limit-header extraction, the OAuth bearer machinery, and any future operations OCP forwards from `cli.js` to Anthropic). For the Class B mapping of each rule, see the Class B section below.
|
||||
|
||||
1. **Rule 1 (Grep First).** Before adding, renaming, or changing any endpoint, header, parameter, or response shape, the author must `grep` the reference `cli.js` and record the exact line numbers in the commit message and PR body. An absent grep hit is itself a finding and must be declared.
|
||||
|
||||
2. **Rule 2 (No Invention).** OCP must not introduce endpoints, headers, request fields, or response fields that are not present in `cli.js`. Speculative "Claude Code probably uses X" statements are prohibited. If the behavior is not observable in `cli.js`, the feature is out of scope.
|
||||
|
||||
3. **Rule 3 (Match the Implementation).** When `cli.js` does perform a given operation, OCP must match it byte-for-byte on the wire: same path, same method, same headers (including casing and ordering constraints), same body schema, same auth scheme. Deviations require an explicit, reviewed exception recorded in this file.
|
||||
|
||||
4. **Rule 4 (Unalignable Features Are Deleted).** Any existing OCP feature that cannot be traced to a concrete `cli.js` reference is deleted. There is no "grandfathering" and no "keep it disabled." The policy is removal, not deprecation. See the Unalignable Policy section below.
|
||||
|
||||
5. **Rule 5 (Cite Line Numbers in Commits).** Every commit that touches `server.mjs` must reference `cli.js` by line number or function name in the form `cli.js:NNNN` or `cli.js vE4 <functionName>`. Commits asserting "Claude Code uses X" without such a citation are blocked by CI and must be reverted on detection.
|
||||
|
||||
---
|
||||
|
||||
## Golden Reference: `cli.js`
|
||||
|
||||
`cli.js` is the Claude Code CLI JavaScript bundle shipped inside the `@anthropic-ai/claude-code` npm package. It is the single source of truth for "what Claude Code actually does."
|
||||
|
||||
### Canonical paths per machine
|
||||
|
||||
| Machine / environment | Path |
|
||||
| --- | --- |
|
||||
| macOS (npm global) | `/usr/local/lib/node_modules/@anthropic-ai/claude-code/cli.js` |
|
||||
| macOS (nvm) | `~/.nvm/versions/node/*/lib/node_modules/@anthropic-ai/claude-code/cli.js` |
|
||||
| Linux (npm global) | `/usr/lib/node_modules/@anthropic-ai/claude-code/cli.js` |
|
||||
| Linux (OCI opc user) | `~/.npm-global/lib/node_modules/@anthropic-ai/claude-code/cli.js` |
|
||||
| Windows (npm global) | `%APPDATA%\npm\node_modules\@anthropic-ai\claude-code\cli.js` |
|
||||
| Raspberry Pi (nvm) | `~/.nvm/versions/node/*/lib/node_modules/@anthropic-ai/claude-code/cli.js` |
|
||||
|
||||
### Current audit pin
|
||||
|
||||
- **Claude Code version under audit:** `2.1.89`
|
||||
- **`cli.js` SHA-256:** `a9950ef6407fdc750bddb673852485500387e524a99d42385cb81e7d17128e01`
|
||||
- **Audit date:** `2026-04-20`
|
||||
- **Auditor:** `project maintainer`
|
||||
|
||||
The audit pin is updated once per year (see Annual Alignment Audit) and whenever a drift incident forces a re-verification.
|
||||
|
||||
### OAuth token-host verification (2026-05-31)
|
||||
|
||||
Motivating evidence: the 2026-05-31 code audit (issues #112 / #119 / #123). The OAuth bearer
|
||||
machinery is a Class A surface (Rules 1–5). Because `cli.js` now ships as a
|
||||
compiled binary, the token-refresh host was re-verified against `claude.exe` (Claude Code
|
||||
`2.1.154`) on 2026-05-31 using the compiled-binary protocol — `strings` on the Mach-O, **no
|
||||
live OAuth probe** (a `refresh_token` grant would rotate the operator's real credentials):
|
||||
|
||||
- **Verified host:** `https://platform.claude.com/v1/oauth/token` — present in the binary
|
||||
byte-for-byte, paired with `OAUTH_CLIENT_ID` in the same `prod` config object (matches
|
||||
`server.mjs` `OAUTH_TOKEN_URL` / `OAUTH_CLIENT_ID`). The legacy `console.anthropic.com/v1/oauth`
|
||||
host is absent (0 hits).
|
||||
- **Pinned wrong-host variant:** `console.anthropic.com/v1/oauth/token` is added to the
|
||||
`alignment.yml` blacklist so a future accidental revert to the legacy host hard-fails CI.
|
||||
|
||||
The blacklist therefore now holds two kinds of token: (1) known hallucinations (e.g.
|
||||
`api.anthropic.com/api/oauth/usage`, the 2026-04-11 drift), and (2) pinned wrong-host variants
|
||||
of a *verified* Class A endpoint. A blacklist hit means either a re-introduced hallucination
|
||||
**or** a drift to a known-wrong host — both are alignment failures under Rules 2 and 3.
|
||||
|
||||
---
|
||||
|
||||
## Historical Lesson: The 2026-04-11 Drift
|
||||
|
||||
On 2026-04-11, commit `b87992f` ("fix: use dedicated /api/oauth/usage endpoint for reliable plan data") was merged. The commit message asserted that `/api/oauth/usage` was "the dedicated usage endpoint that Claude Code CLI uses."
|
||||
|
||||
**This assertion was false.** The string `/api/oauth/usage` does not appear in `cli.js` at any version shipped up to that date. The endpoint was fabricated by an LLM-assisted authoring pass that generalized from adjacent OAuth paths without verifying against `cli.js`. A follow-up commit `cb6c2a8` ("fallback to stale cache on usage API 429 + extend cache to 15min") compounded the error by caching the fabricated response to hide the 4xx failures.
|
||||
|
||||
**Impact:** The `/usage` progress bar in the dashboard was broken for nine days (2026-04-11 through 2026-04-20) before the drift was isolated.
|
||||
|
||||
**Root cause:** LLM hallucination accepted without `grep cli.js` verification, compounded by the absence of a CI blacklist and the absence of this constitution.
|
||||
|
||||
**Fix commit:** `fd7973a` (PR #21 — restored header-based `/usage`); follow-up `01e260c` (PR #24 — OAuth Bearer header correction)
|
||||
|
||||
**Lesson codified:** Rules 1, 2, and 5 of this document; the CI blacklist in `.github/workflows/alignment.yml`; and the PR template evidence section exist to make the 2026-04-11 drift structurally impossible to repeat.
|
||||
|
||||
---
|
||||
|
||||
## Unalignable Policy
|
||||
|
||||
A feature is **unalignable** if, after a good-faith search, it cannot be mapped to a specific `cli.js` line range or function (Class A) or to a specific OpenAI specification section AND an authorizing ADR (Class B).
|
||||
|
||||
- Unalignable features are **deleted**, not disabled, not feature-flagged, not deprecated.
|
||||
- Deletion is the default outcome of an alignment audit finding. The burden of proof is on the feature, not on the auditor.
|
||||
- A deletion PR does not require user-facing deprecation notice, because the feature was never legitimately in scope.
|
||||
- If a user workflow depended on an unalignable feature, the correct remediation is to upstream the behavior into `cli.js` (Class A) or into OpenAI's spec (Class B) or to move it out of OCP into a separate tool. OCP does not retain it.
|
||||
|
||||
---
|
||||
|
||||
## Scope Clarification: OCP-Owned Compatibility Endpoints (Class B)
|
||||
|
||||
OCP has two classes of endpoint. Rules 1–5 above were drafted in the aftermath of the 2026-04-11 forwarding drift and are written in the language of a one-to-one proxy; they apply verbatim to **Class A** endpoints. **Class B** endpoints — the OCP-owned compatibility surface where `cli.js` is not the wire authority — have their own scope discipline, anchored to their own specification authority. The full rationale lives in **ADR 0006 (OpenAI Shim Scope)**.
|
||||
|
||||
**Class A** — `cli.js`-mirror endpoints. The endpoint exists because `cli.js` performs the equivalent operation and OCP forwards, observes, or multiplexes that operation. Rules 1–5 above apply verbatim. Citation format: `cli.js:NNNN` or `cli.js vE4 <functionName>`.
|
||||
|
||||
**Class B** — OCP-owned compatibility endpoints. The endpoint exists because OCP itself surfaces it, with no `cli.js` analogue. Two sub-buckets: **B.1** (OpenAI-compatibility surface — protocol authority is OpenAI's `/v1/chat/completions` specification) and **B.2** (OCP-administrative surface — authority is the ADR that authorized the endpoint's existence).
|
||||
|
||||
### Grandfather provision for existing B.2 inventory
|
||||
|
||||
ADR 0006 retroactively authorizes the B.2 endpoints listed in the inventory table below, **frozen at their current behaviour as of v3.16.4**. This is a one-time provision; it does not extend to new B.2 endpoints or to B.1 endpoints. Any change to the contract (request shape, response shape, semantics) of a grandfathered B.2 endpoint is treated as a new authorization request and requires either a behaviour-preserving refactor PR or its own ADR. Any new B.2 endpoint, or any new method on a grandfathered B.2 endpoint, requires its own ADR before merge.
|
||||
|
||||
### Current Class B inventory
|
||||
|
||||
| Endpoint | Method | Sub-bucket | Authorizing ADR |
|
||||
|---|---|---|---|
|
||||
| `/v1/chat/completions` | POST | B.1 (OpenAI-compat) | ADR 0006 |
|
||||
| `/v1/models` | GET | B.1 (OpenAI-compat) | ADR 0006; content sourced from `models.json` per ADR 0003 |
|
||||
| `/health` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/dashboard` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/sessions` | GET, DELETE | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/logs` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/status` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/settings` | GET, PATCH | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/keys` | GET, POST | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/keys/:id` | DELETE | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/keys/:id/quota` | GET, PATCH | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/usage` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/cache/stats` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/cache` | DELETE | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
|
||||
**Hybrid note.** `/usage` is a hybrid endpoint: the underlying call to `api.anthropic.com/v1/messages` (used to extract `anthropic-ratelimit-unified-*` headers, per the in-file comment block at `server.mjs` line 845–849) is Class A and requires the standard `cli.js` citation; the local synthesis layer that adds `proxy:` stats and `models:` snapshot is Class B and is authorized by ADR 0006. A PR touching only the wire-call layer is Class A; a PR touching only the synthesis layer is Class B; a PR touching both must satisfy both citation requirements.
|
||||
|
||||
### Class B citation requirement
|
||||
|
||||
Class B PRs cite **the relevant specification section + the authorizing ADR**, in place of `cli.js:NNNN`. Examples:
|
||||
|
||||
- B.1: "OpenAI `chat/completions` API, `response_format` parameter (https://platform.openai.com/docs/api-reference/chat/create), authorized by ADR 0006."
|
||||
- B.2 (grandfathered): "Authorized by ADR 0006 (grandfathered as of v3.16.4)."
|
||||
- B.2 (with its own ADR): "Authorized by ADR 00NN (the ADR that originally authorized the endpoint)."
|
||||
|
||||
### Rule mapping for Class B
|
||||
|
||||
| Class A rule | Class B mapping |
|
||||
|---|---|
|
||||
| Rule 1 (Grep First) | Read the cited OpenAI spec section (B.1) or the authorizing ADR (B.2) before writing code. Record the spec URL and ADR number in the PR body. |
|
||||
| Rule 2 (No Invention) | OCP must not introduce fields or behaviour not present in OpenAI's spec for the endpoint (B.1) or outside the scope of the authorizing ADR (B.2). For grandfathered B.2 endpoints, "scope" is the v3.16.4 behaviour snapshot. |
|
||||
| Rule 3 (Match the Implementation) | Match OpenAI's spec wire-format (B.1) or the ADR's specified behaviour (B.2). |
|
||||
| Rule 4 (Unalignable Features Are Deleted) | A Class B endpoint that maps to nothing in OpenAI's spec **and** lacks an authorizing ADR (including not being in the grandfather inventory) is unalignable and is deleted on the same terms as a Class A unalignable feature. |
|
||||
| Rule 5 (Cite Line Numbers in Commits) | Cite the OpenAI spec section URL + authorizing ADR number in the commit body (B.1) or the authorizing ADR number alone (B.2). |
|
||||
|
||||
### New Class B endpoint procedure
|
||||
|
||||
Any new Class B endpoint, or any new method on an existing Class B endpoint (including grandfathered ones), requires its own ADR before merge. An "ADR-less" new Class B endpoint is itself an alignment finding under Rule 4.
|
||||
|
||||
---
|
||||
|
||||
## Annual Alignment Audit
|
||||
|
||||
- **Date:** 11 April each year (the anniversary of the `b87992f` drift).
|
||||
- **Scope (Class A):** Diff the current `cli.js` against the pinned SHA-256 in the Golden Reference section. For every network call in `server.mjs`, re-verify that the corresponding `cli.js` reference still exists at the cited line numbers (adjust citations if line numbers shifted across Claude Code versions).
|
||||
- **Scope (Class B):** Audit B.1 endpoints against OpenAI's current `/v1/chat/completions` specification snapshot. Audit B.2 endpoints against their authorizing ADR — for grandfathered endpoints, verify the endpoint behaviour still matches its v3.16.4 snapshot; for ADR-specific endpoints, verify behaviour still matches the ADR. The B.1 specification pin lives in `docs/openai-compat-pin.md` (created alongside the first B.1 audit; not required for ADR 0006 to land).
|
||||
- **Output:** A signed audit note committed to `docs/alignment-audits/YYYY-04-11.md`, updating the Class A pin and (once `docs/openai-compat-pin.md` exists) the B.1 pin.
|
||||
- **Failure mode:** Any audit finding that cannot be reconciled triggers an immediate deletion PR per the Unalignable Policy.
|
||||
|
||||
---
|
||||
|
||||
## Amendment Procedure
|
||||
|
||||
This constitution is amended only by a PR that (a) cites the evidence motivating the amendment, (b) is reviewed by an independent reviewer per CC Iron Rule 10, and (c) updates the Historical Lesson section if the amendment was driven by an incident. Amendments never retroactively legitimize previously unalignable features.
|
||||
+479
@@ -0,0 +1,479 @@
|
||||
# Changelog
|
||||
|
||||
## v3.24.0 — 2026-07-21
|
||||
|
||||
Minor release. Headline: two long-requested **OpenAI-compat features** land — **multimodal vision** (`image_url` parts) and **structured outputs** (`response_format` / JSON schema). Also: the prompt-char budget now derives from the model SPOT instead of a hand-set constant, an agentic-turn bug that dropped the model's final answer is fixed, and `OCP_LOCAL_TOOLS` supports the OpenClaw-backend use case. Four of the six landed from external contributors (@vvlasy-openclaw). Every code PR carried a fresh-context reviewer (Iron Rule 10); no new endpoint, no new `cli.js` wire behavior.
|
||||
|
||||
### Added
|
||||
|
||||
- **Multimodal vision — OpenAI `image_url` parts (#154, contributed by @vvlasy-openclaw).** `/v1/chat/completions` forwards OpenAI `image_url` content parts to `claude` as native Anthropic image blocks via `--input-format stream-json` (the CLI's own contract — no invented wire shape, verified live). Base64 `data:` URIs by default; remote `http(s)` URLs are off unless `CLAUDE_IMAGE_ALLOW_URL=1` (and even then OCP never fetches them — no SSRF surface). Byte/count caps (`CLAUDE_MAX_IMAGE_BYTES`, `CLAUDE_MAX_IMAGES`, `CLAUDE_MAX_IMAGE_TOTAL_BYTES`), all fail-closed on a misconfigured value. TUI mode returns `400 images_unsupported_in_tui_mode` (it can't carry image blocks) and an image present only in a `system` message returns `400` rather than being silently dropped. README § "Images / Multimodal".
|
||||
- **Structured outputs — OpenAI `response_format` (#153, contributed by @vvlasy-openclaw).** `/v1/chat/completions` honors `response_format: { type: "json_schema" | "json_object" }` so OpenAI-SDK clients (Home Assistant AI Tasks, Honcho, scripts) get machine-parseable JSON in `content`. Validates against the schema (incl. `$ref`/`$defs` + `allOf`/`anyOf`/`oneOf` — the shapes the OpenAI SDK emits), retries with a stronger instruction up to `OCP_STRUCTURED_MAX_ATTEMPTS` (default 3, fail-closed), and on exhaustion returns OpenAI's own `refusal` field (200/`content:null`) rather than an invented error. Cyclic-`$ref` schemas fail closed (no stack overflow); a pathologically deep model reply returns a refusal, not a 500 (#181). Single-flight dedup + structured-keyed cache bound the cost. Class B.1 (ADR 0006). README § "Structured Outputs".
|
||||
- **SPOT-derived prompt-char budget (#179, ADR 0009).** `MAX_PROMPT_CHARS` default now derives from `max(models.json contextWindow) × 3 chars/token` (600,000 today) instead of the hand-set 150,000 (~37.5k tokens) that silently under-delivered the advertised window ~5×. `CLAUDE_MAX_PROMPT_CHARS` and the settings API remain absolute overrides; a garbage value fails closed to the derived default.
|
||||
- **`OCP_LOCAL_TOOLS` — positive local-tools system-prompt wrapper (single-user, loopback only; default off) (#182, contributed by @vvlasy-openclaw).** The `-p` path prepends a wrapper telling the model it has no local filesystem/shell access — correct for a shared gateway, but it makes a personal instance's model (e.g. an OpenClaw agent on its own local OCP) refuse to use the server-side `claude` tools it legitimately has. `=1` swaps in a positive wrapper. Changes **only the prompt**, never the tool surface (`--allowedTools`/`--disallowedTools` untouched; multi-tenant still disallows the FS surface); it does **not** enable client-side `tool_calls` (still unsupported by design). Fail-closed boot gate mirroring `OCP_TUI_FULL_TOOLS` (ADR 0007): refuses to start under `CLAUDE_AUTH_MODE=multi`, a non-loopback bind, or `PROXY_ANONYMOUS_KEY`. Inert (and logged as such) in TUI mode. The active wrapper is folded into the config epoch so toggling it invalidates the standard response cache. No new `cli.js` wire behavior (reuses the already-cited `--system-prompt` flag).
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Agentic turns dropped the model's final answer (#183, contributed by @vvlasy-openclaw).** On a tool-using turn, `/v1/chat/completions` returned only the opening preamble ("I'll find the repo…") and silently discarded the post-tool-use final answer: aggregate-`assistant` extraction was gated on `isFirstDelta` (which flips false after the first text), and OCP runs pure-aggregate mode (no `--include-partial-messages`), so each of an agentic turn's several assistant messages after the first was lost. Now guards on `sawTextDelta` and accumulates every assistant message (streaming and buffered paths assemble byte-identically).
|
||||
- **Deep structured reply returned a 500 instead of a refusal (#181 / #184).** `validateJsonSchema` recurses on the model reply's nesting depth; a ~2000-level-deep reply overflowed the stack → caught `RangeError` → generic 500. A crash-safe façade converts that (only) into a validation miss → refusal; any other throw still surfaces.
|
||||
|
||||
## v3.23.0 — 2026-07-17
|
||||
|
||||
Minor release. Headline: **the default `sonnet` alias now resolves to Claude Sonnet 5** — a behavior change for every request that omits `model` (pin `claude-sonnet-4-6` by full ID to keep the previous default). Also: Windows-safe upgrade snapshots, two upgrade-system reliability fixes from a live fleet update, the `CLAUDE_SYSTEM_PROMPT` env var made functional, cache-key honesty for config changes, a billing-policy status correction (the 2026-06-15 `-p` split is PAUSED by Anthropic), and a major README restructure. No new endpoint; no new `cli.js` wire behavior. Every code PR carried a fresh-context reviewer (Iron Rule 10).
|
||||
|
||||
### Changed
|
||||
|
||||
- **Default `sonnet` alias → `claude-sonnet-5` (#168, contributed by @vvlasy-openclaw).** The `sonnet` alias (the model used for every `/v1/chat/completions` request that omits `model`, and OpenClaw's OCP primary via `ocp-connect`) now resolves to `claude-sonnet-5` instead of `claude-sonnet-4-6`. `claude-sonnet-4-6` remains available by full ID for pinning. Mirrors the shipped Claude CLI's own `latest_per_family` mapping (`sonnet → claude-sonnet-5`, verified from binary 2.1.211). Split out from the additive model entry (#152) per Iron Rule 11.
|
||||
- **`CLAUDE_SYSTEM_PROMPT` is now functional (#175).** The var was read, documented, and echoed on `/health.systemPrompt` but never reached a request (dead since the `APPEND_SYSTEM_PROMPT` retirement). It is now appended (last, trimmed) to the composed system prompt on the default `-p` path via the new pure `lib/prompt.mjs`; TUI-mode panes are unaffected. Unset ⇒ byte-identical composition to before. README § Environment Variables documents it, including the cache caveat below.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Windows-safe upgrade snapshot paths (#167, contributed by @nyxst4ck).** Snapshot directory timestamps now use `-` instead of `:` (Windows forbids `:` in names); legacy colon-named snapshots keep parsing, and `listSnapshots` now orders by **parsed timestamp** (with a deterministic name tie-breaker) so mixed legacy/new names sort chronologically — the initial revision's raw-string sort could delete the newest recovery snapshot at the format boundary and was caught in review; regression tests pin the same-hour mixed-format case.
|
||||
- **`ocp update` reliability — two live-incident fixes (#174, closes #173).** (1) The doctor now runs `git fetch --tags` (offline-tolerant) before computing `latest_version` — previously it compared against the locally cached `origin/main`, so machines that hadn't pulled since a release reported "Already at latest" forever. (2) Post-flight now asserts `/health.version` equals the upgrade target (new `postFlightOk` predicate) instead of accepting any `auth.ok` — a stale orphan process holding the port used to pass post-flight while still serving the old version; the failure message now reports the last-seen version and points at `ss -ltnp`/`lsof -i`.
|
||||
- **Response-cache key now carries a boot-config epoch (#177, closes #176).** The persistent cache keyed on model+key+params+messages but not on server config that shapes answers (`CLAUDE_SYSTEM_PROMPT`, wrapper text, `CLAUDE_ALLOWED_TOOLS`, `CLAUDE_NO_CONTEXT`) — changing any of these and restarting could serve stale-config answers until TTL expiry. A sha256 config-epoch is folded into every key; any config change is an instant whole-cache invalidation. One-time side effect: existing cache entries miss once after this upgrade.
|
||||
|
||||
### Docs
|
||||
|
||||
- **Billing-policy status corrected (#171).** Anthropic **paused** the announced 2026-06-15 `claude -p` billing split on its effective date (official help-article citation in README § How It Works): the default `-p` path currently bills the subscription, and TUI-mode is reframed as the ready-made **hedge** for if/when a reworked change lands. All in-force assertions of the split are now date-stamped and conditioned.
|
||||
- **LAN mode scoped to chat-class workloads (#171).** New "workload fit" paragraph: multi-device OCP is for text-in/text-out workloads; client-machine coding agents are architecturally out of scope (tools execute on the OCP host).
|
||||
- **README restructured, 1205 → ~500 lines (#172).** Operations-manual content moved to `docs/lan-mode.md`, `docs/tui-mode.md`, `docs/troubleshooting.md`, `docs/upgrading.md` (verbatim moves + two canonical dedups; zero content loss verified section-by-section). README keeps the quickstart, the release-kit-pinned reference tables, and summary stubs with links. Plus a staleness sweep (#170): 6-model examples, removal of the never-existed `ocp stop` command, `ocp-connect` claims corrected, current version examples.
|
||||
|
||||
## v3.22.1 — 2026-07-17
|
||||
|
||||
Minor release: TUI-mode latency and streaming features — **all opt-in and off by default**, so the default request path (`-p` / `--output-format stream-json`) is byte-for-byte unchanged — plus hardening from an independent (Codex) re-review of the streaming work, Windows `claude.exe` startup resolution, and the Claude Sonnet 5 model entry. No new `cli.js` wire behavior and no new endpoint; the new surface is entirely OCP-owned TUI-mode configuration (env vars), startup binary discovery, model metadata, and `/health` observation. Every code PR carried a fresh-context reviewer (Iron Rule 10). (Version note: v3.22.0 was prepared but never tagged; its contents ship here as v3.22.1 together with the additions below.)
|
||||
|
||||
### Added
|
||||
|
||||
- **Claude Sonnet 5 in the model SPOT (#152, contributed by @vvlasy-openclaw)** — `claude-sonnet-5` added to `models.json` (`contextWindow` 200000 / `maxTokens` 16384 / `reasoning` true, consistent with existing entries), exposed via `/v1/models` and the OpenClaw sync. Purely additive: the `sonnet` alias still resolves to `claude-sonnet-4-6` (the repoint is tracked separately in #168). `ocp-connect`'s model classifier now matches on the model *family* prefix (`claude-sonnet`/`claude-opus`/`claude-haiku`) instead of version-pinned prefixes, so current and future versioned IDs register with correct `reasoning`/`maxTokens` metadata. New referential-integrity tests guard that every alias target exists in `models[]`.
|
||||
- **Windows `claude.exe` startup resolution (#161, contributed by @nyxst4ck, diagnosis credit #147 @Justinsato)** — on Windows, `resolveClaude()` now discovers a native `claude.exe` (`%USERPROFILE%\.local\bin`, WinGet Links, WindowsApps, then `where.exe`) and rejects npm `.cmd`/`.bat`/`.ps1` shims, which cannot be spawned without a shell — previously startup resolved a shim and failed. A non-`.exe` `CLAUDE_BIN` on Windows is a fatal error with an actionable hint. The macOS/Linux path is byte-for-byte unchanged. Note: this is startup binary resolution only — full Windows support is not yet claimed (snapshot-path portability is tracked in #167).
|
||||
|
||||
### Added — TUI mode (all opt-in, default off)
|
||||
|
||||
- **Spawn effort control — `OCP_TUI_EFFORT` (default `low`) (#156)** — the interactive `claude` is now spawned with an explicit `--effort` flag. `low` cuts measured TTFT p50 by ~40% and collapses run-to-run variance ~15× versus an inherited `xhigh`; proxied requests rarely benefit from extended thinking. Set `inherit` to omit the flag and restore the pre-flag HOME-dependent behaviour. Banner-verified to stay on the subscription pool (`· Claude Max`); an invalid value warns and falls back to `low`. README § "Environment Variables".
|
||||
- **Warm pane pool — `OCP_TUI_POOL_SIZE` (default `0` / off) (#158)** — pre-boots up to 4 single-use `claude` panes so a request skips the cold boot: measured end-to-end p50 `10.17s` → `6.00s` (−41%) on a Mac mini (Sonnet 4.6, `--effort low`). Opt-in because each warm pane is a live idle process held whether or not a request ever arrives. Panes are single-use (one turn, then killed and replaced in the background), port-scoped (`ocp-tui-<port>-p<hex>`), and coexist with the zombie reaper by a synchronous drain→reap→resume sweep. README §§ "Environment Variables" + "How It Works".
|
||||
- **Real SSE streaming — `OCP_TUI_STREAM` (default `0` / off) (#159, #160)** — `stream:true` turns emit real `delta.content` chunks as `claude` generates them, sourced from `claude`'s own `MessageDisplay` hook (registered via `--settings` on the ordinary interactive spawn — banner-verified on the subscription pool). Granularity is block-level, and it moves the *first* byte, not the last. The transcript stays authoritative: streamed text is asserted equal to it at end-of-turn, the auth-banner and truncation gates still run before anything is committed, and a turn whose stream cannot be reconciled is **refused** (SSE error frame, not cached) and counted on `/health` (`tui.streamDivergences`; a silent total-hook-failure is counted separately as `tui.streamZeroDeltaTurns`). Tunables: `OCP_TUI_STREAM_HOLDBACK` (default `100`), `OCP_TUI_STREAM_DIR`, `OCP_TUI_STREAM_POLL_MS`. See ADR 0007 (2026-07-13 amendment). README §§ "Environment Variables" + "How It Works".
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Streaming auth-banner guard: a null `message_id` on the first hook fire (#160)** — a first `MessageDisplay` fire with a null `message_id` could disarm the auth-banner guard; re-landed after a #159 squash dropped it (`lib/tui/stream.mjs`).
|
||||
- **Test suite wrote live, unrevoked API keys into the operator's real key store (#163)** — `npm test` had been opening `~/.ocp/ocp.db` (the running server's DB) and writing two junk `api_keys` rows per run (737 accumulated on the maintainer's host), because the isolation the comments claimed was never wired (ESM import hoisting). `keys.mjs` now honors `OCP_DIR_OVERRIDE` under `NODE_ENV=test` and the suite points at a scratch dir; a child-process probe verifies a production process (no `NODE_ENV`) cannot be redirected.
|
||||
- **Streaming holdback floor + billing-pool observation on failed turns (#164)** — (A1) `OCP_TUI_STREAM_HOLDBACK` now clamps up to the safe floor (`100`) with a boot warning, closing a latent auth-banner leak when an operator set a sub-floor value. (A3) the `cc_entrypoint` (billing-pool) observation is now recorded before the honesty gates that throw, so `/health` no longer goes blind to exactly the failed turns most likely to signal a silent degrade to the metered Agent SDK pool.
|
||||
- **Test-only key-store redirection vars can no longer reach a server OCP launches (#165)** — (A4) `NODE_ENV`/`OCP_DIR_OVERRIDE` are stripped from every service unit `setup.mjs` writes (`plist-merge`'s `NEVER_PRESERVE`) and from the `ocp restart` manual nohup fallback (`env -u`); #163's overstated "a prod server can NEVER be redirected" comments were softened to name the one residual hand-launch path and the loud `getDb()` "NOT the default" backstop.
|
||||
|
||||
### Docs
|
||||
|
||||
- **README billing honesty (#162, closes #136)** — removed a feature bullet that promised what the § "honest limits" section forbids.
|
||||
- **TUI latency plans + streaming-achievability spike (#155, #157)** — measured latency decomposition, backlog, and the `MessageDisplay`-hook streaming prereq spike under `docs/plans/2026-07-13-tui-latency/`.
|
||||
|
||||
## v3.21.1 — 2026-07-07
|
||||
|
||||
Patch release: three bug fixes from an independent concurrency/session-lifecycle audit, each its own PR with a fresh-context reviewer (Iron Rule 10). No new `cli.js` wire behavior, no new endpoint, header, or env var; the `/health` field set is unchanged (only value truthfulness improved).
|
||||
|
||||
### Fixed
|
||||
|
||||
- **TUI session-scope / boot-reap (#148)** — `lib/tui/session.mjs`'s tmux session prefix is now scoped per-instance by listen port (`ocp-tui-<port>-`) instead of a bare host-wide `ocp-tui-` constant, so a second OCP instance on the same host (e.g. a temporary verification instance) can no longer have its live TUI sessions reaped or `kill-server`'d by another instance's boot/periodic sweep. The one-time boot reap also claims exact-shape legacy `ocp-tui-<8hex>` sessions (pre-fix naming) once, to clean up zombies left behind across an in-place upgrade.
|
||||
- **`-p` spawn-token mutex + keychain caching (#150)** — the real-HOME token fallback used when the keychain token is within its 5-minute expiry window is now serialized behind a mutex, so concurrent `-p` spawns no longer race the same single-use refresh token against each other (the credential-fork hazard). Added a 30s TTL cache + last-good-label memoization for the keychain read, cutting per-spawn event-loop blocking. The isolation decision (`/health` isolated/real-home reporting) is now re-evaluated per spawn instead of memoized forever, so `/health` no longer misreports a stale decision. New module `lib/spawn-auth.mjs` extracts the pure, unit-testable primitives (mutex, TTL cache, expiry gate, label ordering).
|
||||
- **Concurrency queue / disconnect handling (#149)** — the shared semaphore now honors a runtime-lowered `maxConcurrent` immediately (previously a decrease was silently ignored until in-flight tasks finished on their own) and wakes queued waiters right away when the limit is raised. Queued `-p`/TUI requests are now linked to the client's HTTP connection via `AbortSignal`; a client that disconnects while queued is spliced out of the queue instead of still spawning `claude` once a slot frees. A singleflight follower whose leader disconnected now retries instead of inheriting a spurious 500, and a queued-then-disconnected request is no longer recorded as a usage failure or logged as an error (quiet disconnect handling).
|
||||
|
||||
## v3.21.0 — 2026-06-25
|
||||
|
||||
Cleanup + docs release: TUI dead-code removal, docs honesty, and release prep. No new `cli.js` wire behavior; the default path (`CLAUDE_TUI_MODE` unset) is byte-for-byte unchanged.
|
||||
|
||||
### TUI dead-code / footgun cleanup
|
||||
|
||||
- **A1 — removed inert entrypoint-env path** (`lib/tui/session.mjs`): deleted `resolveTuiEntrypointEnv()` and the redundant env-strip block in `runTuiTurn`. The `{env}` object passed to `spawnSync` (tmux itself) was the wrong target — tmux does NOT forward the spawning process's environment to the pane; the pane's `claude` gets its env exclusively from the `env` prefix string built inside `buildTuiCmd` (verified live 2026-06-01). The spawnSync env is now intentionally minimal (`HOME` only). Behavior is unchanged: `buildTuiCmd` already handled all claude-specific env vars via its prefix string.
|
||||
- **A2 — removed test-only transcript helpers** (`lib/tui/transcript.mjs`): deleted `encodeCwd()` and `transcriptPath()` exports and the tests that pinned them. Production resolves transcripts exclusively via `findTranscriptPath()` (glob by session-id), which is immune to the exact path-encoding rule. No non-test importers existed (grep confirms). A `// TODO` comment near `findTranscriptPath()` notes that a CI fixture-contract test would make claude-schema drift fail loudly.
|
||||
- **A3 — removed headless-unusable `--dangerously-skip-permissions` branch** (`lib/tui/session.mjs` + `README.md`): `OCP_TUI_FULL_TOOLS=1` now always takes the `--allowedTools` path. The removed branch pushed `--dangerously-skip-permissions` when `CLAUDE_SKIP_PERMISSIONS=true`; on claude v2.1.x this triggers an interactive bypass-acceptance screen that a headless tmux pane cannot answer → the turn hangs to the wallclock cap and bricks the pane. The working path is `--allowedTools` + scratch-home `settings.json` `additionalDirectories`. `CLAUDE_SKIP_PERMISSIONS` for the `-p` path is unchanged (still used in `server.mjs`).
|
||||
|
||||
### Docs
|
||||
|
||||
- **Client-tools boundary** (README `§ How It Works`): OCP is a text-prompt bridge only — it does not pass OpenAI `tools`/`functions` or Anthropic `tool_use` blocks to the client. Clients receive assistant TEXT only; client-local tool execution is not supported by design (bypassing `cli.js` = out of scope per `ALIGNMENT.md`).
|
||||
- **ToS honesty** (README `§ Deployment model & security`): pooling one Claude subscription across multiple distinct people may violate Anthropic's Consumer ToS and risk account suspension by the abuse classifier. The defensible framing is "one person, your own devices" — friends/team sharing is not. The prior language ("account terms are your call") was accurate but understated the risk.
|
||||
- **"Why OCP" posture** (README `§ Why OCP?`): new bullet making explicit that OCP drives the official `claude` CLI as-is — no OAuth token extraction, no binary patching, no protocol invention — so traffic looks like genuine Claude Code (`cc_entrypoint=cli`).
|
||||
- **Promotion plan** (`docs/PROMOTION.md`): "stable & visible" strategy covering goal (polish + low-key OSS visibility, NOT growth-hacking given the live ToS/billing risk), pre-requisites (stability first), honest ToS disclosure requirement, items explicitly skipped (multi-backend routing → OLP; gateway model-discovery; raw API passthrough → ALIGNMENT.md scope), TUI toggle as billing-split insurance, and low-key visibility actions. Framed as a recommendation for the maintainer to review, not a committed plan.
|
||||
|
||||
### Previously shipped (v3.20.x) — documented here for completeness
|
||||
|
||||
- **Default `-p` spawn-home isolation** (v3.20.0 / PR-A): per-request `claude` spawns run in a credential-free minimal scratch HOME (`$HOME/.ocp/spawn-home`, no `.credentials.json`/`settings.json`/plugins) with a neutral cwd and the env token, cutting per-request latency (measured ~10–28s → ~3–7s). Kill-switch: `OCP_SPAWN_REAL_HOME=1`. Active mode shown at startup and on `/health.spawn`.
|
||||
- **Bounded concurrency wait-queue** (v3.20.0 / PR-B): excess `-p` requests queue (up to `CLAUDE_MAX_QUEUE`, default 16) instead of being rejected; a full queue returns `HTTP 429` + `Retry-After` (not an opaque 500). New env vars: `CLAUDE_MAX_QUEUE`, `CLAUDE_QUEUE_RETRY_AFTER`. Surfaced on `/health.concurrency` + `/health.stats.queueRejections`.
|
||||
- **`ocp restart`** macOS `bootout`+`bootstrap` (v3.20.0 / PR-B): safe restart command that forces launchd to re-read the plist (unlike `kickstart -k` which reuses the cached env).
|
||||
- **`/ocp` plugin OpenClaw-2026.5.27 compat** (v3.20.0 / PR-C): gateway plugin updated for the current OpenClaw API version.
|
||||
|
||||
## v3.20.1 — 2026-06-13
|
||||
|
||||
TUI-mode auth hardening: fixes the recurring `Please run /login · API Error: 401` (the PI231 incident) and reaps leaked defunct `claude` sessions. ([#141](https://github.com/dtzp555-max/ocp/pull/141))
|
||||
|
||||
### Fixed
|
||||
|
||||
- **TUI 401 / credential corruption (#141)** — interactive `claude` prefers `~/.claude/.credentials.json` over the `CLAUDE_CODE_OAUTH_TOKEN` env var (unlike `-p` mode, where the env token wins). OCP TUI's per-request spawn + `kill-session` cycle raced claude's single-use refresh-token rotation, corrupting the refresh token to an empty string → permanent 401 that `claude /login` couldn't fix (each new spawn re-corrupted it). This bit Linux/file-based hosts specifically (macOS reads credentials from the Keychain, so Mac mini was immune). **Fix:** when `CLAUDE_CODE_OAUTH_TOKEN` is set, the TUI claude now runs in a credential-free scratch HOME (`<HOME>/.ocp-tui/home`, overridable by `OCP_TUI_HOME`) seeded with onboarding + cwd-trust but **no `.credentials.json`**, so the env token is the only credential and claude never runs the refresh path. Recurrence-proof — a later `claude login` can no longer break TUI. Also: `buildTuiCmd` passes `CLAUDE_CODE_OAUTH_TOKEN` to the spawn, and `reapStaleTuiSessions` reaps defunct `claude` sessions (tmux-server-owned zombies) via `kill-server` when no foreign session remains, plus a 15-min idle-gated periodic reap. When the env token is unset, behaviour is byte-for-byte unchanged (real-home + credentials.json). Two independent fresh-context reviewers (Iron Rule 10) + a live PI231 portability test (works with a corrupt credentials.json present). Authorized by the ADR 0007 PR-D amendment (Class B).
|
||||
|
||||
### Environment variables
|
||||
|
||||
- `CLAUDE_CODE_OAUTH_TOKEN` — when set on a TUI host, TUI authenticates via this long-lived token in a credential-isolated home (recommended; immune to credentials.json corruption).
|
||||
- `OCP_TUI_HOME` — overrides the TUI scratch home; if you previously pointed it at your real home, unset it to get the credential-isolated default.
|
||||
|
||||
## v3.20.0 — 2026-06-10
|
||||
|
||||
TUI-mode billing-safety hardening for the 2026-06-15 Anthropic billing split. A 5-dimension multi-agent audit (adversarial verification + live tests on all three hosts — PI231 / Oracle / Mac mini, claude 2.1.104 / 2.1.114 / 2.1.170) found the TUI subscription-pool path could silently bill the metered Agent SDK pool or poison the cache under realistic failure modes. Three PRs, each with a fresh-context reviewer (Iron Rule 10) and CI; the default path (`CLAUDE_TUI_MODE` unset) is byte-for-byte unchanged.
|
||||
|
||||
### TUI — honesty & cache correctness (#137)
|
||||
|
||||
- **C-1** — `callClaudeTui` now throws on a claude-CLI auth-failure banner (e.g. `Please run /login · API Error: 401 …`, `Failed to authenticate. API Error: 401 …`) instead of returning it as a real answer, so it is never cached, singleflight-shared, or counted as a model success. Conservative detector (whole trimmed text ≤100 chars + `API Error: 4xx` + auth keyword + no code/quote char); overridable via `CLAUDE_TUI_ERROR_PATTERNS`. Live-reproduced on PI231.
|
||||
- **C-2** — `readTuiTranscript` distinguishes a complete turn from a wallclock-truncated partial (`truncated` flag); `callClaudeTui` throws `tui_wallclock_truncated` so a partial is never cached or counted as success.
|
||||
- **C-3** — `verifyEntrypoint` reads the `entrypoint` field from any transcript line, not just `{system, turn_duration}` — some claude builds emit zero turn_duration lines (live-confirmed on Oracle's claude 2.1.114), which previously left the billing-drift assertion blind on those builds.
|
||||
- **C-4 (paste)** — short prompts (e.g. `hi`) could never pass paste-landing detection; threshold lowered. Live-reproduced on PI231.
|
||||
|
||||
### TUI — concurrency & observability (#139)
|
||||
|
||||
- **Concurrency** — `OCP_TUI_MAX_CONCURRENT` (default 2) bounds concurrent interactive `claude` boots via a queuing semaphore (`lib/tui/semaphore.mjs`); the slot is released on throw so honesty-gate / spawn failures never leak it; bounded wait-queue → `tui_queue_full` (503). Independent of the global `MAX_CONCURRENT` (8) — a TUI turn is a heavy per-request cold-boot of tmux+claude + up to 120s wallclock.
|
||||
- **Observability** — additive `/health` `tui` block (`enabled` / `entrypointMode` / `lastEntrypoint` / `entrypointMismatches` / `inflight` / `maxConcurrent`) so an operator can poll for a silent `sdk-cli` metered-pool drift (the audit's top risk) instead of grepping journald. Authorized by the ADR 0007 PR-B amendment under the ALIGNMENT grandfather provision (additive, behaviour-preserving — every pre-existing `/health` field unchanged).
|
||||
|
||||
### Operations (#138)
|
||||
|
||||
- `docs/runbooks/615-canary.md` — the 2026-06-15 credit-balance canary: quiesce, read the Agent SDK credit balance (manual — no programmatic API exists for that pool; OCP's `/usage` headers are subscription rate-limit data, not the credit pool), one TUI canary turn, confirm `entrypoint:cli` in the transcript, green/red decision tree, periodic auto-mode self-classification mini-canary.
|
||||
- `docs/runbooks/tui-flip-rollback.md` — flip/rollback per deployment (systemd `daemon-reload`; launchd `bootout`/`bootstrap`, not `kickstart -k`).
|
||||
- `setup.mjs` auth quick-test gated behind `OCP_SKIP_AUTH_TEST=1` (the `claude -p` probe draws from the metered Agent SDK pool after 6/15).
|
||||
|
||||
### New environment variables
|
||||
|
||||
- `OCP_TUI_MAX_CONCURRENT` — max concurrent interactive TUI turns (default 2) (#139).
|
||||
- `OCP_SKIP_AUTH_TEST` — skip the `claude -p` auth probe in `setup.mjs` (default off) (#138).
|
||||
|
||||
## v3.19.0 — 2026-06-02
|
||||
|
||||
TUI-mode reliability + proxy-purity release. Two fixes diagnosed and verified live on both test hosts (PI231 / Oracle, claude 2.1.104 / 2.1.114), each its own PR with a fresh-context reviewer (Iron Rule 10), then an adversarial multi-host test battery (0 hangs / 0 crashes / 0 injection / 0 leaks). The default path (`CLAUDE_TUI_MODE` unset) is byte-for-byte unchanged.
|
||||
|
||||
### TUI
|
||||
|
||||
- **#130** — Fixed the "stuck typing" hang on large multi-line prompts. Three root causes: (1) terminal-turn detection only recognized `{system, turn_duration}`, which older claude builds (e.g. 2.1.114) don't emit → the reader ran to the wallclock and returned partial text; now also accepts an `assistant` line with a final `stop_reason` (`end_turn`/`stop_sequence`/`max_tokens`), while `tool_use` stays non-terminal. (2) Large prompts pasted via `send-keys -l` delivered embedded newlines as separate Enter events → the prompt never landed; now uses `tmux load-buffer` + `paste-buffer -p` (bracketed paste, atomic). (3) The paste-landed check false-positived on claude's empty curly-quote placeholder → Enter fired into an empty box; now positive-signal-only (`[Pasted text]` / prompt text) with a readiness/paste-verify poll + fast-fail (deterministic ~5s error instead of a 120s wallclock hang).
|
||||
- **#4** — TUI-mode never injects the host's `CLAUDE.md` / auto-memory into proxied turns. OCP is a proxy: the proxied client (OpenClaw / an IDE) owns its own context and memory. `buildTuiCmd` now always sets `CLAUDE_CODE_DISABLE_CLAUDE_MDS` + `CLAUDE_CODE_DISABLE_AUTO_MEMORY` (unconditional — proxy purity is not an opt-in). Verified live with a marker `CLAUDE.md`: obeyed by the proxied turn before the fix, blocked after, on both hosts. Residual host-context vectors (managed-policy / `settings.json` / output-styles) tracked in #133. The env is delivered via an `env`-prefix on the tmux pane command (tmux does not forward the spawning process's environment, and `new-session -e` requires tmux ≥3.2 while the cloud host runs 2.7).
|
||||
|
||||
## v3.18.0 — 2026-06-01
|
||||
|
||||
Hardening release from a multi-agent code audit (1 P0 + 14 P2 + 2 P3 findings, each adversarially verified and independently reviewed) plus three follow-ups (#123–#125). Every change shipped as its own PR with a fresh-context reviewer (Iron Rule 10). The single-user default path (`AUTH_MODE=none`, no TUI) is behavior-identical **except** the `/health` change in #109.
|
||||
|
||||
### Security
|
||||
|
||||
- **#109 (P0)** — `/health` no longer advertises `PROXY_ANONYMOUS_KEY` to remote callers by default. The `anonymousKey` field is gated behind a new `PROXY_ADVERTISE_ANON_KEY=1` opt-in env var; localhost callers are always exempt. Prevents any LAN-reachable device from harvesting a working, quota-spending bearer credential from the unauthenticated `/health` endpoint. **Behavior change:** `ocp-connect` zero-config Path A now requires the server to set `PROXY_ADVERTISE_ANON_KEY=1`; otherwise pass `--key` or use anonymous access.
|
||||
- **#114** — Dashboard escapes all DB-sourced strings (key names, usage rows) before `innerHTML`; the revoke button uses a `data-` attribute + listener instead of an inline `onclick` a quote could break out of; `POST /api/keys` validates key names server-side (`[A-Za-z0-9 ._-]{1,64}`).
|
||||
- **#124** — Dashboard status/plan summary cards escaped too (uniform defense-in-depth over all `innerHTML` sinks).
|
||||
- **#111** — Streaming error paths strip filesystem paths from claude error text / stderr before sending them to clients (`sanitizeError`), matching the non-streaming path.
|
||||
|
||||
### Reliability / correctness
|
||||
|
||||
- **#110** — Non-array `messages` is rejected with a 400 (was silently hanging the connection until socket timeout); OpenAI array `content` is flattened into the prompt instead of dumped as raw JSON; a streamed upstream error now emits an SSE `error` frame instead of a success-looking `finish_reason:"stop"`.
|
||||
- **#111** — `res.on("close")` escalates SIGTERM→SIGKILL on client disconnect (closes a narrow re-occurrence of the #37 concurrency-slot leak on the hottest exit path); `overallTimer` is cleared on semantic completion so a slow-exiting child can't record a spurious post-success timeout; per-key quota is documented as best-effort (bounded overshoot ≤ `MAX_CONCURRENT`, cache hits uncounted).
|
||||
- **#113** — CLI/installer hardening: `ocp-plugin` restart uses the live uid + `dev.ocp.proxy`/`ocp-proxy` labels and drops the unsafe `pkill` fallback; `ocp-connect` quotes + `chmod 600`s the persisted key; `setup.mjs` XML-escapes and newline-validates injected service-unit secrets.
|
||||
|
||||
### Alignment / governance
|
||||
|
||||
- **#112** — OAuth token-refresh host (`platform.claude.com/v1/oauth/token`) re-verified against the compiled cli.js v2.1.154 (`strings`, no live probe) and recorded in `ALIGNMENT.md`; usage-probe and default request model now derive from `models.json` (ADR 0003 SPOT) instead of hardcoded IDs.
|
||||
- **#123** — The legacy `console.anthropic.com/v1/oauth/token` host is pinned in the `alignment.yml` blacklist so a future OAuth-host drift hard-fails CI; the blacklist now documents its dual purpose (known hallucinations + pinned wrong-host variants of a verified Class A endpoint).
|
||||
|
||||
### TUI
|
||||
|
||||
- **#115** — The TUI LAN gate refuses any non-loopback bind (not just literal `0.0.0.0`); the achieved `cc_entrypoint` is asserted each turn and a `tui_entrypoint_mismatch` warning is logged on a silent degrade to the metered sdk-cli pool.
|
||||
|
||||
### Refactor
|
||||
|
||||
- **#125** — `isLoopbackBind` extracted to `lib/net.mjs`, shared by `server.mjs` and the test suite (was duplicated via a copy-paste mirror).
|
||||
|
||||
### New environment variables
|
||||
|
||||
- `PROXY_ADVERTISE_ANON_KEY` — opt-in (default off); advertise `PROXY_ANONYMOUS_KEY` on the public `/health` body for remote zero-config discovery (#109).
|
||||
|
||||
## v3.17.1 — 2026-05-31
|
||||
|
||||
### Fix — code-audit P1/P2 hardening
|
||||
|
||||
Fixes from a multi-agent code audit (3 P1 + 5 P2, adversarially verified). The single-user default path (`AUTH_MODE=none`, no TUI) is behavior-identical.
|
||||
|
||||
**Availability / correctness (P1):**
|
||||
- Guard `proc.stdin` against EPIPE — a fast-failing spawned `claude` (auth error, bad model, large prompt) no longer crashes the single-process daemon.
|
||||
- Add `unhandledRejection`/`uncaughtException`/`clientError` safety nets + wrap all request-body read loops — a client aborting mid-upload no longer crashes the daemon.
|
||||
- TUI transcript reader: only `turn_duration` is terminal (was also `tool_use`), which silently truncated any TUI turn that used a built-in tool.
|
||||
|
||||
**Security gates / cache integrity (P2):**
|
||||
- `AUTH_MODE=multi`: the default spawn now passes `--disallowedTools` (Bash/Read/Write/Edit/…) so a guest prompt cannot drive operator-filesystem tools. Single-user path unchanged.
|
||||
- `/sessions` (DELETE), `/settings` (PATCH), `/logs`, `/usage`, `/status` are now admin-gated (were dispatched before the admin check).
|
||||
- Streaming path no longer caches an `is_error` response as success (cache-poisoning fix).
|
||||
- TUI fail-loud guard extended to `none`+`0.0.0.0` (unless `OCP_TUI_ALLOW_LAN=1`) and `+ PROXY_ANONYMOUS_KEY`.
|
||||
- TUI `send-keys` paste uses `-l` (literal) so a prompt equal to a tmux key token (e.g. `C-c`) is typed, not interpreted.
|
||||
|
||||
---
|
||||
|
||||
## v3.17.0 — 2026-05-31
|
||||
|
||||
### Provider — default claude invocation ported to stream-json + `--system-prompt` (Phase 6c)
|
||||
|
||||
OCP's default (non-TUI) claude spawn moves from `claude -p --output-format text` to `claude --output-format stream-json --verbose --no-session-persistence --system-prompt <wrapper>` (no `-p`). The NDJSON event stream is parsed into the assembled response. Benefits: ~64% per-request cost reduction and anti-hallucination via `--system-prompt` tool-use suppression. Clients see no API change — the OpenAI-compatible request/response shapes are identical. Faithful port of OLP's production-verified implementation; covered by 17 new stream-json parser tests.
|
||||
|
||||
⚠️ **Billing note:** from 2026-06-15 this default path carries `cc_entrypoint=sdk-cli` and bills against the Agent SDK credit pool. Use the new opt-in `CLAUDE_TUI_MODE` (below) to keep traffic on the Pro/Max subscription pool.
|
||||
|
||||
---
|
||||
|
||||
### feat(tui): opt-in CLAUDE_TUI_MODE — serve via interactive claude (cc_entrypoint=cli / subscription pool), single-user only; default stream-json path unchanged
|
||||
|
||||
From 2026-06-15 Anthropic routes `claude -p` / `--output-format` invocations to the Agent SDK credit pool (`cc_entrypoint=sdk-cli`). This feature adds an opt-in bridge: when `CLAUDE_TUI_MODE=true`, OCP serves each request via a real interactive `claude` session (no `-p`, no `--output-format`) so it carries `cc_entrypoint=cli` and bills against the Pro/Max subscription.
|
||||
|
||||
The complete string response is read from claude's native JSONL session transcript and replayed to callers as a normal OpenAI completion or chunked SSE. Clients see no API change. The default stream-json path is byte-for-byte unchanged when `CLAUDE_TUI_MODE` is unset.
|
||||
|
||||
**Security:** single-user / single-operator only. Never enable on a multi-user OCP. See ADR 0007 and README § "Subscription-pool (TUI) mode".
|
||||
|
||||
New env vars: `CLAUDE_TUI_MODE`, `CLAUDE_TUI_WALLCLOCK_MS`, `OCP_TUI_CWD`, `OCP_TUI_HOME`.
|
||||
New ADR: `docs/adr/0007-tui-interactive-mode.md`.
|
||||
New modules: `lib/tui/transcript.mjs`, `lib/tui/session.mjs` (shipped in preceding commits on this branch).
|
||||
|
||||
---
|
||||
|
||||
### Model — add claude-opus-4-8
|
||||
|
||||
Add `claude-opus-4-8` as the newest Opus to `models.json` (index 0, newest first). Repoint `aliases.opus` from `claude-opus-4-7` to `claude-opus-4-8`. `claude-opus-4-7` remains in the list callable by literal id. `legacyAliases.claude-opus-4` left pointing at `claude-opus-4-7` (no change — legacy alias tracks the prior generation). README Available Models table and model-count references updated accordingly.
|
||||
|
||||
---
|
||||
|
||||
## v3.16.4 — 2026-05-13
|
||||
|
||||
### Refactor — port-literal SPOT + CI guardrail
|
||||
|
||||
Closes the structural side of the port-drift cascade addressed by v3.16.2
|
||||
and v3.16.3. Those two releases reverted plist / plugin / scripts back to
|
||||
3456 line-by-line, but the underlying invitation to drift — a hardcoded
|
||||
port literal scattered across six source files — was still intact.
|
||||
|
||||
Changes:
|
||||
|
||||
- **New `lib/constants.mjs`** — single source of truth for shared literals.
|
||||
Exports `DEFAULT_PORT = 3456`, `LOCAL_HOST = "127.0.0.1"`,
|
||||
`OPENAI_API_BASE = "/v1"`, `LOCAL_PROXY_URL`.
|
||||
- **`server.mjs:127`, `setup.mjs:36`, `scripts/upgrade.mjs:137`,
|
||||
`scripts/doctor.mjs:84` + `:205`, `scripts/sync-openclaw.mjs:73`** —
|
||||
all replaced with imports from `lib/constants.mjs`. Behavior is
|
||||
identical; the literal `3456` now exists in exactly one place per
|
||||
language (`lib/constants.mjs` for `.mjs`, `ocp` + `ocp-connect` for
|
||||
bash, `test-features.mjs` for pinned historical-port tests).
|
||||
- **`.github/workflows/alignment.yml`** — extended the path filter to
|
||||
`setup.mjs`, `scripts/**`, `lib/**`, `ocp`, `ocp-connect`. Added a new
|
||||
`port-spot` hard-fail job that greps for any hardcoded `3478` or `3456`
|
||||
literal in `.mjs/.js/.ts/.json` outside the EXEMPT_REGEX (which lists
|
||||
`lib/constants.mjs`, `test-features.mjs`, the bash CLIs, docs, and the
|
||||
workflow itself). Any future PR re-introducing a hardcoded port
|
||||
literal will be blocked at CI before it can cascade.
|
||||
- Doc comments in `server.mjs` env-var summary and `setup.mjs` usage
|
||||
banner reworded so the literal `3456` no longer appears as
|
||||
documentation text (CI grep is intentionally aggressive — it does not
|
||||
parse comments — so doc strings reference `DEFAULT_PORT from
|
||||
lib/constants.mjs` instead).
|
||||
|
||||
No behavior change for any user. `CLAUDE_PROXY_PORT` env var remains
|
||||
the runtime override; the only difference is the unset-env fallback
|
||||
now flows through one shared constant.
|
||||
|
||||
ALIGNMENT.md hard-requirements: this PR modifies `server.mjs` (one-line
|
||||
import + one literal swap, mechanical). No cli.js operation changed;
|
||||
the citation requirement does not apply. SPOT principle (Rule 2 spirit)
|
||||
is the entire motivation.
|
||||
|
||||
## v3.16.3 — 2026-05-13
|
||||
|
||||
### Fixes — completes v3.16.2 port-drift revert
|
||||
|
||||
v3.16.2 reverted the plugin / `openclaw.plugin.json` / README / Mac mini
|
||||
plist back to `3456` (the historical source default since `593d0dc`), but
|
||||
missed three places in `scripts/` that still defaulted to `3478`. Those
|
||||
three lines were the residual cascade source: every time `ocp doctor` or
|
||||
`ocp upgrade` ran without `CLAUDE_PROXY_PORT` in the env, they probed
|
||||
`3478`, reported "OCP not responding" against a healthy 3456 instance,
|
||||
and (in the case of OpenClaw sync follow-ups on the maintainer's host)
|
||||
re-introduced 3478 into downstream config.
|
||||
|
||||
Changes:
|
||||
|
||||
- `scripts/upgrade.mjs:137` — default port `3478` → `3456`.
|
||||
- `scripts/doctor.mjs:84` — default port `3478` → `3456`.
|
||||
- `scripts/doctor.mjs:205` — default port `3478` → `3456`.
|
||||
|
||||
No behavior change for users who set `CLAUDE_PROXY_PORT` explicitly; env
|
||||
still takes precedence. The fix only affects the unset-env fallback,
|
||||
which now matches `server.mjs:126` and the rest of the codebase.
|
||||
|
||||
Test plan: existing `test-features.mjs` cases that pin
|
||||
`CLAUDE_PROXY_PORT=3478` continue to pass — they use the env path, not
|
||||
the default.
|
||||
|
||||
## v3.16.2 — 2026-05-12
|
||||
|
||||
### Fixes — corrects v3.16.1
|
||||
|
||||
The v3.16.1 fix was directionally correct (plugin now reads env first, falls back to a hardcoded default) but **the narrative and the hardcoded default were both wrong**.
|
||||
|
||||
What v3.16.1 said: "OCP server moved to 3478 default in v3.14+; plugin lagged at 3456."
|
||||
What is actually true:
|
||||
- **OCP server source default has been `3456` since `593d0dc` (initial release) and has never changed.** Every line in `server.mjs`, `setup.mjs`, and the `ocp` CLI still uses `3456` as the documented and code-level default.
|
||||
- The single OCP installation observed on `3478` is the maintainer's Mac mini, whose plist was rewritten with `--port 3478` during a PR #71 dogfood smoke-test accident on 2026-05-08 (see `~/.cc-rules/memory/learnings/subagent_setup_mjs_prod_host_collision.md`). The plist drift was never reconciled back to source default, and v3.16.1 incorrectly canonised the post-accident value as if it had been a release decision.
|
||||
|
||||
This release:
|
||||
- Restores the plugin fallback to `http://127.0.0.1:3456` to match server source default.
|
||||
- Updates `openclaw.plugin.json` `configSchema.proxyUrl.default` back to `3456`.
|
||||
- Restores README §"Environment Variables" `CLAUDE_PROXY_PORT` default to `3456`.
|
||||
- Plugin reads `OCP_PROXY_URL` env (full URL) first, then `CLAUDE_PROXY_PORT` env (port only), then falls back to `3456`. Hosts whose OCP plist injects a non-default port must also inject the same `CLAUDE_PROXY_PORT` into the OpenClaw plist for the plugin to follow.
|
||||
- Maintainer's Mac mini plist was reverted from `3478` to `3456` as part of this release deploy (no source change reflects this; it was a one-host correction).
|
||||
|
||||
### Governance
|
||||
|
||||
- No `cli.js` citation needed (no `server.mjs` change). ALIGNMENT.md Rule 2 not engaged.
|
||||
|
||||
## v3.16.1 — 2026-05-12 (superseded — narrative incorrect; see v3.16.2 erratum)
|
||||
|
||||
### Fixes (as shipped — note erratum above)
|
||||
|
||||
- **OCP plugin port lag** — `ocp-plugin/index.js` hard-coded `http://127.0.0.1:3456`. ~~While OCP server moved to 3478 in v3.14+,~~ **(corrected v3.16.2: no such move ever happened.)** The Mac mini's plist was on `3478` only as residue from a dogfood accident. Result: `/ocp` slash commands from the home Telegram bot returned "OCP error: fetch failed". v3.16.1 changed the plugin default to `3478` (wrong direction; v3.16.2 reverts to `3456`).
|
||||
|
||||
### Governance
|
||||
|
||||
- No `cli.js` citation needed (no `server.mjs` change). ALIGNMENT.md Rule 2 not engaged.
|
||||
|
||||
## v3.16.0 — 2026-05-10
|
||||
|
||||
### Features
|
||||
|
||||
- **`ocp doctor --check oauth`** (PR #93) — fast path that runs only the OAuth check, skipping
|
||||
version detection / from-version / git operations / models endpoint. ~50ms vs. full doctor's
|
||||
~200-500ms. Use cases: AI agent repair loops, post-`claude auth login` verify, quick health
|
||||
gates. Help text in `cmd_doctor_help` now reflects working behaviour.
|
||||
- **`ocp update --rollback --gc`** — manually garbage-collect old upgrade snapshots.
|
||||
Retention policy: keep last 5 snapshots OR snapshots newer than 30 days OR the single most
|
||||
recent (always-keep safety net). `--dry-run` previews. Successful `ocp update` runs auto-GC
|
||||
at the end of the full path; light path does not (no snapshot created there).
|
||||
|
||||
### Behavior changes
|
||||
|
||||
- After a successful cross-minor `ocp update`, the auto-GC emits `[gc] removed N old snapshots`
|
||||
to stderr if any were collected. Safe to ignore; manual gc is `ocp update --rollback --gc`.
|
||||
|
||||
### Governance
|
||||
|
||||
- No `cli.js` citation needed (no `server.mjs` change). ALIGNMENT.md Rule 2 not engaged.
|
||||
- PR #93 (--check oauth) merged separately; this release bundles it with the GC feature.
|
||||
|
||||
## v3.15.1 — 2026-05-10
|
||||
|
||||
### Fixes
|
||||
|
||||
- **doctor: dynamic `latest_version` from `origin/main:package.json`** — v3.15.0 doctor used a hard-coded `latest = "v3.14.0"` fallback, which made any v3.15.0+ install report `kind = upgrade` (against a stale value). `ocp update` would then attempt `git checkout v3.14.0` — a downgrade. Doctor now fetches `git -C ~/ocp show origin/main:package.json` to determine the actual latest version; on failure (offline, fresh clone with no remote), falls back to `currentVersion` so `kind = noop` instead of recommending a downgrade.
|
||||
|
||||
## v3.15.0 — 2026-05-10
|
||||
|
||||
### Features
|
||||
|
||||
- **`ocp doctor`** — health & upgrade-readiness check; primary entry for AI-driven debugging.
|
||||
`--json` mode emits a `next_action` with `ai_executable[]` for agents to run verbatim
|
||||
and `human_required[]` for steps requiring the user (typically only OAuth).
|
||||
- **`ocp update` cross-version path** — for cross-minor jumps (e.g. v3.10 → v3.14),
|
||||
`ocp update` now runs doctor → snapshot → `setup.mjs` (with the plist env-merge from
|
||||
PR #90) → service restart → post-flight `/health` + `/v1/models` verification.
|
||||
Same-patch updates retain the existing light path; users see no change for routine
|
||||
patch bumps.
|
||||
- **`ocp update --rollback`** — restore the most recent (or specified) upgrade snapshot.
|
||||
Snapshots are saved to `~/.ocp/upgrade-snapshot-<ISO-ts>/` and never auto-deleted.
|
||||
- **Fresh-install routing** — `ocp update` on installations < v3.4.0 routes to a fresh-install
|
||||
flow (with `--yes` to skip confirmation; AI agents pass this). OAuth survives via Claude
|
||||
Code's credential store; users do not re-OAuth unless their token was independently broken.
|
||||
- **AI prompt blocks in README** — §Installation, §Upgrading, and §Troubleshooting each
|
||||
start with a copy-paste prompt for Claude Code / Cursor / Copilot, so users can drive
|
||||
install / setup / upgrade through their existing AI assistant.
|
||||
|
||||
### Behavior changes
|
||||
|
||||
- `ocp update` may take 10–30s longer when a cross-minor jump triggers the full path
|
||||
(snapshot + post-flight). Patch bumps are unchanged.
|
||||
- Pre-v3.4.0 installs are routed to fresh-install rather than failing silently or
|
||||
half-migrating.
|
||||
|
||||
### Governance
|
||||
|
||||
- No `cli.js` citation needed (no `server.mjs` change). ALIGNMENT.md Rule 2 not engaged.
|
||||
- Depends on PR #90 (plist env merge bug fix; merged before this release).
|
||||
|
||||
## v3.14.0 — 2026-05-10
|
||||
|
||||
### Features (security hardening)
|
||||
|
||||
- **Per-key session isolation** (PR #86, S1) — the `sessions` Map in `server.mjs` is now keyed by `${keyName}|${conversationId}` instead of bare `conversationId`. Before this fix, two clients using distinct API keys but the same `session_id` value (e.g. both defaulting to `"default"`) would share the same `cli.js` subprocess and conversation history, creating a cross-tenant leak path. Post-fix each (key, session) pair is isolated end-to-end, extending the per-key cache isolation shipped in v3.13.0 D1 to the session layer.
|
||||
- **On-disk credential file modes 0700/0600** (PR #87, S2) — `setup.mjs` now creates `~/.ocp` at mode 0700 and both `admin-key` and `ocp.db` at mode 0600. An idempotent `reconcileFileModes()` call in `server.mjs` startup tightens any existing installation to these modes automatically on every launch, so existing prod boxes fix themselves without manual `chmod`. Before this fix, all three files were created at the process's default umask (typically world-readable 0644 / 0755), leaving plaintext credentials readable by other local users.
|
||||
- **`/api/usage` default scope = self; admin all-keys requires `?all=true`** (PR #88, S3) — the usage endpoint now applies a least-privilege default: anonymous callers receive only their own rows, non-admin authenticated callers receive only their own rows, and admin callers receive only their own rows unless they explicitly pass `?all=true`. When `?all=true` is used, an audit log line is emitted. Before this fix, any admin-token holder could silently enumerate usage data for every key on the server.
|
||||
|
||||
### Behavior changes
|
||||
|
||||
- **Breaking change for admin tooling**: `/api/usage` no longer returns all-keys data by default. Existing cron jobs, dashboards, or scripts that rely on the admin token seeing all-keys output must add `?all=true` to their request URL after upgrading to v3.14.0.
|
||||
- **File mode reconcile at server startup** logs a one-line notice per path when mode is tightened (e.g. `[security] tightened ~/.ocp/ocp.db → 0600`). No action is required from the operator; the reconcile is idempotent and silent when modes are already correct.
|
||||
- **`sessions` Map key is now `${keyName}|${conversationId}` internally.** No client-visible wire change — the `session_id` field in request/response is unchanged.
|
||||
|
||||
### Verification
|
||||
|
||||
- Stress-test pass: 11/11 phases including S1/S2/S3 security regression checks (Phase E, I, J). 35-minute sustained run, 60 calls, 0 errors, 0 timeouts. RSS dropped 51→47 MB across the window. Per-key cache isolation, singleflight, cache_control bypass, quota enforcement, file-mode reconcile, and scope guard against escalation all verified against running code.
|
||||
|
||||
### Governance
|
||||
|
||||
- All three PRs (#86, #87, #88) include the explicit `cli.js`-citation-not-applicable disclaimer (per PR #75 pattern) since they are OCP-internal access-control, session-state, and file-permission changes with no corresponding `cli.js` operation to cite.
|
||||
|
||||
### No new env vars / no public API surface change beyond the documented breaking change
|
||||
|
||||
This release adds no new env vars or endpoints. The only externally visible change is the `/api/usage` scope guard (breaking for admin all-keys consumers; see Behavior changes above).
|
||||
|
||||
## v3.13.0 — 2026-05-07
|
||||
|
||||
### Features (cache layer hardening)
|
||||
|
||||
- **Per-key cache isolation** (D1) — the cache key now includes the API key id, so distinct keys never share cache entries. Anonymous/unauthenticated callers share one `anon` pool. Hash format upgraded to `v2`; legacy v1-format rows orphan and are reaped by the existing TTL cleanup interval (no migration script).
|
||||
- **`cache_control` bypass** (D2) — when a request carries an Anthropic `cache_control` annotation (top-level or nested in a content array), OCP skips its own cache entirely. The caller is using Anthropic-side prompt caching deliberately, and OCP must not interfere. A `cache_skipped{reason: cache_control_present}` log line is emitted on bypass.
|
||||
- **Chunked stream replay** (D3) — when a streaming request hits the cache, the cached content is now emitted as multiple SSE chunks (80 codepoints/chunk, codepoint-safe via `Array.from()`) instead of a single large delta. Multibyte characters (CJK / emoji) stay intact.
|
||||
- **Singleflight stampede protection** (D4) — concurrent identical cache-miss requests now share one upstream `cli.js` spawn instead of spawning N processes. Followers receive byte-identical responses to what the leader returns. All-or-nothing failure semantics: if the leader errors, all followers receive the same error. Streaming-path singleflight is explicitly out of scope (TODO left for follow-up).
|
||||
|
||||
### Behavior changes
|
||||
|
||||
- `/cache/stats` response now includes additive fields `inflight` and `requesters` (current in-flight singleflight entries and total waiting callers). Existing fields `entries`, `totalHits`, `sizeBytes` are preserved unchanged.
|
||||
|
||||
### Governance
|
||||
|
||||
- New ADR [`docs/adr/0005-no-multi-provider.md`](docs/adr/0005-no-multi-provider.md): OCP stays single-provider (Anthropic via `cli.js` spawn). Multi-provider gateway refactor explicitly out of scope; cache improvements are explicitly in scope.
|
||||
- Design spec for this release: [`docs/superpowers/specs/2026-05-07-cache-upgrade-design.md`](docs/superpowers/specs/2026-05-07-cache-upgrade-design.md).
|
||||
|
||||
### No new env vars / no public API surface change
|
||||
|
||||
This release adds no new env vars or endpoints. All four improvements are internal correctness/concurrency upgrades to the existing `CLAUDE_CACHE_TTL`-gated cache layer. No client-observable wire shape change.
|
||||
|
||||
## v3.12.0 — 2026-04-25
|
||||
|
||||
### Features
|
||||
|
||||
- **Streaming heartbeat** — opt-in SSE comment frame (`: keepalive\n\n`) emitted during silent windows on the streaming response. Controlled by `CLAUDE_HEARTBEAT_INTERVAL` env var (ms; `0` = disabled, default). Covers both pre-first-byte and mid-stream tool-use pauses. Addresses #47. See [design doc](docs/superpowers/specs/2026-04-25-47-sse-heartbeat-design.md).
|
||||
- **`X-Accel-Buffering: no`** response header added to SSE responses so heartbeats survive nginx/Cloudflare default buffering.
|
||||
|
||||
### Behavior changes
|
||||
|
||||
- SSE headers are now sent immediately after the claude CLI spawns successfully, not on first stdout byte. The rare "spawn succeeded but subprocess died before any byte" path now closes the SSE stream cleanly rather than returning a JSON error.
|
||||
|
||||
### Config additions
|
||||
|
||||
| Variable | Default | Description |
|
||||
|---|---|---|
|
||||
| `CLAUDE_HEARTBEAT_INTERVAL` | `0` (disabled) | Interval in ms for SSE keepalive comment frames on streaming path. Resets on every real frame. |
|
||||
|
||||
## v3.11.1 — 2026-04-21
|
||||
|
||||
### Fixes
|
||||
- Concurrency slot leak on subprocess timeout (#37). The request-timeout handler called `proc.kill("SIGTERM")` without decrementing `stats.activeRequests`. A subprocess stuck in a syscall that ignored SIGTERM would hold its slot until (or beyond) the 5s SIGKILL escalation actually reaped it. Slot release is now wired to `proc.once("exit", cleanup)` so every termination path — normal close, error, SIGTERM, SIGKILL — releases the slot exactly once.
|
||||
|
||||
## v3.11.0 — 2026-04-20
|
||||
|
||||
### Features
|
||||
- `ocp update` now automatically syncs OpenClaw's registry with the latest models (scripts/sync-openclaw.mjs)
|
||||
- Server logs warn if OpenClaw registry drifts from models.json
|
||||
|
||||
### Refactor
|
||||
- models.json is now the single source of truth for model list
|
||||
- server.mjs and setup.mjs derive MODEL_MAP/MODELS from models.json
|
||||
- Adding a new model is now a one-file edit
|
||||
|
||||
### Fixes
|
||||
- OpenClaw's model dropdown now shows all 4 current models (opus-4-7, opus-4-6, sonnet-4-6, haiku-4.5) on existing installs after `ocp update`. Previously setup.mjs only wrote the registry at install time.
|
||||
@@ -0,0 +1,94 @@
|
||||
@AGENTS.md
|
||||
@~/.cc-rules/AGENTS.md
|
||||
|
||||
# OCP Project Session Instructions
|
||||
|
||||
> **WARNING — READ BEFORE WRITING ANY CODE IN THIS REPO**
|
||||
>
|
||||
> Before touching `server.mjs` or any network-facing surface, read [`./ALIGNMENT.md`](./ALIGNMENT.md) in full. The constitution is binding. Non-compliant commits are reverted.
|
||||
|
||||
---
|
||||
|
||||
## Before starting any task
|
||||
|
||||
1. Read `./ALIGNMENT.md`. Internalize the five Rules and the 2026-04-11 drift lesson.
|
||||
2. Run `/dev-start <task description>` to get a pre-flight plan that incorporates the iron rules, `SKILL_ROUTING.md`, this file, and `ALIGNMENT.md`.
|
||||
3. If the task touches `server.mjs`, locate the corresponding `cli.js` reference **before** drafting any code. No code is written ahead of the `grep cli.js` evidence.
|
||||
|
||||
---
|
||||
|
||||
## Hard requirements for `server.mjs` changes
|
||||
|
||||
Every PR that modifies `server.mjs` must satisfy all three of the following. A PR missing any one of them is blocked from merge.
|
||||
|
||||
1. **`cli.js` citation.** The commit message and PR body declare the corresponding `cli.js` function name and line number range, using the format `cli.js:NNNN` or `cli.js vE4 <functionName>`. If `cli.js` does not perform the operation, the PR must state this explicitly and justify scope under `ALIGNMENT.md` Rule 2 (in practice, this almost always means the PR should be closed).
|
||||
2. **CI blacklist pass.** The `alignment.yml` workflow must pass. The workflow greps `server.mjs` for known-hallucinated tokens (currently blocking `api.anthropic.com/api/oauth/usage`) and fails the build on any hit. New tokens are added via PR amendment to `alignment.yml`; removing entries requires an `ALIGNMENT.md` amendment PR. Do not suppress the workflow.
|
||||
3. **Independent reviewer (Iron Rule 10).** The implementation author may not self-approve. A separate reviewer — human or a subagent spawned with a fresh context — must read the diff, verify the `cli.js` citation by opening `cli.js` at the cited lines, and explicitly approve. A review comment that does not confirm the `cli.js` citation was checked is not a valid approval.
|
||||
|
||||
---
|
||||
|
||||
## Iron rules in force
|
||||
|
||||
This repo operates under the CC Development Iron Rules (CC 开发铁律) v1.3. Three rules are load-bearing for OCP work:
|
||||
|
||||
- **Iron Rule 10 (Code Review).** Every implementation phase has an independent reviewer. Self-review does not count. See `server.mjs` hard requirement #3 above.
|
||||
- **Iron Rule 11 (Incremental Diff Review).** Non-trivial work is split into the minimum reviewable unit — one PR per layer per severity. `ALIGNMENT.md`, `CLAUDE.md`, the PR template, and the CI workflow are therefore shipped as the same constitutional PR (they are one layer: governance), but any subsequent `server.mjs` remediation lands as its own PR.
|
||||
- **Iron Rule 12 (Pre-Brainstorm Prior-Art Search).** Before proposing any new endpoint or header, search GitHub, Anthropic docs, and the `cli.js` bundle. For OCP specifically, the `cli.js` grep is the decisive search: if it does not hit, Rule 2 of the constitution applies.
|
||||
|
||||
The full iron rules are at `~/.claude/CC_DEV_IRON_RULES.md` (symlinked from the cc-rules repo on the maintainer's workstations). Load them into session context with `/cc-rules` when needed.
|
||||
|
||||
---
|
||||
|
||||
## Skills relevant to this repo
|
||||
|
||||
- `/dev-start` — pre-flight planning, always first.
|
||||
- `/cc-rules` — load the iron rules into context.
|
||||
- `/agent-dispatch` — pick the correct model (opus for design and review, sonnet for straightforward edits, haiku for mechanical chores) before spawning any subagent.
|
||||
- `/cc-mem search <keyword>` — look up cross-machine memory for prior decisions, especially prior drift incidents.
|
||||
|
||||
---
|
||||
|
||||
## Commit message conventions
|
||||
|
||||
- Subject line uses Conventional Commits (`fix:`, `feat:`, `docs:`, `refactor:`, `chore:`).
|
||||
- Any assertion of the form "Claude Code uses X" or "cli.js uses X" in the body must be immediately followed by a citation in the form `cli.js:NNNN` or `cli.js vE4 <functionName>`. CI performs a soft check for this pattern on all commits in the PR.
|
||||
- Co-author trailer is required for LLM-assisted commits (`Co-Authored-By: Claude <model> <noreply@anthropic.com>`).
|
||||
|
||||
---
|
||||
|
||||
## Project-level escalation
|
||||
|
||||
If a design decision cannot be resolved by reference to `cli.js` and `ALIGNMENT.md`, escalate to the project maintainer via `/cc-chat` rather than guessing. Silent guessing is what produced the 2026-04-11 drift.
|
||||
|
||||
---
|
||||
|
||||
## Release kit overlay (CC 开发铁律 第五律 5.5)
|
||||
|
||||
This project's overlay per iron rule v1.4's 5.5. Machine-checkable declaration.
|
||||
|
||||
```yaml
|
||||
release_kit:
|
||||
version_source: package.json
|
||||
changelog: CHANGELOG.md
|
||||
release_channel:
|
||||
type: github-release
|
||||
tag_format: v{semver}
|
||||
auto_create_on_tag_push: true # via .github/workflows/release.yml
|
||||
docs_source: README.md
|
||||
resource_lists:
|
||||
- name: Available Models table
|
||||
location: README.md § "Available Models"
|
||||
source_of_truth: models.json
|
||||
- name: API Endpoints table
|
||||
location: README.md § "API Endpoints"
|
||||
- name: Environment Variables table
|
||||
location: README.md § "Environment Variables"
|
||||
new_feature_doc_expectations:
|
||||
- new CLI subcommand → README § "All Commands" + usage example
|
||||
- new env var → README § "Environment Variables" table
|
||||
- new auto-sync / hook → dedicated §, must document trigger + manual invocation + opt-out + any bootstrap quirk
|
||||
- new endpoint → README § "API Endpoints" table + any relevant Config/Troubleshooting §
|
||||
- new file / SPOT / schema → Architecture or contributor § with link
|
||||
bootstrap_quirk_policy:
|
||||
- any one-time migration quirk → README § "Troubleshooting"
|
||||
```
|
||||
-14
@@ -1,14 +0,0 @@
|
||||
FROM node:20-alpine
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY server.mjs ./
|
||||
COPY setup.mjs ./
|
||||
COPY package.json ./
|
||||
|
||||
ENV CLAUDE_SESSION_TOKEN="" \
|
||||
CLAUDE_COOKIES=""
|
||||
|
||||
EXPOSE 3456
|
||||
|
||||
CMD ["node", "server.mjs"]
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Tao Deng
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -1,251 +1,613 @@
|
||||
# openclaw-claude-proxy
|
||||
# OCP — Open Claude Proxy
|
||||
|
||||
> **Already paying for Claude Pro/Max? Use it as your OpenClaw model provider — $0 extra API cost.**
|
||||
[](LICENSE) [](https://github.com/dtzp555-max/ocp/releases) [](https://buymeacoffee.com/dtzp555)
|
||||
|
||||
A lightweight, zero-dependency proxy that lets [OpenClaw](https://github.com/openclaw/openclaw) agents talk to Claude through your existing subscription. One command to set up, one file to run.
|
||||
> **Already paying for Claude Pro/Max? Use your subscription as an OpenAI-compatible API — $0 extra cost.**
|
||||
|
||||
**Why?**
|
||||
- **$0 API cost** — uses your Claude Pro/Max subscription, not pay-per-token API
|
||||
- **Zero dependencies** — single Node.js file, no `npm install`
|
||||
- **One command setup** — `node setup.mjs` handles everything
|
||||
- **OpenAI-compatible** — standard `/v1/chat/completions` endpoint
|
||||
- **All Claude models** — Opus 4.6, Sonnet 4.6, Haiku 4
|
||||
- **Streaming support** — real-time SSE responses
|
||||
*Open source from day one, used daily by my family, maintained on nights and weekends. If OCP saves you money too, you can [☕ buy me a coffee](https://buymeacoffee.com/dtzp555) — [full story below](#support-ocp).*
|
||||
|
||||
## How it works
|
||||
*If OCP saves you a setup, a ⭐ helps other folks discover it. Issue reports are even more useful — that's the highest-quality feedback this project gets.*
|
||||
|
||||
OCP turns your Claude Pro/Max subscription into a standard OpenAI-compatible API on localhost. Any tool that speaks the OpenAI protocol can use it — no separate API key, no extra billing.
|
||||
|
||||
```
|
||||
OpenClaw Gateway → proxy (localhost:3456) → claude -p CLI → Anthropic (via OAuth)
|
||||
Cline ──┐
|
||||
OpenCode ───┤
|
||||
Aider ───┼──→ OCP :3456 ──→ Claude CLI ──→ Your subscription
|
||||
Continue.dev ───┤
|
||||
OpenClaw ───┘
|
||||
```
|
||||
|
||||
The proxy translates OpenAI-compatible `/v1/chat/completions` requests into `claude -p` CLI calls. Anthropic sees normal Claude Code usage under your subscription — no API billing, no separate key.
|
||||
One proxy. Multiple IDEs. All models. **$0 API cost.**
|
||||
|
||||
## Prerequisites
|
||||
## Contents
|
||||
|
||||
- **Node.js** ≥ 18
|
||||
- **Claude CLI** installed and authenticated (`claude login`)
|
||||
- **OpenClaw** installed
|
||||
- [Why OCP?](#why-ocp) · [Supported Tools](#supported-tools)
|
||||
- [Quickstart](#quickstart)
|
||||
- [How It Works](#how-it-works)
|
||||
- Reference: [Available Models](#available-models) · [API Endpoints](#api-endpoints) · [Environment Variables](#environment-variables)
|
||||
- Modes & operations: [LAN & multi-user](#lan--multi-user) → [`docs/lan-mode.md`](docs/lan-mode.md) · [Subscription-pool (TUI) mode](#subscription-pool-tui-mode) → [`docs/tui-mode.md`](docs/tui-mode.md) · [Upgrading](#upgrading) → [`docs/upgrading.md`](docs/upgrading.md)
|
||||
- [Built-in Usage Monitoring](#built-in-usage-monitoring) · [Response Cache](#response-cache) · [Structured Outputs](#structured-outputs-openai-response_format) · [Images / Multimodal](#images--multimodal-vision) · [OpenClaw Integration](#openclaw-integration)
|
||||
- [Troubleshooting](#troubleshooting) → [`docs/troubleshooting.md`](docs/troubleshooting.md)
|
||||
- [Repository Layout](#repository-layout) · [Security](#security) · [Governance](#governance) · [Support OCP](#support-ocp) · [License](#license)
|
||||
|
||||
## Quick Start (Node.js)
|
||||
## Why OCP?
|
||||
|
||||
There are several Claude proxy projects. OCP picks a specific lane: **align tightly with what `cli.js` actually does, observe + multiplex what's already there, don't extend the protocol.** What you get:
|
||||
|
||||
- **LAN multi-user keys** (v3.7.0) — reach one Claude Pro/Max subscription from your own devices across the LAN. Each device gets a per-key API token (no OAuth session leak), with independent usage tracking and one-line revocation. Pro/Max are **per-user** accounts — see [Sharing with family / a team — honest limits](docs/lan-mode.md#deployment-model--security-read-this) before extending access to other **people**.
|
||||
- **`ocp-connect` one-shot client setup** — one command on the client machine auto-configures OpenClaw, and detects Cursor, Cline, Continue.dev, and opencode to print ready-to-paste setup hints for each. No hunting for where each tool keeps its `OPENAI_BASE_URL`.
|
||||
- **Response cache with per-key isolation + singleflight** (v3.13.0). Optional SHA-256 prompt cache, isolated per API key (cross-user pollution is impossible by hash construction, not by application logic), with stampede protection on concurrent identical prompts. Off by default. ([PR #65](https://github.com/dtzp555-max/ocp/pull/65), [PR #66](https://github.com/dtzp555-max/ocp/pull/66))
|
||||
- **Per-key request quotas** (v3.8.0). Daily / weekly / monthly limits per key — set a kid's iPad to 20/day, a partner's laptop to 100/week. ([PR #18](https://github.com/dtzp555-max/ocp/pull/18))
|
||||
- **SSE heartbeat for long reasoning** ([v3.12.0](https://github.com/dtzp555-max/ocp/releases/tag/v3.12.0), opt-in). If you've ever watched your IDE die at the 60s idle mark during a long Claude tool-use pause — that's nginx/Cloudflare default behavior. OCP emits an SSE comment frame to keep the connection alive without polluting the response. ([PR #49](https://github.com/dtzp555-max/ocp/pull/49))
|
||||
- **`cli.js` alignment + CI guardrail.** LLM-assisted code drifts easily — it's tempting to invent plausible-looking endpoints that `cli.js` doesn't actually use. [`ALIGNMENT.md`](./ALIGNMENT.md) is binding: every endpoint OCP exposes must cite a `cli.js` line. The [`alignment.yml`](./.github/workflows/alignment.yml) CI workflow blocks PRs that introduce known-hallucinated tokens. The payoff is boring: your setup keeps working when `cli.js` ships its next minor.
|
||||
- **`models.json` single source of truth** (v3.11.0). Adding a model is one file edit; both `/v1/models` and the OpenClaw bootstrap derive from it. ([PR #30](https://github.com/dtzp555-max/ocp/pull/30))
|
||||
- **Drives the official CLI as-is, no binary patching.** OCP spawns the official `claude` CLI (or hosts it in an interactive tmux pane for TUI mode) — it does not extract OAuth tokens from memory, patch the binary, or invent protocol extensions. Traffic therefore looks like genuine Claude Code to Anthropic's classifiers (`cc_entrypoint=cli`). See `ALIGNMENT.md` for why this constraint is load-bearing.
|
||||
|
||||
### Comparison
|
||||
|
||||
OCP and the alternatives serve adjacent but distinct needs. Pick the one that fits your use case:
|
||||
|
||||
| Feature | OCP | claude-code-router | anthropic-proxy |
|
||||
|---|---|---|---|
|
||||
| Forwards Claude Code subscription as OpenAI API | yes | yes | yes |
|
||||
| Routes to multiple model backends (OpenAI, Gemini, etc.) | no | yes | partial |
|
||||
| SSE heartbeat for long reasoning | yes (opt-in) | no | no |
|
||||
| Per-key quota + LAN multi-user keys | yes | no | no |
|
||||
| Response cache | yes (opt-in) | no | no |
|
||||
| OpenClaw / IDE auto-config | yes | no | no |
|
||||
| Model-routing rules / model-switching | no | yes | no |
|
||||
| GitHub stars / ecosystem size | small | large | mid |
|
||||
| Governance discipline (CI-enforced alignment with cli.js) | yes | n/a | n/a |
|
||||
|
||||
**Plain English**: `claude-code-router` is the routing-and-switching power tool — pick it if you want to mix Anthropic, OpenAI, Gemini, and local models behind one endpoint. `anthropic-proxy` is the minimal forwarder. **OCP focuses on disciplined `cli.js`-aligned forwarding plus subscription multiplexing** — pick it if you want to reach one Claude Pro/Max subscription from your own IDEs and devices, with LAN auth, quotas, and a governance contract that prevents endpoint drift.
|
||||
|
||||
### Related: OLP — Open LLM Proxy
|
||||
|
||||
OCP is Claude-only by design. If you want to spread across **multiple LLM providers** (not just Claude), see the sibling project **[OLP — Open LLM Proxy](https://github.com/dtzp555-max/olp)**: the same spawn-the-provider-CLI approach, but across several provider CLIs behind one OpenAI-compatible endpoint, with intelligent fallback chains. It grew out of OCP in response to Anthropic's 2026-06-15 billing split — the idea being to spread subscription/quota risk across more than one provider. OCP remains the focused, Claude-only option; OLP is the multi-provider one.
|
||||
|
||||
OCP is single-maintainer + LLM-assisted, currently pre-1.0. It runs the maintainer's daily Claude Code workflow. If something breaks, [open an issue](https://github.com/dtzp555-max/ocp/issues).
|
||||
|
||||
## Supported Tools
|
||||
|
||||
Any tool that accepts `OPENAI_BASE_URL` works with OCP:
|
||||
|
||||
| Tool | Configuration |
|
||||
|------|--------------|
|
||||
| **Cline** | Settings → `OPENAI_BASE_URL=http://127.0.0.1:3456/v1` |
|
||||
| **OpenCode** | `OPENAI_BASE_URL=http://127.0.0.1:3456/v1` |
|
||||
| **Aider** | `aider --openai-api-base http://127.0.0.1:3456/v1` |
|
||||
| **Continue.dev** | config.json → `apiBase: "http://127.0.0.1:3456/v1"` |
|
||||
| **OpenClaw** [^openclaw] | `setup.mjs` auto-configures |
|
||||
| **Any OpenAI client** | Set base URL to `http://127.0.0.1:3456/v1` |
|
||||
|
||||
[^openclaw]: **OpenClaw** is an IDE-agnostic AI coding agent (sibling project to OCP). When OCP runs on the same machine, OpenClaw can use it as a local provider — see `scripts/sync-openclaw.mjs` and ADR 0004.
|
||||
|
||||
## Quickstart
|
||||
|
||||
The simplest path: ask your AI.
|
||||
|
||||
Paste this prompt to Claude Code / Cursor / Copilot:
|
||||
|
||||
```
|
||||
Install OCP for me. Read README §Quickstart and follow it.
|
||||
Tell me when I need to run `claude auth login`.
|
||||
```
|
||||
|
||||
The AI will run `git clone`, `npm install`, `node setup.mjs`, and tell you when to OAuth.
|
||||
|
||||
**Prerequisites:** macOS or Linux (Windows is not supported), Node.js 22.5+ (Node 23+ recommended), `git`, and the [Claude CLI](https://docs.anthropic.com/en/docs/claude-cli), authenticated:
|
||||
|
||||
```bash
|
||||
git clone https://github.com/dtzp555-max/openclaw-claude-proxy.git
|
||||
cd openclaw-claude-proxy
|
||||
npm install -g @anthropic-ai/claude-code
|
||||
claude auth login # prints a URL + code — open on any browser, sign in, paste code back
|
||||
```
|
||||
|
||||
# Auto-configure OpenClaw + start proxy + install auto-start
|
||||
**Install** (Server role — runs the proxy):
|
||||
|
||||
```bash
|
||||
git clone https://github.com/dtzp555-max/ocp.git
|
||||
cd ocp
|
||||
node setup.mjs
|
||||
```
|
||||
|
||||
That's it. The setup script will:
|
||||
1. Verify Claude CLI is installed and authenticated
|
||||
2. Add `claude-local` provider to `openclaw.json`
|
||||
3. Add auth profiles to all agents
|
||||
4. Start the proxy
|
||||
5. Install auto-start on login (launchd on macOS, systemd on Linux)
|
||||
`setup.mjs` verifies the Claude CLI, starts the proxy on port 3456, and installs auto-start (launchd on macOS, systemd on Linux). The `ocp` CLI lands at `~/ocp/ocp` — symlink it onto your PATH (`sudo ln -sf ~/ocp/ocp /usr/local/bin/ocp`, or `ln -sf ~/ocp/ocp ~/.local/bin/ocp`) or alias it (`alias ocp=~/ocp/ocp`); the rest of the docs assume `ocp` is on your PATH.
|
||||
|
||||
**Verify** — should list 6 models:
|
||||
|
||||
Then set your preferred Claude model as default:
|
||||
```bash
|
||||
openclaw config set agents.defaults.model.primary "claude-local/claude-opus-4-6"
|
||||
openclaw gateway restart
|
||||
curl http://127.0.0.1:3456/v1/models
|
||||
# claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-sonnet-5, claude-sonnet-4-6, claude-haiku-4-5-20251001
|
||||
```
|
||||
|
||||
## Security
|
||||
|
||||
- **Localhost only** — the proxy binds to `127.0.0.1` and is not exposed to the internet or your local network
|
||||
- **Bearer token auth (optional)** — set `PROXY_API_KEY` to require a Bearer token on all requests (except `/health`). When unset, auth is disabled for backwards compatibility
|
||||
- **No API keys for Claude** — authentication to Anthropic goes through Claude CLI's OAuth session, no Anthropic credentials are stored in the proxy
|
||||
- **Auto-start via launchd/systemd** — `node setup.mjs` installs a user-level launch agent (macOS) or systemd user service (Linux) so the proxy starts automatically on login
|
||||
- **Remove auto-start** at any time:
|
||||
**Connect one IDE** — point any OpenAI-compatible tool at the proxy, then reload your shell and start a tool (Cline / Continue / Cursor / OpenCode):
|
||||
|
||||
```bash
|
||||
export OPENAI_BASE_URL=http://127.0.0.1:3456/v1
|
||||
```
|
||||
|
||||
See [Supported Tools](#supported-tools) for per-tool config.
|
||||
|
||||
**LAN / multi-user** — reach OCP from your own devices, with per-key auth, quotas, and anonymous access:
|
||||
|
||||
```bash
|
||||
node setup.mjs --bind 0.0.0.0 --auth-mode multi
|
||||
```
|
||||
|
||||
The full LAN server + client handbook, headless (Pi / NAS / VPS) OAuth, key/quota/anonymous-access management, AI-assisted install prompts, and the deployment/security model live in **[docs/lan-mode.md](docs/lan-mode.md)**. Claude Pro/Max are per-user accounts — read the [honest limits of sharing](docs/lan-mode.md#deployment-model--security-read-this) before extending access to other people.
|
||||
|
||||
### Uninstall
|
||||
|
||||
```bash
|
||||
# From the cloned repo
|
||||
node uninstall.mjs
|
||||
```
|
||||
|
||||
## Manual Install
|
||||
Removes the launchd (macOS) or systemd (Linux) auto-start entry. Handles both legacy (`ai.openclaw.proxy` / `openclaw-proxy`) and current (`dev.ocp.proxy` / `ocp-proxy`) service names. Does not delete `~/.openclaw/`, `~/.ocp/`, or the cloned repo — remove those manually if desired.
|
||||
|
||||
### 1. Start the proxy
|
||||
## How It Works
|
||||
|
||||
```bash
|
||||
node server.mjs
|
||||
# or in background:
|
||||
bash start.sh
|
||||
```
|
||||
Your IDE → OCP (localhost:3456) → claude --output-format stream-json CLI → Anthropic (via subscription)
|
||||
```
|
||||
|
||||
### 2. Configure OpenClaw
|
||||
OCP translates OpenAI-compatible `/v1/chat/completions` requests into `claude --output-format stream-json` CLI calls. Anthropic sees normal Claude Code usage — no API billing, no separate key needed.
|
||||
|
||||
Add to `~/.openclaw/openclaw.json` under `models.providers`:
|
||||
> **Billing-policy status (as of 2026-07).** Anthropic announced (2026-05-14) that from 2026-06-15 the `claude -p` / Agent SDK path would move to a separate metered credit pool — then **paused the change on its effective date**: *"For now, nothing has changed: Claude Agent SDK, `claude -p`, and third-party app usage still draw from your subscription's usage limits"* ([official help article](https://support.claude.com/en/articles/15036540-use-the-claude-agent-sdk-with-your-claude-plan)). So the default path above currently bills your subscription. Anthropic has said it will give notice before any future change; if the split re-lands, OCP's opt-in [subscription-pool (TUI) mode](docs/tui-mode.md#subscription-pool-tui-mode) is the ready-made hedge — see the billing table there.
|
||||
|
||||
```json
|
||||
"claude-local": {
|
||||
"baseUrl": "http://127.0.0.1:3456/v1",
|
||||
"api": "openai-completions",
|
||||
"apiKey": "<your PROXY_API_KEY, or omit if auth disabled>",
|
||||
"models": [
|
||||
{
|
||||
"id": "claude-opus-4-6",
|
||||
"name": "Claude Opus 4.6",
|
||||
"reasoning": true,
|
||||
"input": ["text"],
|
||||
"cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 },
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
{
|
||||
"id": "claude-sonnet-4-6",
|
||||
"name": "Claude Sonnet 4.6",
|
||||
"reasoning": true,
|
||||
"input": ["text"],
|
||||
"cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 },
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
{
|
||||
"id": "claude-haiku-4",
|
||||
"name": "Claude Haiku 4",
|
||||
"reasoning": false,
|
||||
"input": ["text"],
|
||||
"cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 },
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 8192
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
### Client-tools boundary
|
||||
|
||||
### 3. Set as default model
|
||||
OCP is a **text-prompt bridge** to the official `claude` CLI. It does **not** pass through OpenAI `tools`/`functions` payloads or Anthropic `tool_use` blocks to the client. Clients (Cline, Cursor, OpenClaw, etc.) pointed at OCP receive **assistant TEXT only** — they never get `tool_calls` to execute locally.
|
||||
|
||||
```bash
|
||||
openclaw config set agents.defaults.model.primary "claude-local/claude-opus-4-6"
|
||||
openclaw gateway restart
|
||||
```
|
||||
Any tool use happens server-side, under the `--allowedTools` set configured on the OCP host. In default mode (no `CLAUDE_NO_CONTEXT`), the `claude` CLI's own built-in tools are available to the model; in TUI mode, the operator controls the tool surface via `OCP_TUI_FULL_TOOLS`. Either way, the tools run under the operator's credentials on the server, and the client sees only the final text output. Note that on the `-p` path OCP prepends a system-prompt wrapper telling the model it has **no** local access (right for a shared gateway) — a single-user loopback instance whose model *should* use its tools can flip this with `OCP_LOCAL_TOOLS=1` (see Environment Variables).
|
||||
|
||||
**Client-local tool execution is not supported by design.** Supporting it would require bypassing the `claude` CLI to call the raw Anthropic API directly — that is a different product, and is out of scope per `ALIGNMENT.md` (every OCP endpoint must correspond to something `cli.js` actually does).
|
||||
|
||||
**What this means for choosing OCP (workload fit).** LAN/multi-device OCP is built for **chat-class** workloads — Q&A, translation, scripting against the API, chat frontends, home-automation backends — where text in/text out is the whole job. It is **not** the right tool for a coding agent running on a *client* machine that needs the AI to read and edit *that machine's* files: tools execute on the OCP host, so the model can never touch the client's filesystem. For that workload, run `claude` (or a local OCP) directly on the machine where the code lives.
|
||||
|
||||
## Available Models
|
||||
|
||||
| Model ID | Claude CLI model | Notes |
|
||||
|----------|-----------------|-------|
|
||||
| `claude-opus-4-6` | opus | Most capable, slower |
|
||||
| `claude-sonnet-4-6` | sonnet | Good balance of speed/quality |
|
||||
| `claude-haiku-4` | haiku | Fastest, lightweight |
|
||||
| Model ID | Notes |
|
||||
|----------|-------|
|
||||
| `claude-opus-4-8` | Most capable (default for `opus` alias) |
|
||||
| `claude-opus-4-7` | Previous Opus, retained for pinning |
|
||||
| `claude-opus-4-6` | Older Opus, retained for pinning |
|
||||
| `claude-sonnet-5` | Latest Sonnet (default for `sonnet` alias) |
|
||||
| `claude-sonnet-4-6` | Previous Sonnet, retained for pinning |
|
||||
| `claude-haiku-4-5-20251001` | Fastest, lightweight (default for `haiku` alias) |
|
||||
|
||||
## Authentication
|
||||
The proxy supports optional Bearer token authentication via the `PROXY_API_KEY` environment variable.
|
||||
|
||||
**When `PROXY_API_KEY` is set**, all requests (except `GET /health`) must include a valid `Authorization: Bearer <token>` header. Requests with a missing or invalid token receive a `401 Unauthorized` response.
|
||||
|
||||
**When `PROXY_API_KEY` is not set**, authentication is disabled and all requests are accepted (backwards compatible with v1.6.x and earlier).
|
||||
|
||||
For local Claude Max / OAuth usage, avoid exporting `ANTHROPIC_API_KEY` globally into the shell that launches the proxy. `v1.7.1` also sanitizes `ANTHROPIC_*` variables before spawning Claude subprocesses to reduce accidental auth-path pollution.
|
||||
|
||||
|
||||
The proxy supports optional Bearer token authentication via the `PROXY_API_KEY` environment variable.
|
||||
|
||||
**When `PROXY_API_KEY` is set**, all requests (except `GET /health`) must include a valid `Authorization: Bearer <token>` header. Requests with a missing or invalid token receive a `401 Unauthorized` response.
|
||||
|
||||
**When `PROXY_API_KEY` is not set**, authentication is disabled and all requests are accepted (backwards compatible with v1.6.x and earlier).
|
||||
The canonical list lives in [`models.json`](./models.json) — the single source of truth as of v3.11.0. Both `server.mjs` (the `/v1/models` endpoint) and `setup.mjs` (the OpenClaw registration) derive from it. Adding a new model is now a one-file edit:
|
||||
|
||||
```bash
|
||||
# Start with auth enabled
|
||||
PROXY_API_KEY=my-secret-token node server.mjs
|
||||
|
||||
# Configure OpenClaw provider with the matching key
|
||||
# In openclaw.json, set apiKey under the claude-local provider:
|
||||
"claude-local": {
|
||||
"baseUrl": "http://127.0.0.1:3456/v1",
|
||||
"api": "openai-completions",
|
||||
"apiKey": "my-secret-token",
|
||||
...
|
||||
}
|
||||
# 1. Edit models.json — add an entry
|
||||
# 2. Bump version, commit, tag, push
|
||||
# 3. Users get it on next `ocp update`:
|
||||
# - OpenClaw: auto-synced via scripts/sync-openclaw.mjs
|
||||
# - Cline / Aider / Cursor / opencode: live /v1/models, picks up immediately
|
||||
# - Continue.dev: user edits their own config.json
|
||||
```
|
||||
|
||||
The proxy logs auth status on startup: `Auth: enabled (PROXY_API_KEY set)` or `Auth: disabled (no PROXY_API_KEY)`.
|
||||
## API Endpoints
|
||||
|
||||
| Endpoint | Method | Description |
|
||||
|----------|--------|-------------|
|
||||
| `/v1/models` | GET | List available models |
|
||||
| `/v1/chat/completions` | POST | Chat completion (streaming + non-streaming) |
|
||||
| `/health` | GET | Comprehensive health check (includes a `tui` block for TUI-mode drift/concurrency monitoring) |
|
||||
| `/usage` | GET | Plan usage limits + per-model stats |
|
||||
| `/status` | GET | Combined overview (usage + health) |
|
||||
| `/settings` | GET/PATCH | View or update settings at runtime |
|
||||
| `/logs` | GET | Recent log entries (`?n=20&level=error`) |
|
||||
| `/sessions` | GET/DELETE | List or clear active sessions |
|
||||
| `/dashboard` | GET | Web dashboard (always public) |
|
||||
| `/api/keys` | GET/POST | List or create API keys (admin only) |
|
||||
| `/api/keys/:id` | DELETE | Revoke an API key (admin only) |
|
||||
| `/api/keys/:id/quota` | GET/PATCH | View or set per-key quota (admin only) |
|
||||
| `/api/usage` | GET | Per-key usage stats (`?since=&until=&hours=&limit=`); returns self only by default — pass `?all=true` (admin only) for all-keys data |
|
||||
| `/cache/stats` | GET | Cache statistics (admin only) |
|
||||
| `/cache` | DELETE | Clear response cache (admin only) |
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `CLAUDE_PROXY_PORT` | `3456` | Listen port |
|
||||
| `CLAUDE_BIN` | `claude` | Path to claude binary |
|
||||
| `CLAUDE_TIMEOUT` | `120000` | Request timeout (ms) |
|
||||
| `PROXY_API_KEY` | *(unset)* | Bearer token for API authentication; when unset, auth is disabled |
|
||||
| `CLAUDE_PROXY_PORT` | `3456` | Listen port (server-side). Also consumed by the OpenClaw `ocp-plugin` to dial the local proxy. |
|
||||
| `OCP_PROXY_URL` | *(unset)* | Plugin-side full URL override (e.g. `http://10.0.0.5:3456`). Wins over `CLAUDE_PROXY_PORT` when both are set. Read by `ocp-plugin/index.js` only — server ignores it. |
|
||||
| `CLAUDE_BIND` | `127.0.0.1` | Bind address (`0.0.0.0` for LAN access) |
|
||||
| `CLAUDE_AUTH_MODE` | `none` | Auth mode: `none`, `shared`, or `multi` |
|
||||
| `OCP_ADMIN_KEY` | *(unset)* | Admin key for key management (multi mode) |
|
||||
| `CLAUDE_BIN` | *(auto-detect)* | Path to claude binary |
|
||||
| `CLAUDE_TIMEOUT` | `600000` | Request timeout (ms, default: 10 min) |
|
||||
| `CLAUDE_HEARTBEAT_INTERVAL` | `0` | Streaming SSE keepalive interval (ms). `0` = disabled. See ["Streaming heartbeat"](#streaming-heartbeat) below. |
|
||||
| `CLAUDE_MAX_CONCURRENT` | `8` | Max concurrent claude processes (`-p`/stream-json path) |
|
||||
| `CLAUDE_MAX_QUEUE` | `16` | Max requests **waiting** for a `-p` concurrency slot. Beyond `CLAUDE_MAX_CONCURRENT`, requests queue (up to this cap) instead of being rejected; when the queue is **also** full, the request gets `HTTP 429` + `Retry-After` (not an opaque 500). Surfaced on `/health.concurrency` + `/health.stats.queueRejections`. |
|
||||
| `CLAUDE_QUEUE_RETRY_AFTER` | `5` | Seconds advertised in the `Retry-After` header on a `-p` concurrency-overflow `429`. |
|
||||
| `CLAUDE_MAX_PROMPT_CHARS` | *(derived)* | Prompt truncation limit in chars. Default derives from the models.json SPOT: `max(contextWindow) × 3` — currently **600,000** (≈150–200k tokens). Setting this env var (or the runtime settings API) overrides the derivation absolutely. See [ADR 0009](docs/adr/0009-spot-derived-prompt-budget.md). Note: very large prompts burn subscription-window quota quickly and slow TTFT; the TUI-mode paste path is untested beyond ~hundreds of KB. Applies to **text only** — image bytes bypass this budget (see [Images / Multimodal](#images--multimodal-vision)). |
|
||||
| `OCP_STRUCTURED_MAX_ATTEMPTS` | `3` | Max attempts (initial + retries) to coerce a schema-valid JSON reply when a request uses OpenAI `response_format`. Fail-closed: a non-numeric value keeps the default. See [Structured Outputs](#structured-outputs-openai-response_format). |
|
||||
| `CLAUDE_SESSION_TTL` | `3600000` | Session expiry (ms, default: 1 hour) |
|
||||
| `CLAUDE_CACHE_TTL` | `0` | Response cache TTL (ms, 0 = disabled). Set to e.g. `300000` for 5-min cache. See [Response Cache](#response-cache). |
|
||||
| `CLAUDE_ALLOWED_TOOLS` | `Bash,Read,...,Agent` | Comma-separated tools to pre-approve |
|
||||
| `CLAUDE_SKIP_PERMISSIONS` | `false` | Bypass all permission checks |
|
||||
| `CLAUDE_MCP_CONFIG` | *(unset)* | Path to an MCP server config JSON, passed to the spawned `claude` as `--mcp-config` (both the `-p` path and TUI `OCP_TUI_FULL_TOOLS` panes) |
|
||||
| `CLAUDE_MAX_BODY_SIZE` | `5242880` | Max request body size (bytes, default 5 MB). Base64 image payloads inflate ~33%; raise this to admit larger multimodal requests. Fail-closed parsing: a garbage value keeps the default. |
|
||||
| `CLAUDE_IMAGE_ALLOW_URL` | `false` | Allow remote `http(s)` image URLs in `image_url` parts. **Off by default** (v1 supports base64 `data:` URIs only). When on, the URL is passed through to Anthropic as a `url` image source — **OCP does not fetch it** (no OCP-side SSRF surface); unreachable/blocked URLs surface as an API error. |
|
||||
| `CLAUDE_MAX_IMAGE_BYTES` | `5242880` | Per-image decoded-byte cap (default 5 MB). Over-cap images get `HTTP 413`. |
|
||||
| `CLAUDE_MAX_IMAGES` | `20` | Max image parts per request. Over-cap gets `HTTP 413`. |
|
||||
| `CLAUDE_MAX_IMAGE_TOTAL_BYTES` | `20971520` | Aggregate decoded-byte cap across all images in a request (default 20 MB). Over-cap gets `HTTP 413`. |
|
||||
| `CLAUDE_SYSTEM_PROMPT` | *(unset)* | Operator-wide system-prompt text appended (last) to every request's composed system prompt on the default `-p` path. TUI-mode panes are unaffected (they keep the interactive CLI's own system prompt). Echoed truncated on `/health.systemPrompt`. Note: changing this value and restarting auto-invalidates the response cache (the key carries a boot-config epoch, #177). |
|
||||
| `OCP_LOCAL_TOOLS` | *(unset)* | **Single-user, loopback only.** `=1` swaps the default *"you have no local filesystem/shell access"* system-prompt wrapper for a positive one telling the model it **may** use its tools. These are the **server-side `claude` tools** OCP spawns via `-p` (`--allowedTools`) — which, on a loopback instance, run on the operator's own machine, i.e. *local* tools. For a personal instance (e.g. an **OpenClaw** agent on its own local OCP) the default wrapper otherwise makes the model refuse to use tools it legitimately has. Changes **only the prompt**, never the tool surface (governed by `--allowedTools`/`--disallowedTools`; multi-tenant still `--disallowedTools` the whole FS surface). **Does not** enable client-side `tool_calls` for OpenClaw/Cline/etc. — that remains unsupported by design (see § How tools work). Fail-closed: OCP **refuses to boot** if `=1` is combined with `CLAUDE_AUTH_MODE=multi`, a non-loopback bind, or `PROXY_ANONYMOUS_KEY` (mirrors `OCP_TUI_FULL_TOOLS`, ADR 0007). **Inert in TUI mode** (the `-p` wrapper is unused there; the TUI tool surface is `OCP_TUI_FULL_TOOLS`) — a warning is logged. Off by default → the default path is byte-for-byte unchanged. Toggling it auto-invalidates the standard response cache (boot-config epoch, #177). |
|
||||
| `CLAUDE_NO_CONTEXT` | `false` | Suppress CLAUDE.md and auto-memory injection (pure API mode) |
|
||||
| `PROXY_API_KEY` | *(unset)* | Bearer token for shared-mode authentication |
|
||||
| `PROXY_ANONYMOUS_KEY` | *(unset)* | Well-known anonymous key (multi mode) — this exact string bypasses `validateKey()` and grants public access. Exposed via `/health.anonymousKey` only to localhost, or to all callers when `PROXY_ADVERTISE_ANON_KEY=1`. Full setup + security notes: [docs/lan-mode.md § Anonymous Access](docs/lan-mode.md#anonymous-access-optional). |
|
||||
| `PROXY_ADVERTISE_ANON_KEY` | *(unset)* | When `=1`, advertise `PROXY_ANONYMOUS_KEY` in the public `/health` body for remote zero-config discovery. Default off — `/health` is unauthenticated, so this exposes the shared key to any LAN-reachable device (issue #109). Localhost always sees it regardless. |
|
||||
| `CLAUDE_TUI_MODE` | `false` | **Opt-in, single-user only.** Set to `"true"` to serve requests via interactive `claude` (`cc_entrypoint=cli`, subscription pool). Refuses to boot under `AUTH_MODE=multi`. See [Subscription-pool (TUI) mode](docs/tui-mode.md#subscription-pool-tui-mode). |
|
||||
| `CLAUDE_CODE_OAUTH_TOKEN` | *(unset)* | OAuth bearer token — highest-precedence credential for the `-p` path, and the **recommended** credential for TUI-mode hosts (when set with `OCP_TUI_HOME` unset, OCP runs the TUI `claude` in a credential-isolated home). See [docs/tui-mode.md](docs/tui-mode.md#tui-other-vars) and the [permanent-401 fix](docs/troubleshooting.md#tui-401). |
|
||||
| `OCP_SPAWN_REAL_HOME` | *(unset)* | Kill-switch for the default `-p`/stream-json **spawn-home isolation** (latency fix). When unset and an OAuth token is resolvable, OCP runs the per-request `claude` spawn in a **credential-free minimal scratch home** (`$HOME/.ocp/spawn-home`, no `.credentials.json`/`settings.json`/plugins) with a neutral cwd and the env token — so it loads none of the operator's heavy global `~/.claude` (plugins/skills/hooks) or the project `CLAUDE.md`, cutting per-request latency (measured ~10–28s → ~3–7s). Set to `"1"` to force the legacy real-`HOME` spawn (no cwd override) even when a token exists. With **no** resolvable token, OCP falls back to the real `HOME` automatically (zero regression). Active mode is shown at startup and on `/health.spawn`. |
|
||||
| `CLAUDE_TUI_WALLCLOCK_MS` | `120000` | (TUI-mode) Maximum time in ms to wait for the native transcript to signal turn completion. Increase for long Opus thinking turns. |
|
||||
| `OCP_TUI_CWD` | `$HOME/.ocp-tui/work` | (TUI-mode) Scratch working directory where interactive claude sessions run. Transcripts land under `<HOME>/.claude/projects/<encoded-cwd>/`. Created automatically. |
|
||||
| `OCP_TUI_HOME` | *(auto)* | (TUI-mode) `HOME` claude runs under. When unset, OCP auto-picks a credential-isolated scratch home (env token set) or the real home (no token). Full home/credential strategy: [docs/tui-mode.md](docs/tui-mode.md#tui-other-vars). |
|
||||
| `OCP_TUI_ENTRYPOINT` | `cli` | (TUI-mode) Billing-classifier labeling: `cli` pins `cc_entrypoint=cli`; `auto` self-classifies via TTY; `off` leaves inherited env untouched. See [docs/tui-mode.md](docs/tui-mode.md#tui-entrypoint). |
|
||||
| `OCP_TUI_EFFORT` | `low` | (TUI-mode) `--effort` level for the interactive spawn (`low`/`medium`/`high`/`xhigh`/`max`/`inherit`). Explicit `low` cuts TTFT p50 ~40% vs an inherited `xhigh`; invalid values fall back to `low`. See [docs/tui-mode.md](docs/tui-mode.md#tui-other-vars). |
|
||||
| `OCP_TUI_STREAM` | `0` (off) | (TUI-mode) `=1` emits real SSE `delta.content` chunks (block-level) from claude's `MessageDisplay` hook instead of buffering; transcript stays authoritative and divergent turns are refused. Caveats (tool-using turns, zero-delta detection) in [docs/tui-mode.md § `OCP_TUI_STREAM`](docs/tui-mode.md#ocp-tui-stream). |
|
||||
| `OCP_TUI_STREAM_HOLDBACK` | `100` | (TUI-mode, streaming) Characters withheld before the first chunk — keeps the auth-banner gate alive and is the knob for tool-using turns. See [docs/tui-mode.md § `OCP_TUI_STREAM_HOLDBACK`](docs/tui-mode.md#ocp-tui-stream-holdback). |
|
||||
| `OCP_TUI_STREAM_DIR` | `$HOME/.ocp-tui/stream` | (TUI-mode, streaming) Directory for the hook script/settings + per-session delta sink (one sink per session-id, so concurrent turns never interleave). See [docs/tui-mode.md](docs/tui-mode.md#ocp-tui-stream). |
|
||||
| `OCP_TUI_STREAM_POLL_MS` | `100` | (TUI-mode, streaming) Interval at which OCP drains the delta sink; the hook fires at block granularity so a finer poll buys nothing. See [docs/tui-mode.md](docs/tui-mode.md#ocp-tui-stream). |
|
||||
| `OCP_TUI_MAX_CONCURRENT` | `2` | (TUI-mode) Max concurrent interactive TUI turns, independent of `CLAUDE_MAX_CONCURRENT`. Excess turns queue (bounded); a full queue yields 503. See [docs/tui-mode.md](docs/tui-mode.md#tui-other-vars). |
|
||||
| `OCP_TUI_POOL_SIZE` | `0` (off) | (TUI-mode) Number of pre-booted warm `claude` panes (max `4`) so a request skips the cold boot — measured p50 `10.17s` → `6.00s`. Each warm pane is a live idle process; panes are single-use. See [docs/tui-mode.md § `OCP_TUI_POOL_SIZE`](docs/tui-mode.md#ocp-tui-pool-size). |
|
||||
| `OCP_SKIP_AUTH_TEST` | *(unset)* | When `=1`, skip the `claude -p` auth probe during `setup.mjs`. Under the announced (currently **paused**) 2026-06-15 billing split this probe would draw from the metered Agent SDK credit pool; set this to avoid burning a probe on re-installs or `ocp update` runs. Auth is validated at the first real request. |
|
||||
| `OCP_TUI_FULL_TOOLS` | *(unset)* | (TUI-mode, **single-user only**) `=1` grants the interactive session the same tool surface as the `-p` path (`--allowedTools` + optional `--mcp-config`) so a trusted single operator can run a tool-using / MCP agent on the subscription pool. Safe because TUI refuses to boot under `AUTH_MODE=multi`. See [docs/tui-mode.md § `OCP_TUI_FULL_TOOLS`](docs/tui-mode.md#ocp-tui-full-tools). |
|
||||
|
||||
## API Endpoints
|
||||
### Streaming heartbeat
|
||||
|
||||
- `GET /v1/models` — List available models
|
||||
- `POST /v1/chat/completions` — Chat completion (streaming + non-streaming)
|
||||
- `GET /health` — Health check
|
||||
When `CLAUDE_HEARTBEAT_INTERVAL` is set to a positive integer (milliseconds), OCP emits an SSE comment frame (`: keepalive\n\n`) on streaming responses whenever the stream has been idle for that duration. The timer resets on every real chunk, so heartbeats only fire during genuine silent windows (for example, Claude CLI tool-use pauses of 30s–5min, or a long "processing large contexts" delay before the first token).
|
||||
|
||||
## Server / Advanced: Docker
|
||||
Use cases: downstream HTTP clients or load balancers with idle-connection timeouts that would otherwise abort a slow-but-alive request. `CLAUDE_HEARTBEAT_INTERVAL=30000` (30s) is a reasonable starting value if your downstream has a 60s idle timeout.
|
||||
|
||||
For server deployments or if you prefer Docker:
|
||||
Heartbeats are inert SSE comment lines — conforming SSE clients ignore them. If your downstream client's SSE parser crashes on comment frames, leave this disabled (the default) and file an issue so we can consider an alternate frame format.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/dtzp555-max/openclaw-claude-proxy.git
|
||||
cd openclaw-claude-proxy
|
||||
cp .env.example .env # add your CLAUDE_SESSION_TOKEN / CLAUDE_COOKIES
|
||||
docker compose up -d
|
||||
OCP also sends `X-Accel-Buffering: no` on SSE responses so nginx-default proxy buffering does not hold heartbeats in an upstream buffer.
|
||||
|
||||
### Runtime settings (no restart needed)
|
||||
|
||||
Many tunables can be changed live via `ocp settings <key> <value>` (or `PATCH /settings`) without restarting:
|
||||
|
||||
```
|
||||
$ ocp settings maxPromptChars 200000
|
||||
✓ maxPromptChars = 200000
|
||||
|
||||
$ ocp settings maxConcurrent 4
|
||||
✓ maxConcurrent = 4
|
||||
```
|
||||
|
||||
Or as a single command if you already have a `.env` ready:
|
||||
## LAN & multi-user
|
||||
|
||||
Run OCP as a server on an always-on device and reach your one Claude Pro/Max subscription from your own laptops, phones, and Pis across the LAN — with per-key API tokens, per-key usage tracking + quotas, response-cache isolation, and one-command client setup (`ocp connect` / `ocp-connect`). A shared **anonymous key** covers simple trusted-family sharing.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/dtzp555-max/openclaw-claude-proxy.git && cd openclaw-claude-proxy && docker compose up -d
|
||||
node setup.mjs --bind 0.0.0.0 --auth-mode multi
|
||||
ocp keys add laptop # then: ocp lan → prints the LAN IP + connect command
|
||||
```
|
||||
|
||||
Health check: `curl http://localhost:3456/health`
|
||||
⚠️ The per-key modes give usage tracking, quotas, and cache separation — **not** a security isolation boundary. The spawned `claude` runs with the operator's filesystem access and is not sandboxed per key, so only share with people you fully trust, on a trusted network. Pro/Max are per-user accounts; pooling across distinct people may violate Anthropic's ToS.
|
||||
|
||||
## Recovery after OpenClaw upgrade
|
||||
Full server + client handbook, headless OAuth, AI-assisted install prompts, key/quota/anonymous-access management, monitoring dashboard, and the [deployment/security model & honest limits](docs/lan-mode.md#deployment-model--security-read-this): **[docs/lan-mode.md](docs/lan-mode.md)**.
|
||||
|
||||
OpenClaw upgrades (`npm update -g openclaw`) **do not overwrite** the user config at `~/.openclaw/openclaw.json`. However, if the claude-local models stop working after an upgrade, follow these steps:
|
||||
## Subscription-pool (TUI) mode
|
||||
|
||||
### Quick diagnosis
|
||||
**Opt-in, single-user only.** `CLAUDE_TUI_MODE=true` serves requests through interactive `claude` (no `-p`, `cc_entrypoint=cli`) so they bill the Pro/Max **subscription pool** instead of the metered Agent SDK path. Because `claude` runs with the operator's filesystem access, it is **single-operator only** — never enable it on a multi-user OCP (it refuses to boot under `AUTH_MODE=multi`).
|
||||
|
||||
> **⚠️ Status (as of 2026-07): a hedge, not a necessity.** The 2026-06-15 billing split that made this matter was announced, then **paused on its effective date** — the default `-p` path currently bills your subscription. TUI-mode is kept ready for if/when a reworked change lands (Anthropic has promised advance notice).
|
||||
|
||||
Setup, the ~6-second latency floor, real-SSE streaming (`OCP_TUI_STREAM`), the warm-pane pool (`OCP_TUI_POOL_SIZE`), full-tool mode (`OCP_TUI_FULL_TOOLS`), `/health` drift monitoring, and the flip/canary runbooks: **[docs/tui-mode.md](docs/tui-mode.md)**.
|
||||
|
||||
## Upgrading
|
||||
|
||||
Run **`ocp update`** — it smart-picks the path. A **patch bump** (e.g. `v3.21.0 → v3.21.1`) takes the light path (git pull + npm install + restart); a **cross-minor** jump (e.g. `v3.18 → v3.22`) takes the full path (pre-flight, snapshot, `setup.mjs` with plist env-merge, restart, post-flight `/health` + `/v1/models` verification). `ocp update --check` shows available updates without applying.
|
||||
|
||||
Manual flags, rollback (`ocp update --rollback`), snapshots, and the OpenClaw model auto-sync (v3.11.0+): **[docs/upgrading.md](docs/upgrading.md)**.
|
||||
|
||||
## Built-in Usage Monitoring
|
||||
|
||||
Check your subscription usage from the terminal:
|
||||
|
||||
```
|
||||
$ ocp usage
|
||||
Plan Usage Limits
|
||||
─────────────────────────────────────
|
||||
Current session 21% used
|
||||
Resets in 3h 12m (Tue, Mar 28, 10:00 PM)
|
||||
|
||||
Weekly (all models) 45% used
|
||||
Resets in 4d 2h (Tue, Mar 31, 12:00 AM)
|
||||
|
||||
Extra usage off
|
||||
|
||||
Model Stats
|
||||
Model Req OK Er AvgT MaxT AvgP MaxP
|
||||
──────────────────────────────────────────────────────
|
||||
opus 5 5 0 32s 87s 42K 43K
|
||||
sonnet 18 18 0 20s 45s 36K 56K
|
||||
Total 23
|
||||
|
||||
Proxy: up 6h 32m | 23 reqs | 0 err | 0 timeout
|
||||
```
|
||||
|
||||
**Web Dashboard:** open `http://<host>:3456/dashboard` in any browser for real-time per-key usage, request history, plan utilization, and system health (screenshot + details in [docs/lan-mode.md § Monitoring](docs/lan-mode.md#monitoring-server-side)).
|
||||
|
||||
### All Commands
|
||||
|
||||
```
|
||||
ocp usage Plan usage limits & model stats
|
||||
ocp usage --by-key Per-key usage breakdown (LAN mode)
|
||||
ocp status Quick overview
|
||||
ocp health Proxy diagnostics
|
||||
ocp keys List all API keys (multi mode)
|
||||
ocp keys add <name> Create a new API key
|
||||
ocp keys revoke <name> Revoke an API key
|
||||
ocp connect <ip> One-command LAN client setup
|
||||
ocp doctor Health & upgrade-readiness check; primary entry for AI-driven debugging. --json produces a next_action for AI agents.
|
||||
ocp lan Show LAN connection info & IP
|
||||
ocp settings View tunable settings
|
||||
ocp settings <k> <v> Update a setting at runtime
|
||||
ocp logs [N] [level] Recent logs (default: 20, error)
|
||||
ocp models Available models
|
||||
ocp sessions Active sessions
|
||||
ocp clear Clear all sessions
|
||||
ocp restart Restart proxy
|
||||
ocp restart gateway Restart gateway
|
||||
ocp update Update to latest version
|
||||
ocp update --check Check for updates without applying
|
||||
ocp --help Command reference
|
||||
```
|
||||
|
||||
> **Note:** Terminal CLI uses `ocp <command>`; the OpenClaw gateway plugin exposes the same as `/ocp <command>` in Telegram/Discord (see [OpenClaw Integration](#openclaw-integration)).
|
||||
|
||||
## Response Cache
|
||||
|
||||
OCP can cache responses to avoid redundant Claude CLI calls for identical prompts — useful during development when the same prompt is sent repeatedly.
|
||||
|
||||
**Enable** by setting `CLAUDE_CACHE_TTL` (ms), or update at runtime with `ocp settings cacheTTL 300000`:
|
||||
|
||||
```bash
|
||||
# 1. Check if proxy is running
|
||||
curl http://127.0.0.1:3456/health
|
||||
# Expected: {"status":"ok"}
|
||||
|
||||
# 2. Verify Claude CLI works
|
||||
claude -p "hello" --model sonnet --output-format text
|
||||
# Expected: text response
|
||||
|
||||
# 3. Verify OpenClaw config
|
||||
cat ~/.openclaw/openclaw.json | grep -A3 claude-local
|
||||
export CLAUDE_CACHE_TTL=300000 # cache responses for 5 minutes
|
||||
```
|
||||
|
||||
### Common issues and fixes
|
||||
**How it works:**
|
||||
- Cache key = SHA-256 of `v2|<keyId or "anon">|model + messages + temperature + max_tokens + top_p`
|
||||
- **Per-key isolation** — different API keys never share cache entries; anonymous callers share one `anon` pool
|
||||
- Cache hits return instantly — no Claude CLI process spawned. **Streaming hits** are replayed as multiple SSE chunks (80 codepoints each), not one large delta, so incremental render is preserved
|
||||
- **`cache_control` bypass** — a request carrying an Anthropic `cache_control` annotation (top-level or nested in `content[]`) skips OCP's cache entirely, so it doesn't interfere with Anthropic-side prompt caching
|
||||
- **Singleflight stampede protection** — concurrent identical cache-miss requests share one upstream `cli.js` spawn; followers receive byte-identical responses (non-streaming path only; streaming-path singleflight is a known TODO)
|
||||
- Multi-turn conversations (with `session_id`) are never cached; expired entries are reaped automatically every 10 minutes
|
||||
|
||||
| Symptom | Cause | Fix |
|
||||
|---------|-------|-----|
|
||||
| Agent doesn't reply, no proxy logs | Gateway didn't load claude-local provider | Check `models.providers.claude-local` in `openclaw.json` |
|
||||
| Proxy reports `exit 1` | Claude CLI not logged in or token expired | Run `claude login` to re-authenticate |
|
||||
| `🔑 unknown` in `/status` | Normal — no API key, using OAuth | Does not affect functionality, safe to ignore |
|
||||
| `/status` shows Context 0% | Messages not reaching proxy (SSE format issue) | Ensure proxy is latest version with streaming support |
|
||||
| Gateway reports `invalid api type` | OpenClaw renamed API type in new version | Check `api` field is still valid (e.g., `openai-completions`) |
|
||||
| Proxy startup `EADDRINUSE` | Port 3456 already in use | `lsof -i :3456` to find and kill the old process |
|
||||
**Management:**
|
||||
```bash
|
||||
curl http://127.0.0.1:3456/cache/stats # { "entries": 42, "totalHits": 156, "sizeBytes": 284000, "inflight": 0, "requesters": 0 }
|
||||
curl -X DELETE http://127.0.0.1:3456/cache # clear all cached responses
|
||||
ocp settings cacheTTL 0 # disable at runtime
|
||||
```
|
||||
|
||||
### One-command recovery
|
||||
Cache is **disabled by default** (`CLAUDE_CACHE_TTL=0`). All data is stored locally in `~/.ocp/ocp.db`. **Hash format upgrade in v3.13.0:** legacy `v1` cache rows don't match new `v2`-format lookups; they orphan and are reaped by the TTL cleanup interval within one window — no migration script required.
|
||||
|
||||
## Structured Outputs (OpenAI `response_format`)
|
||||
|
||||
`/v1/chat/completions` honors OpenAI's [`response_format`](https://platform.openai.com/docs/api-reference/chat/create#chat-create-response_format) parameter so OpenAI-SDK clients that require machine-parseable JSON (Home Assistant AI Tasks, Honcho, BYO scripts) get JSON in `choices[].message.content` — not prose.
|
||||
|
||||
Supported shapes:
|
||||
|
||||
- `response_format: { "type": "json_schema", "json_schema": { "name", "strict", "schema" } }`
|
||||
- `response_format: { "type": "json_object" }`
|
||||
- `json_mode: true` — non-standard top-level alias honored by several OpenAI-compatible clients; treated as `json_object`.
|
||||
|
||||
When a structured request is detected, OCP:
|
||||
|
||||
1. Appends a strict JSON-only steering instruction to the request (no Markdown, no fences, no prose, must begin with `{` or `[`).
|
||||
2. Extracts the JSON from the model reply (unwraps a stray code fence / prose via a string-aware balanced slice).
|
||||
3. For `json_schema`, validates the result against the supplied schema (types, `required`, `enum`, `const`, `additionalProperties`, nullability, `items`, `min/maxItems`, and `$ref`/`$defs` + `allOf`/`anyOf`/`oneOf` composition — the shapes the official OpenAI SDK emits via `zodResponseFormat` / `client.beta.chat.completions.parse`). For `json_object`, the whole reply must parse as a single JSON value (a stray brace inside prose is not served as the answer).
|
||||
4. On a parse/validation miss, retries with a stronger instruction that names the failure, up to `OCP_STRUCTURED_MAX_ATTEMPTS` (default 3).
|
||||
5. If no valid JSON can be produced, returns OpenAI's assistant **`refusal`** field (`HTTP 200`, `message.content: null`, `message.refusal: "<reason>"`, `finish_reason: "stop"`) — the spec's own mechanism for "the model would not produce the required output" — rather than an invented error type or passing prose through. SDK clients take their written `refusal` branch.
|
||||
|
||||
A reply that carries **more than one** top-level JSON value (e.g. `Schema: {…}` then `Answer: {…}`) is rejected as ambiguous rather than silently serving the first — OCP never serves an unvalidated or arbitrarily-chosen extraction.
|
||||
|
||||
`message.content` for a structured request is the raw JSON string only — no fences, no reasoning, no wrapper. Non-structured requests are completely unaffected (normal conversational behaviour, streaming included). This is a Class B.1 endpoint extension authorized by ADR 0006; the pure logic lives in [`lib/structured-output.mjs`](./lib/structured-output.mjs) and is unit-tested in `test-features.mjs`.
|
||||
|
||||
**Caching & cost.** A structured request can cost up to `OCP_STRUCTURED_MAX_ATTEMPTS` metered `claude` spawns — each retry is a fresh spawn, burning subscription-window quota today and metered credits if the (currently **paused**) 2026-06-15 billing split re-lands (see the billing-policy status note in [How It Works](#how-it-works)) — so this feature adds cost-attack surface. Two guards bound it: (a) identical **concurrent** structured requests share one flight (single-flight dedup, so N callers ≠ N× spawns), and (b) when `CLAUDE_CACHE_TTL > 0`, a **validated** result is cached on a **structured-keyed** hash (the `response_format`/schema is folded into the key, so a JSON reply never collides with the conversational answer and different schemas never share a slot). A refusal is never cached. Operators concerned about cost can lower `OCP_STRUCTURED_MAX_ATTEMPTS` to `1` (no retries) or gate the surface behind per-key quotas (`/api/keys/:id/quota`).
|
||||
## Images / Multimodal (Vision)
|
||||
|
||||
`POST /v1/chat/completions` accepts OpenAI-style multimodal `content` parts, so a
|
||||
message can carry images alongside text and Claude will actually see them. This
|
||||
follows OpenAI's [vision](https://platform.openai.com/docs/guides/vision) /
|
||||
[chat-completions `image_url`](https://platform.openai.com/docs/api-reference/chat/create#chat-create-messages)
|
||||
request shape — no OCP-invented fields. (Class B.1 endpoint; see ADR 0006.)
|
||||
|
||||
Under the hood, when a request carries an image OCP feeds the conversation to the
|
||||
Claude CLI as Anthropic image blocks over `--input-format stream-json`. Text-only
|
||||
requests are completely unaffected (unchanged code path).
|
||||
|
||||
### Supported input
|
||||
|
||||
- **Base64 data URIs** (default, recommended):
|
||||
`data:image/png;base64,<...>`. Media types: `image/jpeg`, `image/png`,
|
||||
`image/gif`, `image/webp`.
|
||||
- **Remote `http(s)` URLs** — **off by default**. Set `CLAUDE_IMAGE_ALLOW_URL=1`
|
||||
to enable; the URL is passed through to Anthropic (OCP never fetches it itself,
|
||||
so there is no OCP-side SSRF surface).
|
||||
- Images may appear in **any** message in the history (multi-turn), not just the
|
||||
last one.
|
||||
- Non-image, non-text parts (audio, files) are **not** yet supported and are
|
||||
replaced with a `[non-text content omitted]` placeholder (deferred to a future
|
||||
version).
|
||||
|
||||
### Example (base64 data URI)
|
||||
|
||||
```bash
|
||||
cd ~/.openclaw/projects/claude-proxy # or wherever you cloned it
|
||||
git pull # pull latest version
|
||||
node setup.mjs # reconfigure OpenClaw + start proxy
|
||||
openclaw gateway restart
|
||||
curl -X POST http://127.0.0.1:3456/v1/chat/completions \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"model": "claude-sonnet-4-6",
|
||||
"messages": [{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{ "type": "text", "text": "What is in this image?" },
|
||||
{ "type": "image_url",
|
||||
"image_url": { "url": "data:image/png;base64,iVBORw0KGgoAAA..." } }
|
||||
]
|
||||
}]
|
||||
}'
|
||||
```
|
||||
|
||||
### Pre-upgrade backup (recommended)
|
||||
### Not supported in TUI mode
|
||||
|
||||
Multimodal images require the default `-p` spawn path. In **TUI / subscription-pool
|
||||
mode** (`CLAUDE_TUI_MODE=true`) the CLI is driven interactively and cannot carry
|
||||
image blocks, so a request with an `image_url` part returns **`400
|
||||
images_unsupported_in_tui_mode`** rather than silently dropping the image and
|
||||
answering about something the model never saw. Remove the images, or run OCP
|
||||
without TUI mode, to use vision.
|
||||
|
||||
Images must also live in a **user or assistant** message, not a `system` message
|
||||
(system content is not forwarded to the CLI as image blocks). An `image_url` part
|
||||
present only in a system message returns **`400 images_unsupported_in_system_messages`**
|
||||
for the same reason — fail loudly rather than answer about an unseen image. This matches
|
||||
the OpenAI vision spec, which does not place images in the system role.
|
||||
|
||||
### Limits
|
||||
|
||||
Images bypass the text `CLAUDE_MAX_PROMPT_CHARS` budget and are instead bounded by
|
||||
their own byte/count caps. The **text** in a multimodal request is still subject to
|
||||
`CLAUDE_MAX_PROMPT_CHARS` (older text is truncated exactly as on the text-only
|
||||
path — only the image bytes are exempt). All numeric caps are parsed **fail-closed**:
|
||||
a malformed value (e.g. `CLAUDE_MAX_BODY_SIZE=unlimited` or `=5MB`) is rejected with
|
||||
a startup warning and the safe default is kept — a misconfigured cap can never
|
||||
silently disable the guard. Requests that violate a cap get a clear `4xx` (never a
|
||||
silent drop):
|
||||
|
||||
| Cap | Env var | Default | Error |
|
||||
|-----|---------|---------|-------|
|
||||
| Request body | `CLAUDE_MAX_BODY_SIZE` | 5 MB | `413` request body too large |
|
||||
| Per-image bytes | `CLAUDE_MAX_IMAGE_BYTES` | 5 MB | `413` `image_too_large` |
|
||||
| Total image bytes | `CLAUDE_MAX_IMAGE_TOTAL_BYTES` | 20 MB | `413` `images_too_large` |
|
||||
| Image count | `CLAUDE_MAX_IMAGES` | 20 | `413` `too_many_images` |
|
||||
| Unsupported media type | — | — | `400` `unsupported_image_type` |
|
||||
| Malformed data URI | — | — | `400` `invalid_data_uri` |
|
||||
| Remote URL while disabled | `CLAUDE_IMAGE_ALLOW_URL` | off | `400` `remote_url_disabled` |
|
||||
|
||||
Base64 payloads are large: a 5 MB image is ~6.7 MB as a data URI, so raise
|
||||
`CLAUDE_MAX_BODY_SIZE` (and, if needed, `CLAUDE_MAX_IMAGE_BYTES`) to admit big
|
||||
images. Vision support depends on the target model — request a current
|
||||
vision-capable Claude model.
|
||||
|
||||
## OpenClaw Integration
|
||||
|
||||
OCP was originally built for [OpenClaw](https://github.com/openclaw/openclaw) and includes deep integration:
|
||||
|
||||
- **`setup.mjs`** auto-configures the `claude-local` provider in `openclaw.json` at install time
|
||||
- **`ocp update`** auto-syncs the `claude-local` model registry from `models.json` (v3.11.0+) — no more stale model dropdowns after upgrades
|
||||
- **Gateway plugin** registers `/ocp` as a native slash command in Telegram/Discord
|
||||
- **Multi-agent** — 8 concurrent requests sharing one subscription
|
||||
- **No conflicts** — uses neutral service names (`dev.ocp.proxy` / `ocp-proxy`) that don't trigger OpenClaw's gateway-like service detection
|
||||
|
||||
**Install the gateway plugin:**
|
||||
|
||||
```bash
|
||||
cp ~/.openclaw/openclaw.json ~/.openclaw/openclaw.json.bak
|
||||
cp -r ocp-plugin/ ~/.openclaw/extensions/ocp/
|
||||
```
|
||||
|
||||
## Notes
|
||||
Add to `~/.openclaw/openclaw.json`, then `openclaw gateway restart`:
|
||||
```json
|
||||
{
|
||||
"plugins": {
|
||||
"allow": ["ocp"],
|
||||
"entries": { "ocp": { "enabled": true } }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
- Cost shows as $0 because billing goes through your Claude subscription
|
||||
- The `🔑` field in `/status` shows the configured auth key (or `unknown` if auth is disabled) — this is normal
|
||||
- Each request spawns a `claude -p` process; concurrent requests are supported
|
||||
- The proxy must run on the same machine as the Claude CLI (uses local OAuth)
|
||||
- The same Claude account can be used on multiple machines (shared usage quota)
|
||||
After installing, use `/ocp` slash commands in your chat: `/ocp status`, `/ocp usage`, `/ocp models`, `/ocp health`, `/ocp keys`, `/ocp keys add <name>`, `/ocp keys revoke <name>`.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
The simplest path: ask your AI — paste `Run `ocp doctor` and follow its `next_action`. Tell me if you hit anything that needs human input.` The doctor emits a JSON `next_action` with `ai_executable[]` (commands to run verbatim) and `human_required[]` (usually just OAuth).
|
||||
|
||||
**Most common issues:**
|
||||
|
||||
- **`EADDRINUSE: port 3456 already in use`** — an old OCP instance is bound. Find it (`lsof -nP -iTCP:3456 -sTCP:LISTEN`) and stop it (`launchctl bootout gui/$(id -u)/dev.ocp.proxy` on macOS, `systemctl --user stop ocp-proxy` on Linux). There is no `ocp stop` — the proxy is a service; `ocp restart` bounces it.
|
||||
- **`node: command not found` / version error** — OCP needs Node.js 22.5+ (`node --version`).
|
||||
- **`claude: command not found`** — install the Claude CLI, run `claude auth login`, then re-run `node setup.mjs`.
|
||||
- **Usage shows "unknown" / 401** — usually an expired Claude CLI session: `claude auth login && ocp restart`. For the *permanent* TUI-mode `Please run /login · API Error: 401` that re-login can't fix, see [docs/troubleshooting.md § permanent TUI-mode 401](docs/troubleshooting.md#tui-401).
|
||||
|
||||
**Bootstrap quirks (one-time migrations):**
|
||||
|
||||
- **A TUI session vanished right after upgrading OCP** — if a pre-3.21.1 and a post-3.21.1 instance ran on the same host at the same time during an upgrade, the new instance's one-time boot reap can, once, kill an old-format (`ocp-tui-<8hex>`) live TUI session belonging to the still-running old instance. Restart the affected session (`ocp restart` or re-run your TUI turn) and it returns under the new instance's port-scoped naming.
|
||||
- **OpenClaw shows old models after `ocp update` (v3.10→v3.11 only)** — the running shell had the old `cmd_update` cached, so the sync hook doesn't fire on that single jump. Run once: `node ~/ocp/scripts/sync-openclaw.mjs && openclaw gateway restart`. Every future update syncs automatically.
|
||||
|
||||
Full manual — setup failures, env-var-not-taking-effect-after-restart (launchd bootout+bootstrap vs `kickstart -k`), stuck sessions, "OpenClaw registry out of sync", and the two-layer TUI-mode 401 root cause + fix: **[docs/troubleshooting.md](docs/troubleshooting.md)**.
|
||||
|
||||
## Repository Layout
|
||||
|
||||
Top-level files a contributor or operator may need to know:
|
||||
|
||||
| Path | Role |
|
||||
|------|------|
|
||||
| `server.mjs` | The proxy itself; every request path lives here. Governed by `ALIGNMENT.md`. |
|
||||
| `setup.mjs` | First-time installer — verifies Claude CLI, patches OpenClaw config, installs auto-start. |
|
||||
| `uninstall.mjs` | Reverses the launchd / systemd auto-start install. |
|
||||
| `keys.mjs` | API-key management module (multi-mode auth: create/list/revoke, quotas, usage tracking). |
|
||||
| `models.json` | Single source of truth for model IDs, aliases, context windows. See ADR 0003. |
|
||||
| `ocp` / `ocp-connect` | User-facing CLI wrappers (server-side / client-side respectively). |
|
||||
| `dashboard.html` | Static dashboard served from `/dashboard`. |
|
||||
| `scripts/sync-openclaw.mjs` | Idempotent OpenClaw registry sync invoked by `ocp update`. See ADR 0004. |
|
||||
| `.claude/skills/` | Project-specific Claude Code skills. |
|
||||
| `ocp-plugin/` | OpenClaw gateway plugin (optional installation). |
|
||||
| `docs/lan-mode.md` | LAN & multi-user operations manual (server/client setup, keys, quotas, anonymous access, security model). |
|
||||
| `docs/tui-mode.md` | Subscription-pool (TUI) mode: setup, latency, streaming, warm-pane pool, drift monitoring. |
|
||||
| `docs/troubleshooting.md` | Full troubleshooting manual, including the permanent TUI-mode 401 root cause + fix. |
|
||||
| `docs/upgrading.md` | Upgrade manual (`ocp update` paths, rollback, OpenClaw auto-sync). |
|
||||
| `docs/adr/` | Architecture Decision Records. Read these before proposing governance or SPOT changes — see [`docs/adr/README.md`](docs/adr/README.md). |
|
||||
| `ALIGNMENT.md` | The constitution. Binding for any `server.mjs` change. |
|
||||
| `AGENTS.md` / `CLAUDE.md` | Agent and Claude-Code-specific session instructions. |
|
||||
|
||||
## Security
|
||||
|
||||
- **Localhost by default** — binds to `127.0.0.1`; set `CLAUDE_BIND=0.0.0.0` to enable LAN access
|
||||
- **3-tier auth** — `none` (trusted network), `shared` (single key), `multi` (per-user keys with usage tracking)
|
||||
- **Timing-safe key comparison** — prevents timing attacks on API keys and admin keys
|
||||
- **Admin-only key management** — creating, listing, and revoking keys requires the admin key
|
||||
- **Public endpoints** — `/health` and `/dashboard` are always accessible without auth
|
||||
- **No API keys needed** — authentication goes through Claude CLI's OAuth session
|
||||
- **Keys stored locally** — `~/.ocp/ocp.db` (SQLite), never sent to external services
|
||||
- **Auto-start** — launchd (macOS) / systemd (Linux)
|
||||
|
||||
## Governance
|
||||
|
||||
OCP runs under a small set of binding documents so contributions stay aligned with what `cli.js` actually does, not what an LLM thinks it does:
|
||||
|
||||
- **[`ALIGNMENT.md`](./ALIGNMENT.md)** — the constitution. Every endpoint OCP exposes must correspond to something `cli.js` actually does, with a line-number citation. Background in [ADR 0002](./docs/adr/0002-alignment-constitution.md).
|
||||
- **[`.github/workflows/alignment.yml`](./.github/workflows/alignment.yml)** — CI guardrail. Greps `server.mjs` for known-hallucinated tokens and fails the build on any hit. Not suppressible without an `ALIGNMENT.md` amendment PR.
|
||||
- **[`AGENTS.md`](./AGENTS.md)** — guidelines any AI coding agent (Claude Code / Cursor / Copilot / Codex / Gemini) should read before touching this repo.
|
||||
- **[`models.json`](./models.json)** — single source of truth for the model registry. See [ADR 0003](./docs/adr/0003-models-json-spot.md).
|
||||
- **[`docs/adr/`](./docs/adr/)** — architecture decision records explaining why current structure exists.
|
||||
|
||||
If you want to contribute: read `ALIGNMENT.md` first, search `cli.js` for the operation you're proposing, and cite the line number in your PR.
|
||||
|
||||
## Support OCP
|
||||
|
||||
OCP has been **open source from day one** — not a freemium tool, not a commercial product turned open, just open. It will stay that way forever. No paid tiers, no premium features, no "Pro" version locked behind a paywall.
|
||||
|
||||
I built it because my family and I needed it. We use OCP every day across our own machines and IDEs — keeping one Claude Pro/Max subscription powering everything, saving the per-token API cost we'd otherwise pay. It's been quietly heartwarming to hear from users online who say OCP has saved them money the same way it saves ours. That's the whole point.
|
||||
|
||||
Behind every version are hundreds of hours that don't show up in commits: building it from scratch, adding new features as the Claude Code ecosystem evolves, debugging across Mac / Windows / Linux machines, validating against half a dozen IDEs (Claude Code, Cursor, Cline, OpenCode, Aider, Continue.dev, OpenClaw), tracking down `cli.js` drift, OAuth refresh edge cases, SSE streaming quirks, concurrency leaks, and the occasional incident that turns into a multi-day investigation (the [2026-04-11 alignment drift](./docs/adr/0002-alignment-constitution.md), the [v3.11.1 concurrency leak](./CHANGELOG.md), the v3.12 SSE replay regression).
|
||||
|
||||
**The commitment**: this project will keep being updated, keep getting new features, and will stay open source as long as I'm able to maintain it.
|
||||
|
||||
**Please try it.** If something breaks or could be better, [open an issue](https://github.com/dtzp555-max/ocp/issues) — feedback is genuinely what keeps the project moving.
|
||||
|
||||
And if OCP saves you (or your team, or your family) real money and you'd like to chip in toward the next debugging session:
|
||||
|
||||
- ☕ **[Buy me a coffee](https://buymeacoffee.com/dtzp555)**
|
||||
|
||||
Donations directly fund the time it takes to keep OCP saving the community money.
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
MIT — see [`LICENSE`](LICENSE).
|
||||
|
||||
+273
@@ -0,0 +1,273 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<title>OCP Dashboard</title>
|
||||
<style>
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body { font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif; background: #0f172a; color: #e2e8f0; padding: 1.5rem; }
|
||||
h1 { font-size: 1.5rem; margin-bottom: 1rem; color: #38bdf8; }
|
||||
h2 { font-size: 1.1rem; margin: 1.5rem 0 0.5rem; color: #94a3b8; }
|
||||
.grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(200px, 1fr)); gap: 1rem; margin-bottom: 1.5rem; }
|
||||
.card { background: #1e293b; border-radius: 8px; padding: 1rem; }
|
||||
.card .label { font-size: 0.75rem; color: #64748b; text-transform: uppercase; }
|
||||
.card .value { font-size: 1.5rem; font-weight: 700; margin-top: 0.25rem; }
|
||||
.card .sub { font-size: 0.8rem; color: #94a3b8; margin-top: 0.25rem; }
|
||||
table { width: 100%; border-collapse: collapse; margin-top: 0.5rem; }
|
||||
th, td { text-align: left; padding: 0.5rem 0.75rem; border-bottom: 1px solid #334155; font-size: 0.85rem; }
|
||||
th { color: #64748b; font-weight: 600; }
|
||||
.tag { display: inline-block; padding: 2px 8px; border-radius: 4px; font-size: 0.75rem; }
|
||||
.tag-ok { background: #065f46; color: #6ee7b7; }
|
||||
.tag-err { background: #7f1d1d; color: #fca5a5; }
|
||||
.btn { background: #2563eb; color: white; border: none; padding: 0.5rem 1rem; border-radius: 6px; cursor: pointer; font-size: 0.85rem; }
|
||||
.btn:hover { background: #1d4ed8; }
|
||||
.btn-sm { padding: 0.25rem 0.5rem; font-size: 0.75rem; }
|
||||
.btn-danger { background: #dc2626; }
|
||||
.btn-danger:hover { background: #b91c1c; }
|
||||
input { background: #1e293b; border: 1px solid #334155; color: #e2e8f0; padding: 0.4rem 0.75rem; border-radius: 6px; font-size: 0.85rem; }
|
||||
.flex { display: flex; gap: 0.5rem; align-items: center; }
|
||||
.mono { font-family: "SF Mono", "Fira Code", monospace; font-size: 0.8rem; }
|
||||
.bar-bg { background: #334155; border-radius: 4px; height: 8px; overflow: hidden; margin-top: 0.5rem; }
|
||||
.bar-fill { height: 100%; border-radius: 4px; transition: width 0.5s; }
|
||||
.bar-blue { background: #3b82f6; }
|
||||
.bar-amber { background: #f59e0b; }
|
||||
.bar-red { background: #ef4444; }
|
||||
#refresh-indicator { font-size: 0.75rem; color: #475569; }
|
||||
.section { margin-bottom: 2rem; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
|
||||
<div class="flex" style="justify-content: space-between; margin-bottom: 1.5rem;">
|
||||
<h1>OCP Dashboard</h1>
|
||||
<div class="flex">
|
||||
<span id="refresh-indicator"></span>
|
||||
<button class="btn btn-sm" onclick="refreshAll()">Refresh</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="grid" id="status-cards"></div>
|
||||
|
||||
<div class="section">
|
||||
<h2>Plan Usage</h2>
|
||||
<div class="grid" id="plan-cards"></div>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<div class="flex" style="justify-content: space-between; align-items: center;">
|
||||
<h2 style="margin: 0;">Usage by Key</h2>
|
||||
<label id="usage-scope-toggle" class="flex" style="display:none; gap: 0.4rem; font-size: 0.8rem; color: #94a3b8; cursor: pointer;">
|
||||
<input type="checkbox" id="usage-show-all" style="cursor: pointer;">
|
||||
<span>Show all keys</span>
|
||||
</label>
|
||||
</div>
|
||||
<table id="key-usage-table">
|
||||
<thead><tr><th>Key</th><th>Requests</th><th>OK</th><th>Err</th><th>Avg Time</th><th>Last Request</th></tr></thead>
|
||||
<tbody></tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="section" id="key-mgmt-section" style="display:none">
|
||||
<h2>API Keys</h2>
|
||||
<div class="flex" style="margin-bottom: 0.75rem;">
|
||||
<input type="text" id="new-key-name" placeholder="Key name (e.g. wife-laptop)">
|
||||
<button class="btn btn-sm" onclick="addKey()">Create Key</button>
|
||||
</div>
|
||||
<table id="keys-table">
|
||||
<thead><tr><th>Name</th><th>Key</th><th>Created</th><th>Status</th><th></th></tr></thead>
|
||||
<tbody></tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="section">
|
||||
<h2>Recent Requests</h2>
|
||||
<table id="recent-table">
|
||||
<thead><tr><th>Time</th><th>Key</th><th>Model</th><th>Prompt</th><th>Response</th><th>Time</th><th>Status</th></tr></thead>
|
||||
<tbody></tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
const BASE = window.location.origin;
|
||||
const headers = {};
|
||||
const urlToken = new URLSearchParams(window.location.search).get("token");
|
||||
const storedToken = urlToken || localStorage.getItem("ocp_token");
|
||||
if (urlToken) { localStorage.setItem("ocp_token", urlToken); history.replaceState(null, "", "/dashboard"); }
|
||||
if (storedToken) headers["Authorization"] = `Bearer ${storedToken}`;
|
||||
|
||||
async function api(path) {
|
||||
const resp = await fetch(BASE + path, { headers });
|
||||
if (resp.status === 401 || resp.status === 403) {
|
||||
const token = prompt("Enter your OCP admin key:");
|
||||
if (token) {
|
||||
localStorage.setItem("ocp_token", token);
|
||||
headers["Authorization"] = `Bearer ${token}`;
|
||||
return api(path);
|
||||
}
|
||||
}
|
||||
return resp.json();
|
||||
}
|
||||
|
||||
async function apiPost(path, body) {
|
||||
const resp = await fetch(BASE + path, {
|
||||
method: "POST", headers: { ...headers, "Content-Type": "application/json" },
|
||||
body: JSON.stringify(body),
|
||||
});
|
||||
return resp.json();
|
||||
}
|
||||
|
||||
async function apiDelete(path) {
|
||||
const resp = await fetch(BASE + path, { method: "DELETE", headers });
|
||||
return resp.json();
|
||||
}
|
||||
|
||||
function fmtTime(ms) {
|
||||
if (!ms) return "-";
|
||||
return ms > 1000 ? (ms/1000).toFixed(1) + "s" : Math.round(ms) + "ms";
|
||||
}
|
||||
|
||||
function fmtChars(n) {
|
||||
if (!n) return "-";
|
||||
return n > 1000 ? (n/1000).toFixed(0) + "K" : String(n);
|
||||
}
|
||||
|
||||
function escapeHtml(s) {
|
||||
return String(s ?? "").replace(/[&<>"']/g, c => ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" }[c]));
|
||||
}
|
||||
|
||||
function barColor(pct) {
|
||||
if (pct >= 80) return "bar-red";
|
||||
if (pct >= 50) return "bar-amber";
|
||||
return "bar-blue";
|
||||
}
|
||||
|
||||
async function refreshStatus() {
|
||||
const data = await api("/status");
|
||||
const p = data.proxy || {};
|
||||
const r = data.requests || {};
|
||||
|
||||
document.getElementById("status-cards").innerHTML = `
|
||||
<div class="card"><div class="label">Status</div><div class="value"><span class="tag ${p.status === 'ok' ? 'tag-ok' : 'tag-err'}">${escapeHtml(p.status || '?')}</span></div><div class="sub">v${escapeHtml(p.version || '?')}</div></div>
|
||||
<div class="card"><div class="label">Uptime</div><div class="value">${escapeHtml(p.uptime || '?')}</div></div>
|
||||
<div class="card"><div class="label">Requests</div><div class="value">${r.total || 0}</div><div class="sub">${r.active || 0} active</div></div>
|
||||
<div class="card"><div class="label">Errors</div><div class="value">${r.errors || 0}</div><div class="sub">${r.timeouts || 0} timeouts</div></div>
|
||||
<div class="card"><div class="label">Sessions</div><div class="value">${p.activeSessions || 0}</div></div>
|
||||
`;
|
||||
|
||||
const plan = data.plan;
|
||||
if (plan && typeof plan === 'object' && !plan.error) {
|
||||
const s = plan.currentSession || {};
|
||||
const w = plan.weeklyLimits?.allModels || {};
|
||||
const sPct = Math.round((s.utilization || 0) * 100);
|
||||
const wPct = Math.round((w.utilization || 0) * 100);
|
||||
document.getElementById("plan-cards").innerHTML = `
|
||||
<div class="card">
|
||||
<div class="label">Session (5h)</div>
|
||||
<div class="value">${escapeHtml(s.percent || '?')}</div>
|
||||
<div class="bar-bg"><div class="bar-fill ${barColor(sPct)}" style="width:${sPct}%"></div></div>
|
||||
<div class="sub">Resets in ${escapeHtml(s.resetsIn || '?')}</div>
|
||||
</div>
|
||||
<div class="card">
|
||||
<div class="label">Weekly (7d)</div>
|
||||
<div class="value">${escapeHtml(w.percent || '?')}</div>
|
||||
<div class="bar-bg"><div class="bar-fill ${barColor(wPct)}" style="width:${wPct}%"></div></div>
|
||||
<div class="sub">Resets in ${escapeHtml(w.resetsIn || '?')}</div>
|
||||
</div>
|
||||
`;
|
||||
}
|
||||
}
|
||||
|
||||
async function refreshUsage() {
|
||||
try {
|
||||
const showAll = localStorage.getItem("ocp_usage_show_all") === "1";
|
||||
const data = await api(showAll ? "/api/usage?all=true" : "/api/usage");
|
||||
const tbody = document.querySelector("#key-usage-table tbody");
|
||||
tbody.innerHTML = (data.byKey || []).map(k => `
|
||||
<tr>
|
||||
<td>${escapeHtml(k.key_name)}</td>
|
||||
<td>${k.requests}</td>
|
||||
<td>${k.successes}</td>
|
||||
<td>${k.errors}</td>
|
||||
<td>${fmtTime(k.avg_elapsed_ms)}</td>
|
||||
<td class="mono">${escapeHtml(k.last_request || '-')}</td>
|
||||
</tr>
|
||||
`).join("") || '<tr><td colspan="6" style="color:#475569">No usage data yet</td></tr>';
|
||||
|
||||
const rtbody = document.querySelector("#recent-table tbody");
|
||||
rtbody.innerHTML = (data.recent || []).slice(0, 20).map(r => `
|
||||
<tr>
|
||||
<td class="mono">${escapeHtml(r.created_at?.slice(11, 19) || '?')}</td>
|
||||
<td>${escapeHtml(r.key_name)}</td>
|
||||
<td>${escapeHtml(r.model)}</td>
|
||||
<td>${fmtChars(r.prompt_chars)}</td>
|
||||
<td>${fmtChars(r.response_chars)}</td>
|
||||
<td>${fmtTime(r.elapsed_ms)}</td>
|
||||
<td><span class="tag ${r.success ? 'tag-ok' : 'tag-err'}">${r.success ? 'ok' : 'err'}</span></td>
|
||||
</tr>
|
||||
`).join("") || '<tr><td colspan="7" style="color:#475569">No requests yet</td></tr>';
|
||||
} catch(e) {
|
||||
console.warn("Usage API not available:", e);
|
||||
}
|
||||
}
|
||||
|
||||
async function refreshKeys() {
|
||||
try {
|
||||
const data = await api("/api/keys");
|
||||
document.getElementById("key-mgmt-section").style.display = "";
|
||||
// Admin-only "Show all keys" toggle for /api/usage scope.
|
||||
document.getElementById("usage-scope-toggle").style.display = "flex";
|
||||
const tbody = document.querySelector("#keys-table tbody");
|
||||
tbody.innerHTML = (data.keys || []).map(k => `
|
||||
<tr>
|
||||
<td>${escapeHtml(k.name)}</td>
|
||||
<td class="mono">${escapeHtml(k.keyPreview)}</td>
|
||||
<td class="mono">${escapeHtml(k.created_at)}</td>
|
||||
<td><span class="tag ${k.revoked ? 'tag-err' : 'tag-ok'}">${k.revoked ? 'revoked' : 'active'}</span></td>
|
||||
<td>${k.revoked ? '' : `<button class="btn btn-sm btn-danger" data-revoke="${escapeHtml(k.name)}">Revoke</button>`}</td>
|
||||
</tr>
|
||||
`).join("");
|
||||
tbody.querySelectorAll("button[data-revoke]").forEach(btn =>
|
||||
btn.addEventListener("click", () => revokeKeyUI(btn.getAttribute("data-revoke")))
|
||||
);
|
||||
} catch(e) { /* not admin */ }
|
||||
}
|
||||
|
||||
async function addKey() {
|
||||
const name = document.getElementById("new-key-name").value.trim();
|
||||
if (!name) return alert("Enter a key name");
|
||||
const result = await apiPost("/api/keys", { name });
|
||||
if (result.key) {
|
||||
alert("New API Key (copy it now — you won't see it again):\n\n" + result.key);
|
||||
document.getElementById("new-key-name").value = "";
|
||||
refreshKeys();
|
||||
}
|
||||
}
|
||||
|
||||
async function revokeKeyUI(name) {
|
||||
if (!confirm(`Revoke key "${name}"?`)) return;
|
||||
await apiDelete(`/api/keys/${encodeURIComponent(name)}`);
|
||||
refreshKeys();
|
||||
}
|
||||
|
||||
async function refreshAll() {
|
||||
document.getElementById("refresh-indicator").textContent = "Refreshing...";
|
||||
await Promise.all([refreshStatus(), refreshUsage(), refreshKeys()]);
|
||||
document.getElementById("refresh-indicator").textContent = `Updated ${new Date().toLocaleTimeString()}`;
|
||||
}
|
||||
|
||||
// Wire "Show all keys" toggle (visibility gated to admin via refreshKeys()).
|
||||
(function setupUsageScopeToggle() {
|
||||
const cb = document.getElementById("usage-show-all");
|
||||
cb.checked = localStorage.getItem("ocp_usage_show_all") === "1";
|
||||
cb.addEventListener("change", () => {
|
||||
localStorage.setItem("ocp_usage_show_all", cb.checked ? "1" : "0");
|
||||
refreshUsage();
|
||||
});
|
||||
})();
|
||||
|
||||
refreshAll();
|
||||
setInterval(refreshAll, 30000);
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -1,7 +0,0 @@
|
||||
services:
|
||||
claude-proxy:
|
||||
build: .
|
||||
ports:
|
||||
- "3456:3456"
|
||||
env_file: .env
|
||||
restart: unless-stopped
|
||||
@@ -0,0 +1,100 @@
|
||||
# OCP Promotion Strategy — "Stable & Visible"
|
||||
|
||||
> **This document is a recommendation for the maintainer to review and adjust, not a committed plan.**
|
||||
> It reflects the project's current posture (post-v3.21.0) and should be revisited whenever
|
||||
> the Anthropic billing / ToS environment changes significantly.
|
||||
|
||||
---
|
||||
|
||||
## 1. Goal: Polish + Low-Key OSS Visibility
|
||||
|
||||
The goal is **stability and quiet discoverability**, not growth-hacking. OCP is a personal power tool
|
||||
that has been open-sourced because others can benefit from it. The right audience finds it via GitHub
|
||||
search, issue threads in related projects, and word of mouth — not viral posts.
|
||||
|
||||
**Explicitly avoid:**
|
||||
|
||||
- HN / Reddit front-page pushes, influencer outreach, or any campaign that would attract a large
|
||||
influx of users before the ToS/billing situation has settled. Anthropic is actively tightening
|
||||
billing and enforcement on subscription-sharing (the June-15 Agent-SDK billing split is
|
||||
*paused*, not cancelled — and consumer-ToS enforcement on multi-person sharing is a live risk).
|
||||
A high-traffic spotlight right now would draw scrutiny that a low-profile project avoids.
|
||||
- Promising features that require bypassing the `claude` CLI (raw API calls, OAuth extraction, etc.)
|
||||
— that would violate `ALIGNMENT.md` and the ToS simultaneously.
|
||||
|
||||
---
|
||||
|
||||
## 2. Pre-Requisite: Stability First
|
||||
|
||||
Do not promote until the house is in order:
|
||||
|
||||
- [x] The concurrency / latency perf fixes are shipped (v3.20.x–v3.21.0).
|
||||
- [x] Docs honesty is complete (client-tools boundary, ToS sharing disclosure, this doc).
|
||||
- [ ] The June-15 Agent-SDK billing split is either confirmed cancelled or OCP has a confirmed
|
||||
stable path (TUI toggle as insurance — see §5 below).
|
||||
|
||||
Promoting a project that has known rough edges in docs or stability only generates support burden
|
||||
and negative first impressions.
|
||||
|
||||
---
|
||||
|
||||
## 3. Honest ToS Disclosure on Sharing
|
||||
|
||||
Any promotion materials must carry the same disclosure as `README.md § "Deployment model & security"`:
|
||||
|
||||
> Pooling a single Claude subscription across **multiple distinct people** may violate Anthropic's
|
||||
> Consumer Terms of Service and risk account suspension. The defensible framing is "one person,
|
||||
> your own devices". Friends/team sharing is not.
|
||||
|
||||
This framing should appear in any README badge, linked blog post, or issue comment that mentions
|
||||
LAN sharing. It is not a disclaimer that discourages usage — it is honest positioning that protects
|
||||
both the project and its users.
|
||||
|
||||
---
|
||||
|
||||
## 4. What to Explicitly Skip
|
||||
|
||||
These items are **not gaps in OCP** — they are deliberate stance decisions:
|
||||
|
||||
- **Multi-backend routing** (routing to OpenAI, Gemini, Llama, etc.) — that is the sibling [OLP
|
||||
project](https://github.com/dtzp555-max/olp)'s role. OCP stays Claude-only by design.
|
||||
- **Gateway model-discovery** (auto-detecting which models a remote server offers) — not needed
|
||||
for OCP's single-provider, single-subscription model. `models.json` is the SPOT.
|
||||
- **Raw Anthropic API passthrough** (bypassing the `claude` CLI) — out of scope per `ALIGNMENT.md`.
|
||||
|
||||
Do not add these to OCP roadmaps or respond to feature requests for them with "planned" — the
|
||||
correct answer is "that's OLP territory" or "out of scope per ALIGNMENT.md".
|
||||
|
||||
---
|
||||
|
||||
## 5. TUI Toggle as Insurance
|
||||
|
||||
The `CLAUDE_TUI_MODE` opt-in is the primary mitigation if the June-15 billing split reactivates
|
||||
and makes the default `-p` path draw from the metered Agent SDK credit pool.
|
||||
|
||||
Keep the TUI toggle:
|
||||
- Functional and tested across the three deployment hosts.
|
||||
- Documented in the README, including the security constraints (single-user only).
|
||||
- Easily discoverable for users who get unexpectedly metered.
|
||||
|
||||
If the split reactivates, the recommended operator path is: set `CLAUDE_TUI_MODE=true` +
|
||||
`CLAUDE_CODE_OAUTH_TOKEN` → credential-isolated scratch home → subscription pool. That path is
|
||||
already shipped and documented.
|
||||
|
||||
---
|
||||
|
||||
## 6. Low-Key Visibility Actions (when §2 pre-requisites are met)
|
||||
|
||||
- Keep the GitHub README polished and honest — it is the primary landing page.
|
||||
- Respond promptly to issues and PRs — the project's reputation is built on reliability, not
|
||||
marketing.
|
||||
- Add OCP to the `awesome-claude` / `awesome-llm-tools` lists if they exist and allow self-PRs
|
||||
— low-effort, targeted, reaches the right audience.
|
||||
- When related projects (Cline, OpenCode, OpenClaw, Continue.dev) post about local Claude proxies,
|
||||
a short factual comment linking to OCP is appropriate — not spam.
|
||||
- Maintain the `CHANGELOG.md` with clear, honest summaries — users who are already running OCP
|
||||
are the best vector for word-of-mouth.
|
||||
|
||||
---
|
||||
|
||||
*Last updated: v3.21.0 cleanup cycle. Maintainer should re-read before any external promotion.*
|
||||
@@ -0,0 +1,70 @@
|
||||
# 0002 — Alignment Constitution
|
||||
|
||||
- **Date**: 2026-04-20
|
||||
- **Status**: Accepted
|
||||
- **Authors**: project maintainer (with AI drafting assistance)
|
||||
- **Related**: PR #20, commit 2853088; supersedes implicit "keep the proxy honest" discipline
|
||||
|
||||
## Context
|
||||
|
||||
On 2026-04-11 an OCP commit (`b87992f`, "fix: use dedicated `/api/oauth/usage` endpoint for reliable plan data") was merged. The commit asserted that `/api/oauth/usage` was "the dedicated usage endpoint that Claude Code CLI uses." The assertion was false: the string `/api/oauth/usage` does not appear anywhere in `cli.js`. The endpoint was fabricated by an LLM-assisted authoring pass generalizing from adjacent OAuth paths, without anyone running `grep` against `cli.js`.
|
||||
|
||||
The hallucination was not an isolated slip. It persisted across nine days and two additional commits of compensation:
|
||||
|
||||
- `cb6c2a8` extended the stale cache to 15 minutes and added a fallback path on HTTP 429 — a workaround that masked the fabricated endpoint's 4xx failures rather than investigating them.
|
||||
- The dashboard `/usage` progress bar was broken for the entire window (2026-04-11 through 2026-04-20).
|
||||
|
||||
Root cause analysis identified three structural gaps:
|
||||
|
||||
1. No binding rule that OCP must mirror `cli.js` behavior exactly. "Proxy-only" was aspirational, not enforced.
|
||||
2. No CI check that would fail builds containing known-hallucinated tokens.
|
||||
3. No reviewer gate that required the reviewer to verify the `cli.js` citation before approving.
|
||||
|
||||
Without all three, the same class of drift was re-occurrence-probable rather than preventable.
|
||||
|
||||
## Decision
|
||||
|
||||
Adopt `ALIGNMENT.md` as the project constitution. It encodes five binding Rules:
|
||||
|
||||
1. **Grep First** — before changing any endpoint/header/parameter/response shape, the author must `grep` `cli.js` and record the line numbers.
|
||||
2. **No Invention** — OCP must not introduce surface area not present in `cli.js`. Speculative "Claude Code probably uses X" statements are prohibited.
|
||||
3. **Match the Implementation** — where `cli.js` does perform the operation, OCP matches it byte-for-byte on the wire.
|
||||
4. **Unalignable Features Are Deleted** — features that cannot be traced to a `cli.js` reference are removed, not deprecated, not feature-flagged.
|
||||
5. **Cite Line Numbers in Commits** — every `server.mjs`-touching commit references `cli.js:NNNN` or `cli.js vE4 <functionName>`.
|
||||
|
||||
Supporting mechanisms:
|
||||
|
||||
- `CLAUDE.md` enshrines hard requirements for `server.mjs` PRs: `cli.js` citation, CI blacklist pass, independent reviewer who opens `cli.js` at the cited lines.
|
||||
- `.github/workflows/alignment.yml` greps `server.mjs` on every PR for the known-hallucinated token set (`api/oauth/usage`, `api/usage`, et al.) and fails the build on any hit.
|
||||
- `.github/PULL_REQUEST_TEMPLATE.md` makes the `cli.js` citation and the reviewer's cli.js-opened confirmation mandatory fields.
|
||||
- A bootstrap audit pin: Claude Code `2.1.89`, `cli.js` SHA-256 `a9950ef6407fdc750bddb673852485500387e524a99d42385cb81e7d17128e01`, auditor: project maintainer, date 2026-04-20. The pin is refreshed annually on 11 April (the drift anniversary) or on any re-verification event.
|
||||
- A documented Historical Lesson section in `ALIGNMENT.md` that names the drift commits by SHA, so the incident cannot be rewritten or quietly forgotten.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive**
|
||||
|
||||
- Every `server.mjs` change is now provably aligned to a specific `cli.js` line range before merge.
|
||||
- CI hard-fails any reintroduction of the specific hallucinated tokens; the failure mode is loud and immediate, not silent-and-cached.
|
||||
- The reviewer gate makes self-approval a policy violation, which structurally prevents the "fix my own hallucination" cycle that produced `cb6c2a8`.
|
||||
- The constitution becomes the foundation that later governance (iron rule v1.4, per-project release kit, cross-device dev system) builds on.
|
||||
|
||||
**Negative**
|
||||
|
||||
- `server.mjs` changes are meaningfully slower: the `grep` step and the reviewer's cli.js verification are real costs on every PR.
|
||||
- New contributors face a steeper ramp — they must read `ALIGNMENT.md` fully before their first server-side PR will pass review.
|
||||
- The CI blacklist is a moving target; as future drift patterns are discovered, the list grows, and each addition is governance work.
|
||||
|
||||
**Follow-ons**
|
||||
|
||||
- ADR 0003 (models.json SPOT) and ADR 0004 (OpenClaw auto-sync) both lean on the constitution's "one reviewable layer" structure.
|
||||
- Annual audit on 11 April is a recurring calendar obligation; failure to perform it is itself an alignment violation.
|
||||
- The `cli.js` bundle became opaque at v2.1.90 (binary packaging). Future audits require a different verification strategy — see Alternatives (b) below.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
**(a) Pure human discipline — no CI, no template, no ADR.** the maintainer would simply commit to grepping `cli.js` on every change, and reviewers would commit to verifying. Rejected: the 2026-04-11 drift already happened under exactly this regime. the maintainer is meticulous, and the drift still shipped. Social discipline alone cannot prevent LLM hallucination from slipping through, especially when the LLM's output is superficially plausible.
|
||||
|
||||
**(b) Automatic `cli.js` diff on every PR.** A CI step that diffs `server.mjs`'s network surface against a parsed `cli.js` AST, blocking on any mismatch. Rejected as too fragile: `cli.js` v2.1.90+ ships as a minified/obfuscated binary, making AST-level grep invalid without an unofficial unbundling step. Any such pipeline becomes a maintenance burden on Anthropic's release cadence, and would routinely false-positive. The blacklist approach is lower-precision but dramatically more robust.
|
||||
|
||||
**(c) Freeze OCP and fork a new `ocp-v2` from scratch.** Start over with alignment baked in from day one. Rejected: the existing user base depends on OCP, and the drift affected one endpoint, not the architecture. Retrofitting a constitution onto the existing repo is cheaper and preserves user trust.
|
||||
@@ -0,0 +1,65 @@
|
||||
# 0003 — `models.json` as Single Source of Truth
|
||||
|
||||
- **Date**: 2026-04-20
|
||||
- **Status**: Accepted
|
||||
- **Authors**: project maintainer (with AI drafting assistance)
|
||||
- **Related**: PR #30, commit c6f7850; precursor to ADR 0004
|
||||
|
||||
## Context
|
||||
|
||||
OCP's model catalog (the mapping from short aliases like `sonnet` and `opus` to full model IDs with context-window metadata) had organically drifted into three independent locations:
|
||||
|
||||
1. `server.mjs` — `MODEL_MAP` and `MODELS` arrays, hardcoded at the top of the file. This was the runtime authority for `/v1/models` responses and alias resolution.
|
||||
2. `setup.mjs` — a separate `MODELS` constant, unchanged since the v3.0 era. Used only at first-install time to seed user config; by v3.10 it was stale and listed no Claude 4.x models at all.
|
||||
3. `~/.openclaw/openclaw.json` (on user machines) — written exactly once by `setup.mjs` during initial OCP install and never refreshed. A user who installed OCP in v3.0 and ran `ocp update` faithfully through v3.10 still had their OpenClaw config listing only three pre-Claude-4 models.
|
||||
|
||||
By the v3.10.0 release, Opus 4.7 was correctly present in location (1) and absent from (2) and (3). The symptom reaching users: native Claude Code saw the new model immediately (because it queries `/v1/models` live from server.mjs), but OpenClaw users saw nothing new, and new-installers via `setup.mjs` got an incomplete initial config. Three distinct bug reports in the two weeks following v3.10.0.
|
||||
|
||||
The drift was structural, not a bug in any one file. The files disagreed because there was no mechanism requiring them to agree.
|
||||
|
||||
## Decision
|
||||
|
||||
Extract all model metadata into `models.json` at the repo root. `server.mjs` and `setup.mjs` both read this file and derive their in-memory `MODEL_MAP`/`MODELS` structures from it. The file is committed to the repo; it is neither generated nor cached.
|
||||
|
||||
Shape (summarized):
|
||||
|
||||
- `models` — array of entries, each with `id` (full model ID), `alias` (short name), `context_window`, and flags where relevant.
|
||||
- `default_alias` — which alias resolves when the client sends an unknown or empty model.
|
||||
|
||||
Migration approach:
|
||||
|
||||
1. Hand-populate `models.json` from the v3.10.0 `server.mjs` `MODEL_MAP` values.
|
||||
2. Rewrite `server.mjs` to load and index `models.json` at startup.
|
||||
3. Rewrite `setup.mjs` to derive its `MODELS` constant from the same file.
|
||||
4. Verify byte-equivalence: the derived `MODEL_MAP` in v3.11.0 must be a byte-identical superset of the v3.10.0 hardcoded `MODEL_MAP`. This is checked by a one-shot comparison script during the refactor PR; no regression is permitted.
|
||||
|
||||
Post-refactor, the contract for adding a model is: edit `models.json`, open a PR, reviewer sanity-checks the `id` string against Anthropic's model announcement, merge. No other file changes.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive**
|
||||
|
||||
- Single edit point eliminates the "updated one place, forgot the other" failure mode structurally.
|
||||
- `setup.mjs`'s latent staleness is repaired as a side effect — new-installers now get a fresh model list.
|
||||
- Opens the door to ADR 0004 (OpenClaw auto-sync), which requires a file-based SPOT to sync from.
|
||||
- The `models.json` format is stable, Markdown-friendly JSON, easy to diff in code review.
|
||||
|
||||
**Negative**
|
||||
|
||||
- One additional file to load at server startup (negligible cost, but now a startup dependency).
|
||||
- Schema drift risk: if anyone adds a new field to `models.json` that `server.mjs` or `setup.mjs` doesn't know about, the field is silently ignored. A future schema version tag may be warranted if the format grows.
|
||||
- `models.json` parse failure is now a fatal startup error; previously, bad model config required editing source. Consider this a feature, not a regression.
|
||||
|
||||
**Follow-ons**
|
||||
|
||||
- ADR 0004 (OpenClaw auto-sync) consumes `models.json` directly in `scripts/sync-openclaw.mjs`.
|
||||
- Future additions (per-model pricing, per-model capability flags, etc.) belong in `models.json`, not scattered back across `server.mjs`.
|
||||
- The README "Available Models" table is now derived documentation and its source of truth should be pinned to `models.json` in the release_kit overlay.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
**(a) Keep the three locations, enforce sync by manual review discipline.** A reviewer checklist item: "did you update all three places?" Rejected: the drift had already demonstrated that manual discipline is insufficient when the three files are in unrelated sections of the diff. Human reviewers routinely miss the third file. The 2026-04-11 alignment drift had already taught the project that discipline-only approaches fail.
|
||||
|
||||
**(b) YAML with SOPS field-level encryption.** Some projects prefer YAML for multi-line string readability and use SOPS to encrypt sensitive fields. Rejected: OCP's model catalog contains no secrets — model IDs, aliases, and context windows are all public information published by Anthropic. YAML adds a parser dependency and SOPS adds a decryption step at startup, both for zero benefit. JSON is already native to Node, and `models.json` is easy to diff line-by-line in GitHub review UI.
|
||||
|
||||
**(c) Fetch the model list live from Anthropic at server start.** Rejected: `cli.js` does not perform this operation, so per `ALIGNMENT.md` Rule 2 it is out of scope for OCP. Additionally, a live fetch introduces a startup-time network dependency and an availability coupling to Anthropic that OCP is explicitly designed to avoid (OCP is the gateway, not another consumer).
|
||||
@@ -0,0 +1,67 @@
|
||||
# 0004 — OpenClaw Auto-Sync on `ocp update`
|
||||
|
||||
- **Date**: 2026-04-20
|
||||
- **Status**: Accepted
|
||||
- **Authors**: project maintainer (with AI drafting assistance)
|
||||
- **Related**: PR #31, commit 5ef163a; builds on ADR 0003
|
||||
|
||||
## Context
|
||||
|
||||
v3.10.0 added Claude Opus 4.7 to OCP's `server.mjs` `MODEL_MAP`. Native Claude Code users and other IDE consumers (Cline, Aider, Cursor) saw the new model immediately, because every one of those clients queries `/v1/models` live at session start.
|
||||
|
||||
OpenClaw is different. OpenClaw caches its provider/model list in `~/.openclaw/openclaw.json`, written exactly once during OCP's `setup.mjs` run, then treated as immutable until the user manually edits it. An OpenClaw user who installed OCP in, say, v3.7 and diligently ran `ocp update` through v3.10 still saw only the pre-Claude-4 model list. From their perspective, `ocp update` "did not do what it said."
|
||||
|
||||
Within two weeks of v3.10.0, three separate bug reports surfaced, all with the same root cause: OpenClaw's cache was stale. Users tried the obvious workarounds (reinstall OpenClaw, edit the JSON by hand) and reported those as additional bugs when they misformatted the file.
|
||||
|
||||
The underlying asymmetry: every other IDE integration is pull-based (asks OCP for models on demand); OpenClaw is push-based (was told once, caches forever). OCP had no mechanism for a subsequent push.
|
||||
|
||||
Additionally, ADR 0003 had just landed `models.json` as the single source of truth — meaning the data a sync mechanism would need was now available in a machine-readable file rather than scattered across `server.mjs`.
|
||||
|
||||
## Decision
|
||||
|
||||
Add `scripts/sync-openclaw.mjs`, invoked automatically at the end of `ocp update`, plus a passive drift self-check in `server.mjs` startup. Design constraints:
|
||||
|
||||
1. **Strictly scoped.** The script only touches two sub-trees of `~/.openclaw/openclaw.json`:
|
||||
- `models.providers["claude-local"].models` — the provider's model list.
|
||||
- `agents.defaults.models["claude-local/*"]` — per-agent defaults that reference claude-local models.
|
||||
All other OpenClaw config (user-defined agents, non-claude-local providers, UI preferences) is left untouched.
|
||||
|
||||
2. **Idempotent.** Running the script twice with the same `models.json` produces the same file both times — byte-identical. The script diffs before writing and no-ops if there is nothing to change.
|
||||
|
||||
3. **Safe.** Before any write, the script creates a timestamped backup at `~/.openclaw/openclaw.json.bak.<ISO8601>`. The user can always roll back.
|
||||
|
||||
4. **Non-fatal.** If `~/.openclaw/openclaw.json` is missing (OpenClaw not installed), malformed, or otherwise unwriteable, the script logs a single-line warning and exits 0. `ocp update` never fails because of sync.
|
||||
|
||||
5. **Manually invocable.** `node scripts/sync-openclaw.mjs` runs the sync as a standalone operation, for users who want to trigger it without a full `ocp update`.
|
||||
|
||||
6. **Passive drift self-check.** On server startup, `server.mjs` reads the `claude-local` model list from `openclaw.json` (if present) and compares against the models derived from `models.json`. Mismatches produce a single WARN log line — enough to alert the user without taking action. This is the "we noticed" signal; the fix is to run `ocp update`.
|
||||
|
||||
Implementation source: the sync script reads the SPOT (`models.json`), produces the canonical claude-local model list, merges it into the OpenClaw config in the two scoped locations, writes atomically (write-to-temp then rename), and logs the diff.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive**
|
||||
|
||||
- Users get new models on the next `ocp update` with no manual action. The invariant OCP's update flow was advertising is now actually true.
|
||||
- Manual invocation remains available for users who want to sync without updating OCP itself (edge case, but cheap to support).
|
||||
- Passive self-check means even users who somehow skip `ocp update` receive a runtime heads-up instead of silent drift.
|
||||
- The script is short (under 150 lines) and testable in isolation.
|
||||
|
||||
**Negative**
|
||||
|
||||
- One-time bootstrap quirk: users upgrading from v3.10 → v3.11 have a cached `cmd_update` in their existing installation that does not yet invoke the new script. The first `ocp update` to v3.11 still misses the sync; the second `ocp update` (now running v3.11's code) performs it. This is documented in README § "Troubleshooting" per the release_kit `bootstrap_quirk_policy`.
|
||||
- A new script to maintain. If OpenClaw's config schema changes, this script needs updating. The strict-scope constraint bounds the maintenance surface.
|
||||
- Non-fatal-on-error means a broken `openclaw.json` silently stays broken from OCP's perspective. Accepted trade-off: `ocp update` failing because of a sibling tool's config would be worse.
|
||||
|
||||
**Follow-ons**
|
||||
|
||||
- If OpenClaw ever adopts live `/v1/models` polling upstream, this script becomes redundant and can be deleted per ADR 0002's Rule 4 (unalignable-to-upstream features are deleted).
|
||||
- Similar sync needs for future sibling tools would follow this pattern: separate script, strictly scoped, idempotent, non-fatal, invoked by `ocp update`.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
**(a) Modify OpenClaw itself to poll `/v1/models` live.** The "correct" fix at the architecture level. Rejected: OCP is a tenant in OpenClaw's plugin model, not its owner. Opening an upstream PR creates a cross-repo coordination dependency (review timeline, release timeline, version matrix) that leaves current OCP users broken for weeks or months. The sync script is something OCP can ship unilaterally and remove later if the upstream change lands.
|
||||
|
||||
**(b) Re-run `setup.mjs` in full.** `setup.mjs` already knows how to write `openclaw.json` from scratch. Rejected: `setup.mjs` has many side effects beyond OpenClaw registration — it rewrites user shell rc files, regenerates systemd units, touches credential storage. It is explicitly not idempotent, and running it a second time on an already-configured system produces duplicate entries or regressions. The sync script's strict scope is the whole point; re-running `setup.mjs` would blow past it.
|
||||
|
||||
**(c) Do nothing — tell users to manually edit `~/.openclaw/openclaw.json`.** Rejected for two reasons. First, UX: OCP's value proposition includes "`ocp update` keeps your toolchain current," and asking users to hand-edit a third party's JSON breaks that promise. Second, error rate: the three bug reports that motivated this ADR included two malformed-JSON follow-ups from users who tried the manual approach. A machine-written file is strictly safer than a hand-edited one.
|
||||
@@ -0,0 +1,79 @@
|
||||
# 0005 — OCP Stays Single-Provider; No Multi-Provider Refactor
|
||||
|
||||
- **Date**: 2026-05-06
|
||||
- **Status**: Accepted
|
||||
- **Authors**: project maintainer (with AI advisory drafting)
|
||||
- **Related**: ADR 0002 (Alignment Constitution), ADR 0003 (`models.json` SPOT)
|
||||
|
||||
## Context
|
||||
|
||||
OCP's `server.mjs` reached 1667 lines and now provides response cache, per-key quota, session tracking, model-level stats, and SSE heartbeat — all targeting a single backend path: `spawn` the locally installed `cli.js` and let it transact with `api.anthropic.com`. This architecture is the source of OCP's only real differentiator: **`cli.js` behavior-level alignment** (session create-vs-resume semantics, tool_use id reuse, SSE quirks, etc.) — none of which a generic LLM gateway has, because none of them speak this protocol.
|
||||
|
||||
The maintainer evaluated extending OCP to support OpenAI / Gemini / OpenRouter / Together / Groq / Ollama — i.e., turning OCP into a multi-provider gateway resembling Helicone, LiteLLM, OpenRouter, or Portkey. The motivation for that extension: reduce dependency on Anthropic, broaden OCP's commercial surface, and stop being grayscale-positioned (the `cli.js` spawn pattern depends on the local Pro/Max subscription, which Anthropic could fingerprint and disable).
|
||||
|
||||
The honest engineering estimate for that extension:
|
||||
|
||||
| Phase | Net New LOC | Calendar Time (part-time) |
|
||||
|---|---|---|
|
||||
| Provider abstraction + OpenAI | ~1230 + schema migration | 2 weeks |
|
||||
| Add Gemini | ~550 | +1.5 weeks |
|
||||
| OpenAI-compatible family (OpenRouter / Together / Groq) | ~300 | +1 week |
|
||||
| Tests, docs, hardening | — | +1.5 weeks |
|
||||
| **Multi-provider v1** | ~2080 | **~7 weeks focused** |
|
||||
|
||||
That number is not the real cost. The real cost is **strategic**:
|
||||
|
||||
1. **Loss of unique value.** `cli.js` behavior alignment is meaningless for OpenAI / Gemini / Ollama traffic. Going multi-provider means OCP's only moat applies to ~30% of its surface; the other 70% is generic gateway code already done better by Helicone / LiteLLM.
|
||||
|
||||
2. **Hybrid architecture awkwardness.** A multi-provider OCP would have two paths: `spawn(cli.js)` for Claude (still grayscale, depends on Pro subscription), and direct API call for everyone else (clean, BYOK). Customers asking "what is OCP?" would hear two different answers depending on which model they pick. This is worse than either pure path.
|
||||
|
||||
3. **Direct competition with funded incumbents.** Helicone (~$5M raised, YC W23), OpenRouter (~$1B valuation), LiteLLM (significant enterprise revenue), Portkey, Langfuse, Cloudflare AI Gateway — all already do multi-provider gateway with mature dashboards, audit logs, SOC2, and team features. OCP would enter that market 2+ years late with one engineer.
|
||||
|
||||
4. **The grayscale problem isn't solved by adding providers.** As long as OCP keeps the `cli.js` spawn path for Anthropic, it remains grayscale for that path; adding OpenAI alongside doesn't make the Anthropic path any less dependent on a Pro/Max subscription that wasn't licensed for proxying.
|
||||
|
||||
The maintainer's separate decision (recorded in personal notes, not this repo) is that **OCP itself will not be commercialized**; it will remain a personal power tool plus open-source contribution. Any commercial gateway work, if pursued, will start from a clean codebase with BYOK from day one — not from OCP.
|
||||
|
||||
Given that, the multi-provider extension would buy OCP nothing: not a moat, not commercial readiness, not even meaningfully better personal utility (the maintainer overwhelmingly uses Claude).
|
||||
|
||||
## Decision
|
||||
|
||||
OCP stays single-provider. Specifically:
|
||||
|
||||
1. **No new providers added to `server.mjs`.** The dispatch path remains `spawn(cli.js) → api.anthropic.com`. Pull requests that introduce a `providers/` directory or a model-to-provider router are declined on the basis of this ADR.
|
||||
|
||||
2. **`models.json` schema stays Anthropic-only.** No `provider` field, no per-model cost/capability metadata that anticipates other providers. If non-Anthropic models ever need to be referenced (e.g., for OpenClaw provider list completeness), they live in a separate file or in OpenClaw's own config — not in OCP's SPOT.
|
||||
|
||||
3. **Cache improvements are in scope.** The existing response cache (in `keys.mjs`: `cacheHash` / `getCachedResponse` / `setCachedResponse` / `clearCache`) is acceptable to upgrade with stream replay, stampede protection (singleflight), per-key isolation, and Anthropic `cache_control` awareness. These reinforce the single-provider position; they do not create provider-extension surface area.
|
||||
|
||||
4. **Anthropic alignment work continues to be encouraged.** Anything that deepens `cli.js` behavior alignment — session lifetime, tool_use id semantics, SSE behavior, multi-account routing, model-tier observability — is the project's actual value and should be prioritized over generic-gateway features.
|
||||
|
||||
5. **Commercial work, if pursued, starts elsewhere.** A separate repository, separate name, BYOK from day one, no `cli.js` spawn. That repo is out of scope for OCP and is not bound by this ADR.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive**
|
||||
|
||||
- Project scope stays bounded. The maintainer can keep evolving OCP at part-time pace without the multi-provider maintenance burden (every provider's API breaks at some point and demands attention).
|
||||
- The unique value (`cli.js` alignment) is preserved and continues to compound — every new alignment fix increases OCP's distance from generic gateways.
|
||||
- Future contributors reading the code see one architecture, not a hybrid; debugging stays tractable.
|
||||
- Decisions about commercialization are decoupled from OCP's technical evolution. OCP can stay grayscale-personal-tool indefinitely without that being a blocker for any future commercial product.
|
||||
|
||||
**Negative**
|
||||
|
||||
- OCP cannot serve any user who needs OpenAI / Gemini / local LLM access. Those users must route through a different gateway (Helicone, LiteLLM, OpenRouter) or call providers directly.
|
||||
- If Anthropic substantially changes `cli.js` (e.g., adds client attestation, removes the spawn-and-forward pattern, or migrates `claude` to a non-CLI form factor), OCP's core architecture breaks and there is no second backend to fall back to.
|
||||
- The maintainer must resist a recurring temptation: "while I'm in here, let me just add OpenAI." The whole point of this ADR is to make that temptation cost a documented amendment, not a quiet PR.
|
||||
|
||||
**Neutral**
|
||||
|
||||
- This ADR records a non-decision in code: nothing in `server.mjs` changes today. Its purpose is to make future contributors (including the maintainer) explain themselves before going against it. Per the project's PR template and Iron Rule 11, an amendment to this ADR is the gating step before any provider-extension PR.
|
||||
|
||||
## Trigger conditions for revisiting this ADR
|
||||
|
||||
This ADR should be revisited (and possibly amended or superseded) if any of the following occur:
|
||||
|
||||
1. Anthropic ships a feature that breaks the `cli.js` spawn pattern OCP depends on, and the maintainer wants to keep OCP useful.
|
||||
2. The maintainer makes a deliberate decision to commercialize OCP (rather than start a separate codebase). This requires explicit re-scoping; "let me try" is not enough.
|
||||
3. A genuine user need emerges — e.g., the maintainer themselves starts using OpenAI / Gemini frequently from Claude Code workflows — that single-provider OCP cannot serve.
|
||||
|
||||
In all three cases, the response is **first amend this ADR**, then write code. Order is not optional.
|
||||
@@ -0,0 +1,132 @@
|
||||
# 0006 — OpenAI Shim Scope: Class A vs Class B Endpoints
|
||||
|
||||
- **Date**: 2026-05-20
|
||||
- **Status**: Proposed — owner reviewing
|
||||
- **Authors**: project maintainer (with AI drafting assistance)
|
||||
- **Related**: `ALIGNMENT.md` (the constitution); ADR 0002 (Alignment Constitution provenance, PR #20, commit 2853088); PR #99 by external contributor (triggering incident — OpenAI `response_format` honoring on `/v1/chat/completions`)
|
||||
|
||||
## Context
|
||||
|
||||
`ALIGNMENT.md` was drafted in the aftermath of the 2026-04-11 drift (commit `b87992f` — fabricated `/api/oauth/usage` endpoint) and ratified in PR #20 / commit 2853088. Its five Rules are written in the language of a one-to-one proxy: Rule 1 (Grep First), Rule 2 (No Invention), Rule 3 (Match the Implementation), Rule 4 (Unalignable Features Are Deleted), Rule 5 (Cite Line Numbers in Commits). All five anchor explicitly on `cli.js` as the golden reference. This is correct and binding for the endpoints OCP was originally designed to forward — `/v1/messages`, `/api/oauth/*`, and the rate-limit-header extraction path that backs `/usage` — because for those, `cli.js` is the literal wire authority and any deviation is a drift risk.
|
||||
|
||||
OCP also exposes a second class of endpoint that the constitution does not currently distinguish: **OpenAI-compatible surface** that exists so non-Claude-Code clients (Honcho, OpenWebUI, OpenAI SDK consumers, BYO scripts) can talk to claude via OCP. The flagship is `/v1/chat/completions`, which translates between OpenAI's request/response schema and `cli.js`'s native protocol. `cli.js` never speaks OpenAI's wire format — by construction it cannot, because OpenAI and Anthropic are different vendors with different protocols. There is no `cli.js:NNNN` to cite for OpenAI's `messages[].role` field handling, OpenAI's streaming `delta` shape, OpenAI's `stop` event names, or OpenAI's `response_format` parameter. The protocol authority for these is OpenAI's published specification, not `cli.js`.
|
||||
|
||||
The structural gap surfaced when PR #99 (external contributor `jaekwon-park`) added support for the OpenAI `response_format` request field on `/v1/chat/completions`. A strict reading of Rule 2 ("OCP must not introduce request fields that are not present in `cli.js`") blocks the PR. But the same strict reading also blocks the existence of `/v1/chat/completions` itself — every OpenAI-shaped field on that endpoint is, by definition, not in `cli.js`. The endpoint has been in OCP since before the constitution was written and is used by real downstream consumers. The constitution and the endpoint cannot both be correct under the current reading.
|
||||
|
||||
The 2026-04-11 drift remains the cautionary tale that drove the constitution and remains binding. The drift was not "OCP exposed an endpoint that wasn't in `cli.js`" — it was specifically "OCP claimed to forward `cli.js`'s `/api/oauth/usage` call when no such call exists in `cli.js`." That is a Class A failure mode: a forwarding endpoint that lied about what it was forwarding. The fix to that failure mode (Rules 1, 2, 3, 5; CI blacklist; reviewer gate) was correct then and is correct now. This ADR does not relitigate that decision and does not soften Rules 1–5 for the class of endpoint they were designed to discipline.
|
||||
|
||||
What this ADR does is acknowledge that OCP has two classes of endpoint, and that the discipline that fits Class A does not fit Class B without distortion. Class B needs its own anchor (OpenAI's specification) and its own authorization gate (an ADR per endpoint), so contributors know exactly which rule set applies to their PR and so Class B never becomes a backdoor for "OCP can do anything OpenAI-shaped."
|
||||
|
||||
## Decision
|
||||
|
||||
Introduce an explicit two-class taxonomy of OCP endpoints:
|
||||
|
||||
- **Class A — `cli.js`-mirror endpoints.** Endpoints that exist because `cli.js` performs the equivalent operation and OCP forwards, observes, or multiplexes that operation. Rules 1–5 of `ALIGNMENT.md` apply verbatim. The citation requirement is `cli.js:NNNN` (or `cli.js vE4 <functionName>`).
|
||||
|
||||
- **Class B — OCP-owned compatibility endpoints.** Endpoints that exist because OCP itself surfaces them, with no `cli.js` analogue. They fall into two sub-buckets:
|
||||
- **B.1 — OpenAI-compatibility surface.** Endpoints implementing OpenAI's published API contract so non-Anthropic clients can use OCP. The protocol authority is OpenAI's specification.
|
||||
- **B.2 — OCP-administrative surface.** Endpoints that exist purely to operate the proxy itself (health, dashboard, key management, cache control). The authority for these is the ADR that authorized the endpoint's existence.
|
||||
|
||||
For Class B endpoints, the citation requirement shifts from `cli.js:NNNN` to **(a)** the relevant specification section (OpenAI spec section for B.1, or the authorizing ADR for B.2) **and (b)** the ADR that authorized the endpoint's existence in the first place.
|
||||
|
||||
### Grandfather provision for existing B.2 inventory
|
||||
|
||||
ADR 0006 retroactively authorizes the existing B.2 endpoints listed in the inventory table below, **frozen at their current behaviour as of v3.16.4**. This is a one-time grandfather provision intended to avoid a 12-ADR back-fill burden for endpoints that have existed in OCP since before any constitutional governance was written.
|
||||
|
||||
The grandfather provision is narrowly scoped:
|
||||
|
||||
- It covers only the B.2 endpoints enumerated in the inventory table as of this ADR's merge date.
|
||||
- It freezes those endpoints at their **current behaviour**. Any change to the request shape, response shape, or semantics of a grandfathered B.2 endpoint is treated as a new authorization request and requires either (a) a behaviour-preserving refactor PR with no contract change, or (b) its own ADR.
|
||||
- It does **not** authorize new B.2 endpoints. Any new B.2 endpoint, or any new method on a grandfathered B.2 endpoint, requires its own ADR before merge.
|
||||
- It does **not** extend to B.1 (OpenAI-compat) endpoints. B.1 endpoints are bounded by OpenAI's published specification, not by a behaviour snapshot — there is no grandfather equivalent for them.
|
||||
|
||||
The structural intent is: take the one-time hit of declaring "current B.2 surface is authorized" cleanly, then make every future addition pay the ADR-per-endpoint cost. This prevents Class B from becoming a backdoor for general OCP-owned-surface invention while not blocking the present ADR on twelve back-fill PRs.
|
||||
|
||||
### Current Class B inventory (enumerated from `server.mjs`)
|
||||
|
||||
The following endpoints exist today in `server.mjs` and are Class B (no `cli.js` analogue):
|
||||
|
||||
| Endpoint | Method | Sub-bucket | Authorizing ADR |
|
||||
|---|---|---|---|
|
||||
| `/v1/chat/completions` | POST | B.1 (OpenAI-compat) | ADR 0006 |
|
||||
| `/v1/models` | GET | B.1 (OpenAI-compat) | ADR 0006; content sourced from `models.json` per ADR 0003 |
|
||||
| `/health` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/dashboard` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/sessions` | GET, DELETE | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/logs` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/status` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/settings` | GET, PATCH | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/keys` | GET, POST | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/keys/:id` | DELETE | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/keys/:id/quota` | GET, PATCH | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/api/usage` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/cache/stats` | GET | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
| `/cache` | DELETE | B.2 (administrative) | ADR 0006 (grandfathered as of v3.16.4) |
|
||||
|
||||
For Class A reference, the current Class A inventory is `/v1/messages` (forwarded directly to `api.anthropic.com/v1/messages`) and the OAuth bearer / rate-limit-header machinery used by `handleUsage()` (which calls `https://api.anthropic.com/v1/messages` to extract `anthropic-ratelimit-unified-*` headers, per the in-file comment at line 845–849). The `GET /usage` endpoint surface itself is Class B (administrative augmentation: it adds `proxy:` and `models:` blocks not present in any upstream API), but the data fetch underlying it is Class A — see "Hybrid endpoints" below.
|
||||
|
||||
### Hybrid endpoints
|
||||
|
||||
`/usage` is a hybrid: the wire call out to `api.anthropic.com/v1/messages` is Class A and must continue to cite `cli.js`; the local synthesis on top (`proxy:` stats block, `models:` snapshot, response shape) is Class B and is authorized by this ADR. Any future change strictly to the wire-call layer is Class A; any change strictly to the synthesis layer is Class B. A PR touching both must satisfy both citation requirements.
|
||||
|
||||
## What does NOT change
|
||||
|
||||
The following continue to apply verbatim and are not weakened by this ADR:
|
||||
|
||||
- **Rules 1, 2, 3, 4, 5 of `ALIGNMENT.md`** for all Class A endpoints. The 2026-04-11 drift discipline is unchanged. Class A PRs still require `cli.js:NNNN` citations, still must match `cli.js`'s wire format byte-for-byte, and still face the Unalignable Policy if the citation cannot be produced.
|
||||
- **CI blacklist** (`.github/workflows/alignment.yml`). The known-hallucinated token list (currently `api.anthropic.com/api/oauth/usage`) continues to be greppable-and-failable on every PR.
|
||||
- **Reviewer gate** (CLAUDE.md hard requirements + Iron Rule 10). Implementation author may not self-approve; a fresh-context reviewer opens `cli.js` at the cited lines for Class A PRs.
|
||||
- **Annual Alignment Audit** on 11 April. The Class A audit (re-verify each `server.mjs` Class A reference against the pinned `cli.js` SHA-256) continues unchanged.
|
||||
- **Unalignable Policy.** A Class A endpoint that cannot be traced to a `cli.js` reference is still deleted, not deprecated.
|
||||
- **Historical Lesson section in `ALIGNMENT.md`.** The 2026-04-11 drift remains the named cautionary incident, with commit SHAs intact.
|
||||
|
||||
## What additionally applies to Class B
|
||||
|
||||
The following are new and apply only to Class B endpoints:
|
||||
|
||||
1. **OpenAI specification as protocol authority (B.1).** The OpenAI compatibility surface follows OpenAI's published `/v1/chat/completions` specification (https://platform.openai.com/docs/api-reference/chat/create) — not OCP imagination, not "OpenAI probably does X," not generalization from adjacent OpenAI endpoints. The same anti-invention discipline that Rule 2 imposes for `cli.js` applies, with OpenAI's spec substituted as the reference.
|
||||
|
||||
2. **ADR-authorized endpoint existence.** Any new Class B endpoint, or any new Class B endpoint method, requires its own ADR before merge. The grandfather provision above covers existing B.2 inventory only. An "ADR-less" Class B endpoint added after this ADR merges is itself an alignment finding and is subject to deletion under a Class B equivalent of the Unalignable Policy (see Rule 4 mapping in `ALIGNMENT.md`'s new section).
|
||||
|
||||
3. **Class B citation format.** Class B PRs cite (a) the relevant specification section and (b) the authorizing ADR. Example for B.1: "OpenAI `chat/completions` API, `response_format` parameter (https://platform.openai.com/docs/api-reference/chat/create), authorized by ADR 0006." Example for B.2: "Authorized by ADR 0006 (grandfathered)" for grandfathered endpoints, or "Authorized by ADR 00NN" for endpoints with their own ADR.
|
||||
|
||||
4. **Class B audit cadence.** Class B endpoints are audited annually alongside the Class A audit. B.1 endpoints are audited against OpenAI's current `/v1/chat/completions` specification snapshot. B.2 endpoints (grandfathered or ADR-specific) are audited against their authorizing ADR — for grandfathered endpoints, the audit verifies the endpoint behaviour still matches its v3.16.4 snapshot; for ADR-specific endpoints, the audit verifies behaviour still matches the ADR. The B.1 specification pin lives in `docs/openai-compat-pin.md` (to be created alongside the first B.1 audit; not a prerequisite for this ADR to land).
|
||||
|
||||
5. **Reviewer expectation.** The fresh-context reviewer for a Class B PR opens the cited OpenAI spec section (B.1) or the authorizing ADR (B.2) instead of opening `cli.js`. The "I am not the commit author" rule and the "explicit approval comment naming the verified reference" rule continue.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive**
|
||||
|
||||
- PR #99 becomes mergeable with a one-line scope declaration ("Class B — extends `/v1/chat/completions` per ADR 0006") plus the existing alignment-evidence section adapted to Class B citation format. The structural ambiguity that blocked it is removed.
|
||||
- Future Class B contributors have a clear template and a defensible scope: "extend `/v1/chat/completions` for an OpenAI-spec field that's already in OpenAI's spec" is a well-formed PR; "add a new OCP-invented field that looks OpenAI-shaped" is not, and the same anti-invention discipline that protects Class A protects Class B.
|
||||
- The Class A surface is structurally unchanged. Reviewers reading the new `ALIGNMENT.md` see a clean Class A regime with all five Rules intact, plus an enumerated and explicitly scoped Class B carve-out.
|
||||
- The administrative endpoint surface (B.2) is no longer in a "is this even allowed under the constitution?" limbo. The grandfather provision cleanly authorizes the current inventory; new B.2 endpoints must earn their ADR.
|
||||
|
||||
**Negative**
|
||||
|
||||
- OCP now maintains a second alignment surface. OpenAI's `/v1/chat/completions` specification is also a moving target (OpenAI ships changes more than once per year, including breaking ones), so the B.1 audit has real work attached.
|
||||
- The grandfather provision freezes the current B.2 behaviour. If a grandfathered B.2 endpoint has a latent bug or undesirable behaviour, "fixing" it is a contract change and now requires an ADR (or a behaviour-preserving refactor). This is intentional friction to prevent silent contract drift.
|
||||
- Contributors must now choose Class A or Class B on every PR. Some will misclassify. The PR template's required Class A/B radio (see PR template update) and the reviewer's spec-or-cli verification step are the structural counter-measures.
|
||||
|
||||
**Mitigations**
|
||||
|
||||
- The Class B inventory is small (currently 14 endpoints) and is enumerated explicitly in `ALIGNMENT.md`. New entries require an ADR per item 2 above, so the inventory cannot grow silently.
|
||||
- Anthropic-side change frequency (which drives Class A audit cost) is structurally higher than OpenAI's `chat/completions` shape, which has been stable across multiple OpenAI API versions. The marginal B.1 audit cost is low. The grandfathered B.2 audit cost is also low — most of those endpoints have not changed in months.
|
||||
- The B.1 specification pin in `docs/openai-compat-pin.md` lets the audit anchor on a specific OpenAI spec snapshot, the same way the Class A pin anchors on a specific `cli.js` SHA-256. Drift detection then works the same way for both classes.
|
||||
|
||||
## Historical Lesson — explicit non-relitigation
|
||||
|
||||
This ADR does not relitigate the 2026-04-11 drift. The drift commit `b87992f` was Class A — it claimed `cli.js` forwarded a call that `cli.js` did not in fact make. The fix (constitution + CI blacklist + reviewer gate) was correct and remains binding for Class A. This ADR carves out Class B because the discipline that fits Class A does not fit a class of endpoint where `cli.js` is not the wire authority — not because the discipline was wrong, and not because the drift lesson is any less load-bearing.
|
||||
|
||||
A reviewer or future maintainer reading this ADR should not infer: "OCP relaxed its alignment rules." The Class A regime is structurally identical to the version that shipped in PR #20. What changed is that the constitution now names the scope of that regime precisely (the class of endpoint for which `cli.js` is the wire authority) instead of implicitly applying it to every endpoint, including ones the regime was never designed for.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
**(a) Refuse Class B as a category; close PR #99 and delete `/v1/chat/completions`.** This would resolve the structural ambiguity by enforcing Rule 2 maximally — if `cli.js` doesn't speak OpenAI's protocol, neither does OCP. Rejected: there is an existing user base on the OpenAI-compat surface, the surface is genuinely useful (it is OCP's bridge to non-Claude-Code agents), and deletion would be a load-bearing user-facing breakage in service of a doctrinal point that the constitution was never designed to make. The constitution was a response to the 2026-04-11 forwarding drift, not a charter against any OCP-owned surface.
|
||||
|
||||
**(b) Soften Rule 2 to "OCP must not introduce surface area not present in `cli.js` OR not authorized by an ADR."** This is the obvious diff and would unblock PR #99 with the smallest possible textual change. Rejected because it loses the precision that Class A needs. A combined Rule 2 means a Class A reviewer has to read the PR description twice to figure out which authority applies. The Class A/B split makes the question explicit at the PR-template level (the author picks the class) and at the reviewer level (the reviewer opens the appropriate reference). The cost of the split is one new section in `ALIGNMENT.md`; the benefit is no ambiguity in either class.
|
||||
|
||||
**(c) Move `/v1/chat/completions` and all OpenAI-compat surface out of OCP into a separate "ocp-openai-shim" repository.** This would cleanly resolve the scope question by moving Class B out of OCP entirely. Rejected as premature: the maintainer is one person, the OpenAI-compat surface today is a single endpoint plus its support, and the operational cost of two repositories (separate releases, separate CI, separate version coordination) exceeds the cost of one constitution with two named classes. If the OpenAI-compat surface ever grows to the size where a separate repo is justified, ADR 0006 is the natural pivot point — at that future date, the carve-out becomes a separation.
|
||||
|
||||
**(d) Twelve-ADR back-fill for the existing B.2 inventory before this ADR can merge.** Considered and rejected on cost grounds. Each back-fill ADR would be a short paragraph explaining what an existing endpoint does and why it's allowed; the educational value is low and the merge friction is high (12 PRs through the reviewer gate). The grandfather provision above achieves the same authorization outcome in one paragraph, while still requiring an ADR for any future B.2 endpoint. The trade-off: grandfathered endpoints are not individually documented to ADR-depth. Mitigation: the inventory table in `ALIGNMENT.md` lists every grandfathered endpoint by path and method, so the audit surface remains explicit.
|
||||
@@ -0,0 +1,393 @@
|
||||
# ADR 0007 — TUI Interactive Mode (subscription-pool bridge)
|
||||
|
||||
**Date:** 2026-05-31
|
||||
**Status:** Accepted — amended by PR-4 (entrypoint hardening), PR-B (observability + concurrency), PR-C (env-token auth + defunct-reaping), PR-D (credential-isolated home — corrects PR-C)
|
||||
**Deciders:** project maintainer
|
||||
**Authority:** claude CLI v2.1.158 interactive mode — verified live on the test host that sessions launched without `-p` / `--output-format` carry `cc_entrypoint=cli` (subscription pool), not `cc_entrypoint=sdk-cli` (Agent SDK credit pool). Mechanism verified on cli.js v2.1.104; live-confirmed on v2.1.158.
|
||||
|
||||
---
|
||||
|
||||
## Context
|
||||
|
||||
On 2026-05-14 Anthropic announced (effective 2026-06-15) a billing split that routes requests by `cc_entrypoint`:
|
||||
|
||||
| `cc_entrypoint` value | Billing pool |
|
||||
|-----------------------|-------------|
|
||||
| `cli` | Pro/Max subscription pool |
|
||||
| `sdk-cli` | Agent SDK credit pool (~$20/mo on Pro = easily exhausted) |
|
||||
|
||||
OCP's existing path (`claude --output-format stream-json -p`) sets `cc_entrypoint=sdk-cli`. After 2026-06-15 every OCP request will draw from the Agent SDK pool rather than the subscription.
|
||||
|
||||
The structural response: add an opt-in mode that drives a real **interactive** `claude` session (no `-p`, no `--output-format`), which carries `cc_entrypoint=cli` and therefore bills against the subscription. The response text is read from claude's native JSONL transcript instead of from `stdout`.
|
||||
|
||||
This is a personal-use A-path feature (single-user, single-subscription host). It is **not** a multi-tenant isolation layer.
|
||||
|
||||
### Source-verified entrypoint mechanism (PR-4 amendment)
|
||||
|
||||
Claude CLI's `main()` calls a startup function (`t$A` in the compiled bundle) that sets
|
||||
`process.env.CLAUDE_CODE_ENTRYPOINT` **only if unset** to:
|
||||
|
||||
```
|
||||
(argv has -p/--print/--init-only/--sdk-url OR !process.stdout.isTTY) ? "sdk-cli" : "cli"
|
||||
```
|
||||
|
||||
The billing header reads `cc_entrypoint = process.env.CLAUDE_CODE_ENTRYPOINT ?? "unknown"`.
|
||||
The `"unknown"` branch is dead code for any real `main()` spawn — the startup function always
|
||||
sets a value on unset env. The **real risk** is not `"unknown"`: it is a **lost TTY** (e.g. stdout
|
||||
redirected or a non-PTY spawn) silently flipping the self-classification to `"sdk-cli"` and
|
||||
drawing from the metered pool.
|
||||
|
||||
`cc_entrypoint` is one of ~6 upstream run-mode signals. The **dominant discriminator** is the
|
||||
system-prompt identity block ("official CLI" vs "Claude Agent SDK"), which is driven by genuine
|
||||
interactivity (no `-p`, no `--output-format`, real PTY) and is overridable by no env var. This
|
||||
is the real reason the tmux/no-`-p` approach works: the spawn is genuinely interactive, not just
|
||||
labelled as such.
|
||||
|
||||
---
|
||||
|
||||
## Decision
|
||||
|
||||
Add `CLAUDE_TUI_MODE=true` as an opt-in flag in `server.mjs`.
|
||||
|
||||
### How it works
|
||||
|
||||
1. Each request spawns a fresh tmux session running `claude --model <M> --session-id <UUID> --strict-mcp-config --disallowedTools 'mcp__*'` (no `-p`, no `--output-format`).
|
||||
2. The spawn result is checked immediately: if `tmux new-session` returns a non-zero exit status (or a falsy result), the request is aborted with `tui_spawn_failed: tmux session not created` **before** the boot sleep. This is the spawn/PTY gate — OCP must not issue a billing request without a verified interactive session.
|
||||
3. The serialized prompt (from `messagesToPrompt`) is pasted via `tmux send-keys … "$(cat file)"` + a separate `Enter` key event.
|
||||
4. The answer is read from claude's native JSONL transcript at `<HOME>/.claude/projects/<encoded-cwd>/<session-id>.jsonl`, polling until a `turn_duration` system event or the wall-clock cap (`CLAUDE_TUI_WALLCLOCK_MS`, default 120 s).
|
||||
5. The string answer is returned to OCP's existing downstream (singleflight → cache write-back → `completionResponse` / `streamStringAsSSE`) — **same contract as `callClaude`**.
|
||||
6. Streaming requests are buffered then replayed as chunked SSE (no real token streaming — deliberate; "don't build fragile features"). **Superseded for `stream:true` when `OCP_TUI_STREAM=1` — see the 2026-07-13 amendment below. The buffered path remains the default and is unchanged.**
|
||||
|
||||
### Billing-classifier labeling (`OCP_TUI_ENTRYPOINT`, PR-4)
|
||||
|
||||
`CLAUDE_CODE_ENTRYPOINT` on the spawn env is managed by `resolveTuiEntrypointEnv(env, mode)`
|
||||
(exported from `lib/tui/session.mjs`, pure, testable). The function **always deletes any
|
||||
inherited value first** so a stray env var from OCP's own parent process can never leak in and
|
||||
mislabel the billing header. Then:
|
||||
|
||||
| `OCP_TUI_ENTRYPOINT` | Behaviour |
|
||||
|----------------------|-----------|
|
||||
| `cli` (default) | Sets `CLAUDE_CODE_ENTRYPOINT=cli` deterministically — subscription-pool classification. **Honest only because the spawn is a genuine interactive PTY** (tmux pane, no `-p`, stdout not redirected, `new-session` verified). |
|
||||
| `auto` | Deletes the key → claude self-classifies via `t$A` (TTY → `cli`). Use to observe/diagnose the real TTY-derived value. |
|
||||
| `off` | Leaves the env exactly as inherited — diagnostics / honesty audit only. |
|
||||
|
||||
**Governing rule (verbatim):** *OCP may make a true value deterministic; it may never assert a
|
||||
value the spawn's real state contradicts. When it cannot make the claim true (e.g. cannot
|
||||
guarantee a PTY), it fails/drops the request — it does not force the signal.*
|
||||
|
||||
This is why the spawn/PTY gate (step 2 above) is load-bearing for `mode="cli"`: if `new-session`
|
||||
fails, there is no PTY, so asserting `cli` would be dishonest. Abort rather than lie.
|
||||
|
||||
OCP never suppresses the billing header (anti-fingerprinting: we do not mask the spawn).
|
||||
|
||||
### 2026-06-15 verification protocol
|
||||
|
||||
Run one quiesced canary request in TUI-mode and watch the **Agent SDK credit balance** (not the
|
||||
request header). If the balance drops, the subscription pool is unreachable via spawn. Per the
|
||||
constitution (`ALIGNMENT.md`), the response is to **drop the Anthropic provider** rather than
|
||||
escalate spoofing.
|
||||
|
||||
Version caveat: mechanism verified on cli.js v2.1.104 + live on v2.1.158. Re-verify after any
|
||||
major cli.js upgrade.
|
||||
|
||||
### Default behaviour is unchanged
|
||||
|
||||
When `CLAUDE_TUI_MODE` is unset (the default), no code path touches `callClaudeTui` or `runTuiTurn`. `upstreamCall === callClaude` and streaming uses `callClaudeStreaming` — byte-for-byte identical to the pre-TUI code path.
|
||||
|
||||
### Kill-switch
|
||||
|
||||
Unset `CLAUDE_TUI_MODE` (or set it to any value other than `"true"`) → stream-json path restored immediately on next restart.
|
||||
|
||||
### Home strategy
|
||||
|
||||
> **Superseded by the PR-D amendment below for the env-token case.** As of PR-D, `TUI_HOME`
|
||||
> is computed by `resolveTuiHome()`: when `CLAUDE_CODE_OAUTH_TOKEN` is set (and `OCP_TUI_HOME`
|
||||
> is unset) the default is a **credential-free scratch home**, not the real home. The
|
||||
> descriptions below remain accurate for the **no-env-token** case and the **explicit
|
||||
> `OCP_TUI_HOME` override** case.
|
||||
|
||||
- **Real-home (default when NO env token, `OCP_TUI_HOME` unset):** claude runs with the operator's own `~/.claude/` — shared credentials, existing onboarding, no OAuth fork risk. `ensureTuiCwdTrusted` seeds the trust record for the scratch cwd in the real `~/.claude.json` (atomic write).
|
||||
- **Scratch-home opt-in (`OCP_TUI_HOME=<path>`, no env token):** a dedicated `HOME` that symlinks `~/.claude/.credentials.json` from the real home (token is never copied) and seeds a stripped `~/.claude.json` (no project history, trusts only the scratch cwd). **Caveat:** claude rewrites `.credentials.json` on OAuth token refresh, replacing the symlink with a regular file — this forks the credentials. Use this legacy symlink mode only with a dedicated OAuth or for ephemeral testing. (The PR-D env-token mode avoids this caveat entirely — no credentials file to fork.)
|
||||
|
||||
### Working directory
|
||||
|
||||
`TUI_CWD = OCP_TUI_CWD || $HOME/.ocp-tui/work` (dedicated scratch cwd). Transcripts land under `<HOME>/.claude/projects/<encoded-cwd>/` — a stable, single location separate from the operator's real project histories. The directory is created automatically on first request.
|
||||
|
||||
### MCP hard-disable
|
||||
|
||||
`--strict-mcp-config` (no `--mcp-config` argument) prevents account-attached managed MCP servers from connecting. Belt-and-braces: `--disallowedTools 'mcp__*'` blocks any MCP tool invocation even if a server were somehow loaded. Built-in tools (Bash, Read, etc.) are left enabled on the A-path (single-user, acceptable).
|
||||
|
||||
### Session namespace
|
||||
|
||||
All tmux sessions use the prefix `ocp-tui-`. The prefix-scoped reaper (`reapStaleTuiSessions`) kills only `ocp-tui-*` sessions, never `olp-tui-*` or any other prefix. A stale-session cleanup runs once at OCP boot when `TUI_MODE` is on.
|
||||
|
||||
---
|
||||
|
||||
## SECURITY — PROMINENT WARNING
|
||||
|
||||
**TUI-mode is SINGLE-USER / SINGLE-OPERATOR ONLY.**
|
||||
|
||||
`claude` runs as the OCP process owner with full filesystem access regardless of `HOME` setting. Home selection is **not** user isolation. If OCP is serving multiple users or guest API keys:
|
||||
|
||||
- A guest prompt would run `claude` with the **operator's** filesystem access.
|
||||
- An adversarial prompt could exfiltrate files, run shell commands, or exhaust the subscription.
|
||||
|
||||
**Never enable `CLAUDE_TUI_MODE=true` on an OCP instance that serves untrusted callers or multiple users.**
|
||||
|
||||
The B-path (multi-tenant isolation) requires:
|
||||
1. `--tools ""` (no built-in tools)
|
||||
2. Per-key ephemeral `HOME` (isolated credentials + no cross-key project pollution)
|
||||
3. Sandbox runtime (e.g. `@anthropic-ai/sandbox-runtime`)
|
||||
|
||||
B-path is **deferred** and is not implemented in this ADR. Until B-path lands, TUI-mode must only be enabled on a personal single-user OCP.
|
||||
|
||||
---
|
||||
|
||||
## Observability and concurrency (PR-B amendment)
|
||||
|
||||
**Date:** 2026-06-10
|
||||
**Status:** Accepted — amends ADR 0007.
|
||||
**Motivation:** the post-PR-A code audit, findings C-4 (P1) and C-5 (P1).
|
||||
|
||||
### C-4 — independent concurrency bound for the TUI path
|
||||
|
||||
The global `MAX_CONCURRENT` gate lives in `spawnClaudeProcess()` (the `-p` / stream-json
|
||||
path). `callClaudeTui()` never calls `spawnClaudeProcess` — it calls `runTuiTurn()`, which
|
||||
cold-boots a full interactive `claude` inside a fresh tmux session. So the TUI path had **no**
|
||||
concurrency bound: N concurrent TUI requests spawned N simultaneous cold-boot tmux+claude
|
||||
processes. On a small host (e.g. a Pi 4 serving a family) a burst of ~5 is an OOM risk and
|
||||
also multiplies subscription rate-limit pressure.
|
||||
|
||||
PR-B adds an **independent** limiter for the TUI path (`lib/tui/semaphore.mjs`,
|
||||
`TuiSemaphore`):
|
||||
|
||||
- **`OCP_TUI_MAX_CONCURRENT`, default `2`.** Rationale: a TUI turn is heavy — a per-request
|
||||
cold-boot of tmux+claude plus up to `CLAUDE_TUI_WALLCLOCK_MS` (120 s) of wallclock — so a
|
||||
small host cannot run many at once. `2` is the conservative default that keeps a Pi-class
|
||||
host alive under a family burst while still allowing some overlap. It is deliberately **not**
|
||||
the same knob as `MAX_CONCURRENT` (default 8): the two pools have different shapes (a
|
||||
stream-json spawn is cheap and fast; a TUI turn is a heavy cold-boot + long wallclock), so
|
||||
coupling them would mis-size one of the two paths.
|
||||
- **Queue, don't reject.** The limiter **queues** (awaits a slot), mirroring the spirit of
|
||||
`MAX_CONCURRENT` — requests are not dropped on contention. To bound memory against a runaway
|
||||
client, the wait queue itself is capped (`maxQueue`, default 32× the limit); when the queue
|
||||
is full `run()` rejects with `tui_queue_full`, surfaced as a 503 — deterministic backpressure
|
||||
rather than silent OOM.
|
||||
- **Slot released in a `finally`.** `TuiSemaphore.run(fn)` releases the slot in a `finally`, so
|
||||
any throw — PR-A's honesty gates (`tui_wallclock_truncated`, `tui_upstream_error`), a
|
||||
`tui_paste_not_landed`, or a `tui_spawn_failed` — can never leak a slot.
|
||||
|
||||
This limiter has **zero effect when `TUI_MODE` is off**: `callClaudeTui` is never reached, so
|
||||
the semaphore is never entered. The default stream-json path is untouched.
|
||||
|
||||
### C-5 — operator-visible drift surface on `/health` (additive)
|
||||
|
||||
The `tui_entrypoint_mismatch` warning only reached journald. After the 2026-06-15 flip, a
|
||||
silent `sdk-cli` drift (the documented top risk in this ADR — a lost TTY flipping the
|
||||
self-classification to the metered Agent SDK pool) would drain metered credits **invisibly**.
|
||||
PR-B adds a `tui` block to the `/health` JSON response so an operator can poll it:
|
||||
|
||||
```
|
||||
tui: {
|
||||
enabled: <TUI_MODE>,
|
||||
entrypointMode: <OCP_TUI_ENTRYPOINT>, // cli | auto | off
|
||||
lastEntrypoint: <last observed cc_entrypoint, e.g. "cli", or null>,
|
||||
entrypointMismatches: <count of cli-expected-but-got-other turns>,
|
||||
inflight: <current concurrent TUI turns>,
|
||||
queued: <turns waiting for a slot>,
|
||||
maxConcurrent: <OCP_TUI_MAX_CONCURRENT>
|
||||
}
|
||||
```
|
||||
|
||||
`lastEntrypoint` is recorded and `entrypointMismatches` incremented inside `callClaudeTui` in
|
||||
the same mismatch branch that already emits the journald warning (via `recordTuiEntrypoint`).
|
||||
`inflight` / `queued` / `maxConcurrent` come from the C-4 semaphore. When `TUI_MODE` is off the
|
||||
block still appears with `enabled:false` (cheap, harmless) so the response shape is stable for
|
||||
consumers regardless of mode.
|
||||
|
||||
### ALIGNMENT authorization for the `/health` change
|
||||
|
||||
`/health` is a **grandfathered B.2 endpoint** under ADR 0006, frozen at its v3.16.4 behaviour.
|
||||
`ALIGNMENT.md`'s grandfather provision states: *"Any change to the contract (request shape,
|
||||
response shape, semantics) of a grandfathered B.2 endpoint is treated as a new authorization
|
||||
request and requires either a behaviour-preserving refactor PR or its own ADR."*
|
||||
|
||||
This amendment **is** that authorization. The argument:
|
||||
|
||||
- The change is **additive**: it adds one new top-level field (`tui`) containing only new
|
||||
sub-fields. **No existing `/health` field is changed, renamed, removed, or re-typed**, and no
|
||||
existing semantics change. Existing `/health` consumers (the dashboard, `ocp-connect`,
|
||||
monitoring) read the fields they already read and are unaffected — the change is
|
||||
**behaviour-preserving** for them, which is exactly the bar the grandfather provision sets for
|
||||
a non-ADR contract change.
|
||||
- The TUI observability surface is an **intrinsic part of the TUI feature** whose authorizing
|
||||
authority is **this ADR (0007)**, not a brand-new B.2 endpoint. We are not adding a new B.2
|
||||
endpoint or a new method (which would each require their own fresh ADR under the New Class B
|
||||
endpoint procedure) — we are extending the response of an existing grandfathered endpoint with
|
||||
fields that report state owned by an ADR-0007 feature. ADR 0007 is the natural home for that
|
||||
authority, and this amendment records it explicitly.
|
||||
- `cli.js` does not perform this operation — `/health` is OCP-owned (Class B), so no `cli.js`
|
||||
citation applies; the citation is this ADR + ADR 0006 (grandfathered B.2) per
|
||||
`ALIGNMENT.md`'s Class B citation requirement.
|
||||
|
||||
### `OCP_TUI_MAX_CONCURRENT` summary
|
||||
|
||||
| Env var | Default | Meaning |
|
||||
|---|---|---|
|
||||
| `OCP_TUI_MAX_CONCURRENT` | `2` | Max concurrent interactive TUI turns. Independent of `CLAUDE_MAX_CONCURRENT` (the stream-json path). Excess turns queue (bounded); a full queue yields a 503. |
|
||||
|
||||
---
|
||||
|
||||
## Authentication + defunct-reaping (PR-C amendment)
|
||||
|
||||
**Date:** 2026-06-13
|
||||
**Status:** Accepted — amends ADR 0007.
|
||||
**Motivation:** the PI231 production incident — TUI-mode returned `Please run /login · API Error: 401` for days; re-login never stuck.
|
||||
|
||||
### How the TUI `claude` authenticates
|
||||
|
||||
The spawned interactive `claude` obtains its OAuth bearer in one of two ways, in this order of preference:
|
||||
|
||||
1. **`CLAUDE_CODE_OAUTH_TOKEN` in env (PREFERRED).** If the env var is set on the OCP process, `buildTuiCmd` adds `CLAUDE_CODE_OAUTH_TOKEN=<shq-escaped token>` to the pane command's `env` prefix. claude then authenticates via this long-lived token and **never touches the credentials-refresh path**. This is the stable mode — it is exactly how the oracle and Mac-mini hosts already run (and how `server.mjs`'s own `getOAuthCredentials()` takes the same env at highest precedence). cli.js is **not** the authority here: this is a Class B, OCP-owned TUI spawn — see the Class B citation below.
|
||||
2. **`<HOME>/.claude/.credentials.json` (FALLBACK).** When the env var is unset, claude falls back to the credentials file and its short-lived access token, renewing via the single-use refresh token.
|
||||
|
||||
The token MUST be set explicitly in `buildTuiCmd` because **tmux does not forward the parent process's environment to the pane** (verified live 2026-06-01 — the same reason the whole env is delivered as an `env` prefix). A token sitting in the OCP process env is invisible to the pane unless `buildTuiCmd` re-emits it.
|
||||
|
||||
### Why the fallback path corrupts (the PI231 incident)
|
||||
|
||||
When the env token is absent, every per-request spawn drives claude through the credentials.json refresh path. OAuth refresh tokens are **single-use / rotating**: a refresh consumes the old refresh token and writes a new one. The per-request `kill-session` teardown can race / interrupt claude mid-rotation, and over many spawn+kill cycles the refresh token ended up an **empty string** — at which point renewal is impossible and the host returns a permanent 401. Re-login writes a fresh token, but the next spawn re-corrupts it. **Proof the env-token fix works:** on the broken PI231 host, `CLAUDE_CODE_OAUTH_TOKEN=<oat01 token> claude -p ...` returned a real answer *despite* the corrupt credentials.json (control without the env token = 401).
|
||||
|
||||
**Operator guidance:** set `CLAUDE_CODE_OAUTH_TOKEN` on any TUI-mode host. The credentials.json fallback is retained only for hosts that intentionally rely on it; it is not recommended for a long-running TUI deployment.
|
||||
|
||||
**Security note:** with the token in the pane command, it is visible in `ps`. This is acceptable for the **single-user A-path** (it mirrors the existing plaintext-token practice for `server.mjs`), and the **multi-user B-path is already refused at boot** (`CLAUDE_TUI_MODE=true` + `AUTH_MODE=multi` is a hard FATAL), so a guest can never reach this spawn.
|
||||
|
||||
### Defunct `<claude>` reaping
|
||||
|
||||
The connected leak: the pane's `claude` process is a child of the long-lived **tmux server** daemon, not of the OCP node process (`tmux new-session -d` returns the instant the server forks the pane). Node can therefore never `waitpid()`/reap it — a SIGKILL still needs the *parent* (the tmux server) to reap. `kill-session` destroys the session but leaves the pane's `claude` (and its grandchildren) as `<defunct>` zombies that only the server reaps; over 30 days on PI231 this accumulated to **25 defunct `<claude>`** (a live `tmux kill-server` dropped it 25→3).
|
||||
|
||||
The node-reachable action that *actually reaps* — rather than merely re-signalling — is to stop the tmux server: on server exit the kernel reparents survivors to init (PID 1), which reaps them. `reapStaleTuiSessions` therefore, after killing our own `ocp-tui-*` sessions, issues `kill-server` **only when no foreign session of any prefix remains** (coexistence: never disrupt a co-hosted `olp-tui-*` instance). This runs at boot (existing) and now on a 15-min periodic interval gated on TUI-mode and on the TUI path being idle (`inflight === 0 && queued === 0`) so a live turn's pane is never torn down. Residual: a request whose pane is created in the narrow window between the idle-check and `kill-server` would fail cleanly via the existing honesty gates (rare; documented in the server comment).
|
||||
|
||||
### ALIGNMENT authorization (Class B)
|
||||
|
||||
Both changes are **Class B** (OCP-owned TUI spawn). `cli.js` does not perform either operation — there is no `cli.js` analogue for "how the TUI pane authenticates" or "reaping tmux-server-owned zombies"; this surface is authorized by **this ADR (0007)** per `ALIGNMENT.md`'s Class B citation requirement. No Class A wire surface, no endpoint shape, no `alignment.yml` blacklist token, and no `models.json` entry is touched.
|
||||
|
||||
---
|
||||
|
||||
## Credential-isolated home for env-token auth (PR-D amendment)
|
||||
|
||||
**Date:** 2026-06-13
|
||||
**Status:** Accepted — amends ADR 0007. **Corrects** the PR-C rationale and the original "Home strategy" section's scratch-home caveat.
|
||||
**Motivation:** PR-C's env-token passing alone did **not** fix the PI231 401. Decisive live evidence (claude 2.1.104, PI231):
|
||||
|
||||
| Condition | Result |
|
||||
|---|---|
|
||||
| env token passed + a broken `~/.claude/.credentials.json` present | **401** (`Please run /login · API Error: 401`) |
|
||||
| env token passed + `credentials.json` moved aside | **works** (real answer) |
|
||||
|
||||
### Corrected root cause
|
||||
|
||||
**Interactive `claude` PREFERS `~/.claude/.credentials.json` over the `CLAUDE_CODE_OAUTH_TOKEN` env var.** A stale/corrupt `credentials.json` therefore **shadows** the env token. (This is *unlike* `-p` mode, where the env token wins — which is why `server.mjs`'s own `getOAuthCredentials()` is unaffected and why PR-C's premise looked sufficient.) So passing the token (PR-C, `buildTuiCmd`) is **necessary but insufficient**: the TUI `claude` must additionally run in a HOME that has **no `credentials.json`**, so the env token is the only credential and is authoritative.
|
||||
|
||||
This also fixes the original incident at the **root**, more completely than PR-C claimed: with no `credentials.json` in the home, claude never runs the token-refresh path at all, so the single-use refresh token can never be rotated — and therefore never corrupted — by the spawn+`kill-session` cycle. The 25-zombie / empty-refresh-token failure mode becomes structurally impossible, not merely avoided.
|
||||
|
||||
### Decision
|
||||
|
||||
When `CLAUDE_CODE_OAUTH_TOKEN` is set, the TUI `claude` runs in a **credential-free scratch home** by default:
|
||||
|
||||
- `resolveTuiHome({ realHome, configuredHome, envTokenSet })` (exported from `lib/tui/session.mjs`, pure) decides the home:
|
||||
- **`OCP_TUI_HOME` set** → that path (explicit override, back-compat — an operator who configured it keeps exactly that home).
|
||||
- **else env token set** → `<realHome>/.ocp-tui/home` — a dedicated scratch home seeded with a minimal `.claude.json` (`hasCompletedOnboarding=true` + trust **only** the scratch cwd) and its own `projects/` dir, and **deliberately NO `.credentials.json`** (no symlink, no copy).
|
||||
- **else (no env token)** → the operator's real home — **byte-for-byte the pre-fix behaviour** for hosts that intentionally rely on `credentials.json`.
|
||||
- `prepareTuiHome(realHome, tuiHome, cwd, { envTokenMode })` gates the credential handling: in `envTokenMode` it creates the scratch `projects/` dir and seeds the minimal trusted `.claude.json` but **never** creates the credentials symlink. `runTuiTurn` sets `envTokenMode = !!CLAUDE_CODE_OAUTH_TOKEN && ehome !== realHome`.
|
||||
- `readTuiTranscript` reads from the **same** home claude runs under (`ehome`), so transcripts land under `<scratch home>/.claude/projects/` and `findTranscriptPath` globs them there — the home is threaded through consistently. (We chose scratch-`HOME` over `CLAUDE_CONFIG_DIR`: the binary supports `CLAUDE_CONFIG_DIR`, but it relocates the transcript root to `<CONFIG_DIR>/projects/` rather than `<HOME>/.claude/projects/`, which would fork the transcript-resolution rule across modes for no benefit. The scratch-HOME lever reuses the existing, tested `prepareTuiHome`/`ehome` plumbing.)
|
||||
|
||||
### This RESOLVES — not reintroduces — the scratch-home caveat
|
||||
|
||||
The original "Home strategy" section and PR-C's `prepareTuiHome` comment warned that scratch-home is unsafe because *claude rewrites a **symlinked** `.credentials.json` on token refresh → forks/corrupts the OAuth credentials*. **That caveat does not apply to env-token mode**: there is no `credentials.json` in the home to fork, and claude never refreshes (it uses the long-lived env token), so there is no rotation and no corruption. The fork risk was inherent to the *symlink* approach; removing the credentials file entirely removes the risk. The legacy symlink mode is retained **only** for an operator who explicitly sets `OCP_TUI_HOME` without an env token, and its caveat is preserved for exactly that path.
|
||||
|
||||
### ALIGNMENT authorization (Class B)
|
||||
|
||||
**Class B** (OCP-owned TUI spawn). `cli.js` has no analogue for the TUI pane's auth/home strategy; authorized by **this ADR (0007)** per `ALIGNMENT.md`'s Class B citation requirement. `server.mjs` is touched only to compute `TUI_HOME` via `resolveTuiHome()` (TUI wiring) and to surface the auth mode in the boot log — no Class A wire surface, no endpoint shape, no `alignment.yml` blacklist token, and no `models.json` entry is touched.
|
||||
|
||||
---
|
||||
|
||||
## Consequences
|
||||
|
||||
### Positive
|
||||
|
||||
- After 2026-06-15, requests in TUI-mode bill against the Pro/Max subscription pool (`cc_entrypoint=cli`) rather than the Agent SDK credit pool.
|
||||
- Kill-switch is immediate (unset env var + restart); zero code change required.
|
||||
- Default stream-json path is untouched — no regression risk for existing deployments.
|
||||
|
||||
### Negative / trade-offs
|
||||
|
||||
- **No token streaming:** responses are buffered then replayed as chunked SSE. Clients see a delay then the full response arrives; real-time token streaming is not available in TUI-mode.
|
||||
- **Billing unmeasurable until 2026-06-15:** the `cc_entrypoint=cli` signal is verified, but the credit deduction from the correct pool cannot be confirmed until the billing split activates.
|
||||
- **tmux dependency:** the host must have `tmux` installed. CI / Docker images that lack tmux cannot use TUI-mode (the default stream-json path is unaffected).
|
||||
- **Wall-clock cap:** long Opus thinking turns may hit the 120 s cap. Increase `CLAUDE_TUI_WALLCLOCK_MS` if needed (no quiescence heuristic — the reader polls until terminal marker or cap).
|
||||
- **Grey-area usage:** running an interactive `claude` session headlessly to serve HTTP requests is not an officially documented use case. If Anthropic policy changes to block this pattern, OCP must fall back to the stream-json path (unset `CLAUDE_TUI_MODE`).
|
||||
|
||||
### Coexistence
|
||||
|
||||
- tmux prefix `ocp-tui-` is registered. Any co-hosted OLP test instance must use `olp-tui-`. Never run two TUI proxies on the same OAuth concurrently — stop one instance during integration testing.
|
||||
|
||||
---
|
||||
|
||||
## Amendment (2026-07-13) — real SSE streaming via the `MessageDisplay` hook (`OCP_TUI_STREAM`)
|
||||
|
||||
**Supersedes**: Request-flow step 6 above ("no real token streaming — deliberate"), for `stream:true`
|
||||
requests when `OCP_TUI_STREAM=1`. The buffered path stays the default and is byte-for-byte unchanged.
|
||||
|
||||
**Context.** Step 6 was written when the interactive CLI appeared to expose no byte-faithful
|
||||
incremental source. A prereq spike (`docs/plans/2026-07-13-tui-latency/streaming-spike.md`) confirmed
|
||||
three obvious sources are dead ends — the transcript JSONL grows one *whole event* at a time (the
|
||||
answer lands as a single line ~0.3 s before the terminal marker); `tmux capture-pane` yields a
|
||||
*rendered* view whose markdown source is unrecoverable (an H2 and a bold span produce identical ANSI);
|
||||
`--debug-file` logs stream *timing*, never stream *content*. Every interface that does emit
|
||||
`text_delta` (`--output-format stream-json`) requires `-p`, which moves the request to the **metered**
|
||||
`sdk-cli` pool — precisely what TUI-mode exists to avoid.
|
||||
|
||||
**Decision.** Consume `claude`'s own **`MessageDisplay`** hook, registered via `--settings` on the
|
||||
ordinary interactive spawn (no `-p`, no `--bare`). Each fire delivers the **raw markdown source** of an
|
||||
incremental `delta` on the hook's stdin. Verified live (claude 2.1.207, sonnet-4-6): banner stays
|
||||
`· Claude Max` and the transcript `entrypoint` stays `cli` (subscription pool); `concat(deltas) === T`
|
||||
byte-exactly; `T.startsWith(concat(deltas[0..n]))` at every *n*. This is **forwarding, not inventing**
|
||||
— ALIGNMENT.md **Class B**. No `cli.js` citation applies: the TUI spawn is OCP-owned surface (this
|
||||
ADR), the hook payload is claude's own published contract, and the SSE wire shapes are the OpenAI
|
||||
chat/completions streaming spec adopted by **ADR 0006** (the emitters are literally the `-p` path's).
|
||||
|
||||
**The transcript remains authoritative.** It is still the terminal-turn signal, still the source of the
|
||||
returned/cached text `T`, and still the input to the honesty gates (auth-banner detection C-1,
|
||||
`truncated` C-2). The delta stream is a low-latency **mirror**, never a replacement. At end of turn OCP
|
||||
asserts the streamed bytes against `T`: equal → serve; a strict *prefix* of `T` → top up from the
|
||||
transcript (client still receives exactly `T`); **not** a prefix → **refuse the turn** (SSE error frame,
|
||||
no cache, `tui.streamDivergences++`). Serving text the transcript disagrees with is the failure class
|
||||
ALIGNMENT.md exists to prevent, so streaming fails loud rather than degrading quietly.
|
||||
|
||||
**Consequences / constraints recorded for future authors:**
|
||||
|
||||
- **Opt-in, default OFF.** The buffered path is stable production; streaming does not change it.
|
||||
- **Per-`session_id` sink is mandatory, not an optimization.** `OCP_TUI_MAX_CONCURRENT` defaults to
|
||||
**2** — two `claude` panes already run concurrently. A single shared sink would interleave one
|
||||
client's deltas into another's stream. The hook writes to `<dir>/<session_id>.jsonl`, the path
|
||||
delivered through the *pane's own env* (`OCP_TUI_STREAM_FILE`); OCP reads only its own turn's file.
|
||||
Verified with two concurrent streamed turns (ALPHA/BRAVO): zero cross-contamination.
|
||||
- **Warm-pool compatible (a separate in-flight PR depends on this).** The hook script and the settings
|
||||
file are **static** — nothing request-specific is baked in at spawn time. The sink path derives from
|
||||
the session-id, which for a pre-booted pane is fixed at boot.
|
||||
- **The hook is synchronous** (`forceSyncExecution: true` — `claude` *blocks* on it). The hook script
|
||||
must write and exit; it does one `cat` append and nothing else. Measured: p50 **7.2 ms** per fire,
|
||||
~50 ms across a whole turn — noise against a 6–10 s turn. Do not add work to it.
|
||||
- **Thinking blocks do not fire the hook** — verified on a substantive Opus/`xhigh` reasoning turn (see
|
||||
the PR evidence), not merely inferred from the `final:true` call site. This must be **re-verified** if
|
||||
the hook is ever pointed at a new model/effort tier: a thinking delta reaching a client would be
|
||||
unretractable, and the `concat === T` assertion can only *detect* that after the fact, never prevent
|
||||
it. The first-bytes **holdback** (`OCP_TUI_STREAM_HOLDBACK`, default 100 chars) is the same
|
||||
prevention-not-detection reasoning applied to the auth-banner gate.
|
||||
- **Block-level granularity**, scaling with answer length — not token-level. Do not promise otherwise.
|
||||
- **It moves the first byte, not the last.** Only a progressively-rendering consumer benefits; it does
|
||||
not move TUI-mode's ~6 s TTFT floor.
|
||||
|
||||
## Provenance
|
||||
|
||||
TUI-mode originated in a prototype contributed via PR #101 (see the PR for author attribution). The productionization design is in `docs/superpowers/specs/2026-05-30-tui-mode-production-design.md`. Spikes S1–S6 / T1–T6 were validated live on the test host against `claude v2.1.158`.
|
||||
@@ -0,0 +1,208 @@
|
||||
# ADR 0008 — TUI Warm Pane Pool
|
||||
|
||||
**Date:** 2026-07-13
|
||||
**Status:** Proposed
|
||||
**Extends:** [ADR 0007](0007-tui-interactive-mode.md) (TUI interactive mode). This ADR does not
|
||||
change ADR 0007's billing-pool argument, security posture, or kill-switch — it adds a latency
|
||||
optimization *inside* the TUI spawn machinery ADR 0007 owns.
|
||||
|
||||
---
|
||||
|
||||
## Context
|
||||
|
||||
TUI mode (ADR 0007) serves every request by cold-booting a fresh `tmux` session running an
|
||||
interactive `claude`, submitting one prompt, reading the native transcript, and killing the
|
||||
session. That cold boot is paid on **every** request.
|
||||
|
||||
[`docs/plans/2026-07-13-tui-latency/`](../plans/2026-07-13-tui-latency/README.md) measured the
|
||||
TUI path and listed a warm pane pool as backlog item #3, costed at "**~1.0 s**" (the observed
|
||||
boot-to-input-bar time). Instrumenting the real request path showed that estimate is **~4×
|
||||
too low**. Phase decomposition of the cold path (n=6 medians, Sonnet 4.6, `--effort low`,
|
||||
through a real OCP instance):
|
||||
|
||||
| Phase | Median |
|
||||
|---|---|
|
||||
| prep (trust cwd, write prompt file) | 2 ms |
|
||||
| `tmux new-session` | 27 ms |
|
||||
| **boot → input bar ready** | **1232 ms** |
|
||||
| paste (`load-buffer` + `paste-buffer`) | 8 ms |
|
||||
| paste-verify poll | 426 ms |
|
||||
| **submit → transcript terminal** | **8458 ms** |
|
||||
| teardown | 8 ms |
|
||||
| **total** | **10162 ms** |
|
||||
| *claude's own reported `turn_duration`* | *5539 ms* |
|
||||
| **OCP-side overhead** | **4490 ms** |
|
||||
|
||||
The `submit → terminal` phase exceeds claude's own `turn_duration` by **~2.9 s**. That gap is
|
||||
**post-input-bar initialization inside `claude`** — work that a pane which has merely *sat idle
|
||||
for a few seconds* has already completed. A direct spike confirmed it: an identical pane, idle
|
||||
12 s before receiving the same prompt, completed its turn in a median 5537 ms versus 7980 ms
|
||||
cold.
|
||||
|
||||
So a warm pane recovers **~1.26 s of boot *and* ~2.9 s of in-`claude` cold start** — not the
|
||||
~1.0 s the plan predicted.
|
||||
|
||||
The reason this was worth a pool rather than a "keep one session and reuse it" cache is a
|
||||
hazard already flagged in the code. `lib/tui/transcript.mjs` returns the **last text-bearing
|
||||
assistant entry in the whole transcript file**, which is correct *only* under OCP's
|
||||
one-session-per-request model, and it says so:
|
||||
|
||||
> *"If a future warm-pool ever reuses a session WITHOUT a fresh session-id / clear, earlier-turn
|
||||
> text could leak — that author must add user-line scoping here."*
|
||||
|
||||
Reusing a pane for a second turn puts two exchanges in one transcript and would leak the earlier
|
||||
turn's text into the later turn's answer — a **cross-request data leak**, not merely a bug.
|
||||
|
||||
---
|
||||
|
||||
## Decision
|
||||
|
||||
Add an **opt-in pool of pre-booted, single-use `claude` panes**, `OCP_TUI_POOL_SIZE` (default
|
||||
`0` = off, max `4`). Implementation: `lib/tui/pool.mjs`.
|
||||
|
||||
### 1. Panes are SINGLE-USE. This is the load-bearing rule.
|
||||
|
||||
A pooled pane serves **exactly one turn**, then is killed and replaced in the background. Each
|
||||
pane is booted with its **own fresh `--session-id`**, fixed at spawn, and the turn locates its
|
||||
transcript by that id.
|
||||
|
||||
This preserves one-session-per-request exactly, so the `transcript.mjs` hazard above **does not
|
||||
arise** and no user-line scoping was needed. The warning in `transcript.mjs` is deliberately
|
||||
left standing, now annotated: it still binds anyone who later wants a pane to serve a second
|
||||
turn, or to reset a session with `/clear` and reuse it. **Neither is permitted without first
|
||||
adding user-line scoping to the transcript reader.**
|
||||
|
||||
Rejected alternative — *reuse a pane for N turns, `/clear` between* — is strictly cheaper
|
||||
(no re-boot per request) and was rejected on exactly this basis. The latency win is not worth a
|
||||
cross-request text-leak surface guarded only by a `/clear` that we cannot verify landed.
|
||||
|
||||
### 2. The pool is keyed by model, and a MISS is always safe.
|
||||
|
||||
`--model` is fixed at spawn, so a pane can only serve the model it booted with. A pool miss
|
||||
falls back to the existing cold-boot path with **zero behavioural difference**. There is no
|
||||
boot-time pre-warm and no configured model: OCP cannot know which model the next caller wants,
|
||||
so the pool warms the **most recently requested** model. Consequence, stated plainly: **the
|
||||
first request after start, and the first after any model switch, is always a cold miss.**
|
||||
|
||||
### 3. The pool and the session reaper coexist by an explicit invariant.
|
||||
|
||||
This is the subtle part. `reapStaleTuiSessions()` kills every session matching this instance's
|
||||
`ocp-tui-<port>-` prefix, and issues `tmux kill-server` when no foreign session remains (the
|
||||
only mechanism that can reap `<defunct>` `claude` zombies — the pane's `claude` is a child of
|
||||
the tmux *server*, not of node). A warm pooled pane **is** one of our own sessions, alive and
|
||||
idle **by design** — and the periodic sweep runs precisely **when the instance is idle**, i.e.
|
||||
exactly when the pool is full.
|
||||
|
||||
The invariant, stated in a comment above `reapStaleTuiSessions` and pinned by tests:
|
||||
|
||||
1. **A live pooled pane is never reaped — including one that is still BOOTING.** The reaper
|
||||
takes a `spare` set of **exact session names** supplied by the pool's live registry.
|
||||
2. **An orphaned pooled pane IS still reaped.** Membership is by **exact name from a live
|
||||
in-memory registry, never by name shape**. A pane the pool no longer owns — handed out,
|
||||
dropped, cancelled, or left behind by a previous process generation (whose registry died with
|
||||
it) — is absent from `spare` and is killed like any other stale session. **Fail-safe:
|
||||
omitting `spare` reaps *more*, never less.** Pool panes are named `ocp-tui-<port>-p<hex>`
|
||||
purely for operator legibility; that shape is *not* the exemption mechanism.
|
||||
3. **`kill-server` is suppressed while any pane is spared** (it would kill a live child of the
|
||||
tmux server). Therefore **the pool is DRAINED immediately before every sweep**, so `spare` is
|
||||
empty on the normal tick and `kill-server` still fires. Without the drain, a permanently-full
|
||||
pool would **permanently disable zombie reaping** — the pool would silently break the thing
|
||||
the sweep exists to do. The drain costs one pane re-boot per tick (15 min).
|
||||
|
||||
The `spare` mechanism is belt-and-braces given the drain: it makes it impossible for a reap call
|
||||
site that *forgets* to drain to kill a live pane.
|
||||
|
||||
### 4. The pool tracks its in-flight boot BY NAME, not as a count.
|
||||
|
||||
`bootTuiPane` creates the tmux session **synchronously** and only *then* waits (up to
|
||||
`POOL_BOOT_MS`, 20 s) for the input bar. So **a pooled tmux session can be live for ~20 s before
|
||||
its boot resolves.** A pool that tracked in-flight boots as a *count* could not name that
|
||||
session, and this produced two real bugs (both caught in review, both now regression-tested):
|
||||
|
||||
- the periodic sweep **killed the booting pane** (it could not be spared), then left the pool
|
||||
empty with nothing scheduled, and logged the exact `tui_pool_boot_failed` warning operators are
|
||||
told to alert on — for a completely healthy drain;
|
||||
- graceful shutdown **orphaned a live, authenticated, idle `claude`**: `gracefulShutdown` calls
|
||||
`process.exit(0)` in the same tick as the drain (TUI panes are tmux children, so node's
|
||||
`activeProcesses` set is empty and the "wait for children" path exits immediately), so any
|
||||
cleanup deferred to a `.then()` never ran.
|
||||
|
||||
The pool therefore **mints each pane's identity up front** (`{sessionId, name}`, name derived
|
||||
from the session-id so `tmux ls` correlates to the transcript file) and holds it in
|
||||
`_bootingPane`. `liveNames()` includes it; `drain()` kills it **synchronously**. A generation
|
||||
counter distinguishes *"cancelled by us"* from *"genuinely failed"*, so a drain never inflates
|
||||
`bootFailures` and `resume()` reliably starts a fresh boot.
|
||||
|
||||
### 5. Refills take no concurrency slot, and are serialized.
|
||||
|
||||
A refill boot deliberately does **not** take a `TuiSemaphore` slot: those slots bound concurrent
|
||||
*turns* and belong to real requests, and charging a background pre-boot against them would let
|
||||
the pool starve the traffic it exists to speed up. It cannot leak a slot either, since it never
|
||||
holds one. Boots are **serialized** (one at a time): two cold boots racing an in-flight turn were
|
||||
observed to overrun even the generous pool readiness cap. A genuinely failed boot does **not**
|
||||
re-kick the chain (backoff — a broken `claude` must not respawn forever).
|
||||
|
||||
Background boots get a more generous readiness cap (`POOL_BOOT_MS` = 5 × `BOOT_MS`): `BOOT_MS` is
|
||||
tight because a *client* is blocked on it, which is not true of a pre-boot. Slow ≠ broken.
|
||||
|
||||
---
|
||||
|
||||
## Consequences
|
||||
|
||||
### Cost — standing processes, paid whether or not a request arrives
|
||||
|
||||
**A warm pane is a live idle `claude` process.** Peak process count is
|
||||
`OCP_TUI_POOL_SIZE` + `OCP_TUI_MAX_CONCURRENT` + 1 (booting replacement). This is the whole
|
||||
reason the pool is **default-off**: an operator must opt into holding processes for traffic that
|
||||
may never come. Size is clamped to `POOL_MAX_SIZE` = 4; an unparseable value **disables** the
|
||||
pool rather than guessing.
|
||||
|
||||
Panes carry a 10-minute TTL and are health-checked at hand-out; a dead or degraded pane becomes
|
||||
a **miss** (cold path), never a hung turn.
|
||||
|
||||
### Benefit
|
||||
|
||||
Measured end-to-end through a real OCP instance (Sonnet 4.6, `--effort low`):
|
||||
**p50 10.17 s (n=6, pool off) → 6.00 s (n=12 warm hits) — −4.2 s / −41%.**
|
||||
|
||||
### The floor is unchanged
|
||||
|
||||
The pool does not touch the **~6 s TTFT floor** documented in the latency plan (claude always
|
||||
prefills the full Claude Code system prompt). TUI mode remains unsuitable for interactive /
|
||||
real-time consumers; it is for batch and background work. This ADR does not change that
|
||||
conclusion.
|
||||
|
||||
### Observability
|
||||
|
||||
`/health`'s `tui` block gains a `pool` sub-object (`null` when off): `size`, `warm`, `booting`,
|
||||
`model`, `hits`, `misses`, `boots`, `bootFailures`, `cancelled`, `dropped`. A climbing
|
||||
`bootFailures` means panes are not reaching their input bar — the pool then degrades safely to
|
||||
the cold path, but latency reverts to the un-pooled numbers. A steadily climbing `dropped` is
|
||||
**normal** (the 15-min sweep drains and re-boots the pool on every tick, by design — see
|
||||
Decision 3).
|
||||
|
||||
### ALIGNMENT authorization
|
||||
|
||||
- **Class B / OCP-owned.** The warm pool is process management around the `claude` CLI — the
|
||||
same category as the existing tmux session lifecycle and the defunct-session reaper it extends.
|
||||
**`cli.js` does not perform this operation, and no `cli.js` citation applies**; the authority
|
||||
is ADR 0007 (which owns the TUI spawn machinery) plus this ADR. This is `ALIGNMENT.md` Rule 2's
|
||||
Class B citation requirement, discharged explicitly rather than by silence.
|
||||
- **The `/health` extension** adds sub-fields to the `tui` block. That block is **owned by ADR
|
||||
0007** and post-dates ADR 0006's v3.16.4 grandfather snapshot, so it is not part of the frozen
|
||||
B.2 inventory. The change is additive — every pre-existing `/health` field keeps a
|
||||
byte-identical value, and `pool` is `null` unless the operator opts in — which is the
|
||||
behaviour-preserving bar ADR 0006 sets. This ADR records that authorization.
|
||||
- **No spawn argument changed.** `buildTuiCmd` is byte-identical; the pool calls it with the same
|
||||
arguments. Banner-verified on live pooled panes: `· Claude Max`, never `API Usage Billing`
|
||||
(the `--bare` trap documented in the latency plan).
|
||||
|
||||
### What a future contributor must not undo
|
||||
|
||||
- **Do not let a pane serve a second turn** (or `/clear`-and-reuse one) without first adding
|
||||
user-line scoping to `lib/tui/transcript.mjs`. That is a cross-request text leak, not a perf
|
||||
tweak. See Decision 1.
|
||||
- **Do not remove the drain-before-sweep.** It is what keeps `kill-server` zombie reaping alive.
|
||||
See Decision 3.
|
||||
- **Do not go back to counting in-flight boots.** The pool must be able to *name* a session that
|
||||
exists but has not finished booting. See Decision 4.
|
||||
@@ -0,0 +1,54 @@
|
||||
# ADR 0009 — Prompt-char budget derives from the models.json SPOT
|
||||
|
||||
Date: 2026-07-18
|
||||
Status: Accepted (maintainer directive, 2026-07-18: "37.5k 截断未免太短了吧 … 这个在现在还适用吗")
|
||||
|
||||
## Context
|
||||
|
||||
`MAX_PROMPT_CHARS` (the tail-first truncation guard in `messagesToPrompt`) defaulted to a
|
||||
hand-set constant of 150,000 chars ≈ 37.5k English tokens — set in the 200k-window era as a
|
||||
runaway-context guard. Meanwhile `models.json` advertises `contextWindow: 200000` for every
|
||||
model (and the underlying CLI registry carries 1M native windows for Opus 4.8 / Sonnet 5), and
|
||||
`scripts/sync-openclaw.mjs` feeds that 200k into OpenClaw's compaction budget. The result was
|
||||
a standing dishonesty identified in the PR #152 review: **no advertised contextWindow value was
|
||||
true**, because the proxy silently guillotined every request at ~37.5k tokens — roughly 5×
|
||||
below the advertised window — logging only a server-side warning the client never sees.
|
||||
|
||||
Raising the constant to another hand-set number would rot the same way. Following the model's
|
||||
native 1M directly is also wrong: chars ≠ tokens (CJK runs ~1–1.5 chars/token vs ~4 for
|
||||
English, so a 1M-token char cap would let CJK text sail past the model's real window into an
|
||||
upstream rejection), single near-window requests can consume a large fraction of a 5-hour
|
||||
subscription quota window, and the TUI paste path is untested at megabyte scale.
|
||||
|
||||
## Decision
|
||||
|
||||
The default budget **derives from the SPOT** instead of being a constant:
|
||||
|
||||
```
|
||||
MAX_PROMPT_CHARS (default) = max(models.json models[].contextWindow) × 3 chars/token
|
||||
= 200000 × 3 = 600,000 chars today
|
||||
```
|
||||
|
||||
Implemented as the pure `derivePromptCharBudget(models, {charsPerToken = 3, floor = 150000})`
|
||||
in `lib/prompt.mjs` (unit-tested; floor guards degenerate SPOT states). The multiplier ×3 is
|
||||
deliberately conservative: full window for English, and CJK text reaches the model's real
|
||||
window at roughly the same point the cap fires — so OCP truncates gracefully (tail-first)
|
||||
instead of the upstream rejecting outright.
|
||||
|
||||
`CLAUDE_MAX_PROMPT_CHARS` (env) and the runtime settings API remain **absolute overrides**;
|
||||
the derivation applies only when neither is set.
|
||||
|
||||
## Consequences
|
||||
|
||||
- The advertised `contextWindow: 200000` becomes honest: the proxy now actually accepts
|
||||
prompts of that order (English ≈150–200k tokens) before truncating.
|
||||
- If `models.json` ever advertises a larger window (e.g. 1M for the 1M-native models), the
|
||||
budget scales automatically — no code change. Whether to advertise 1M is a **separate,
|
||||
deliberate decision** (quota burn per request, OpenClaw compaction memory, TUI paste
|
||||
limits) and is explicitly NOT made by this ADR; the current recommendation is to keep
|
||||
200000 advertised until a real >200k use case appears.
|
||||
- One-time behavior change: requests between 150k and 600k chars that were previously
|
||||
truncated now pass through whole — longer TTFT and higher quota consumption for those
|
||||
requests, by design.
|
||||
- The truncation mechanism, logging, and the multimodal-path budget threading (PR #154's F2,
|
||||
pending) are unchanged — only the default value's provenance changed.
|
||||
@@ -0,0 +1,42 @@
|
||||
# Architecture Decision Records
|
||||
|
||||
This directory holds the OCP Architecture Decision Records (ADRs) — short documents that capture the **why** behind structural choices.
|
||||
|
||||
Read these before proposing governance, SPOT (single-source-of-truth), or process changes.
|
||||
|
||||
## Numbering
|
||||
|
||||
ADRs start at `0002`. The first one (`0001`) was reserved for an early
|
||||
internal proposal that was superseded before publication; `0002` is
|
||||
deliberately the first published record so the archived `0001` slot
|
||||
remains a placeholder rather than being silently renumbered.
|
||||
|
||||
New ADRs increment from the highest existing number. Filenames are
|
||||
`NNNN-<short-slug>.md`.
|
||||
|
||||
## Index
|
||||
|
||||
| ADR | Title | What it covers |
|
||||
|---|---|---|
|
||||
| [0002](0002-alignment-constitution.md) | Alignment Constitution | The `ALIGNMENT.md` constitution: why every `server.mjs` change requires `cli.js` citation + independent reviewer + CI blacklist pass. Background: the 2026-04-11 drift incident. |
|
||||
| [0003](0003-models-json-spot.md) | `models.json` as SPOT | Why model IDs / aliases / context windows live in a single JSON file (not duplicated in `server.mjs` and `setup.mjs` arrays). v3.11.0 refactor. |
|
||||
| [0004](0004-openclaw-auto-sync.md) | OpenClaw Auto-Sync | Why `scripts/sync-openclaw.mjs` runs on `ocp update`, what its scope boundary is (writes only `models.providers["claude-local"].models` and `agents.defaults.models["claude-local/*"]`), and the idempotency contract. |
|
||||
| [0005](0005-no-multi-provider.md) | No Multi-Provider | Why OCP stays single-provider (Anthropic-via-cli.js) and does not extend to OpenAI / Gemini / OpenRouter. Cost estimate: ~7 weeks for a v1 that buys neither moat nor commercial readiness. Separate commercial work starts in a separate repo. |
|
||||
| [0006](0006-openai-shim-scope.md) | OpenAI Shim Scope | The Class A / Class B taxonomy. Class A endpoints (`cli.js`-mirror) keep Rules 1–5 verbatim; Class B endpoints (OCP-owned compatibility surface — `/v1/chat/completions`, `/v1/models`, admin endpoints) are anchored to OpenAI's spec (B.1) or to an authorizing ADR (B.2). Triggered by PR #99 (external `response_format` honoring). Grandfathers the existing B.2 inventory at v3.16.4. |
|
||||
| [0007](0007-tui-interactive-mode.md) | TUI Interactive Mode | Why TUI-mode spawns an interactive `claude` in a tmux pane (no `-p`) to reach the **subscription** billing pool (`cc_entrypoint=cli`) rather than the metered Agent SDK pool. Owns the TUI spawn machinery: entrypoint labeling, credential-isolated home, MCP hard-disable, session namespace + defunct-session reaping, the independent concurrency bound, and the `/health` `tui` block. **Single-user only** — hard FATAL on multi-user configs. |
|
||||
| [0008](0008-tui-warm-pane-pool.md) | TUI Warm Pane Pool | Why `OCP_TUI_POOL_SIZE` pre-boots **single-use** `claude` panes (one turn each, own `--session-id`) — and why reuse is forbidden (`transcript.mjs` returns the last assistant entry in the file, so a reused session leaks the earlier turn's text). Measured −41% end-to-end. Defines the pool↔reaper invariant (exemption by exact name from a live registry; drain before every sweep so `kill-server` zombie reaping survives) and the standing idle-process cost. Extends ADR 0007. |
|
||||
| [0009](0009-spot-derived-prompt-budget.md) | SPOT-Derived Prompt Budget | Why `MAX_PROMPT_CHARS`'s default is `max(models.json contextWindow) × 3 chars/token` (600k chars today) instead of a hand-set constant — the old 150k silently under-delivered the advertised window ~5×. ×3 is the CJK-safe multiplier; env/settings stay absolute overrides; whether to advertise 1M windows is explicitly a separate decision. |
|
||||
|
||||
## When to write a new ADR
|
||||
|
||||
Open one whenever:
|
||||
|
||||
- A structural rule is being added or changed (e.g., new SPOT, new boundary, new CI guardrail).
|
||||
- A decision encodes a lesson from an incident or drift.
|
||||
- A future contributor reading the code alone could plausibly undo or re-litigate the choice.
|
||||
|
||||
Skip ADRs for routine implementation choices (algorithm pick, naming) — those belong in commit messages.
|
||||
|
||||
## Format
|
||||
|
||||
Keep ADRs short — Context / Decision / Consequences is the standard skeleton. Cite incidents, PRs, or commits where useful.
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 366 KiB |
@@ -0,0 +1,396 @@
|
||||
Part of [OCP](../README.md) — LAN & multi-user: server setup, client connect, API-key management, per-key quotas, anonymous access, and the deployment/security model (including the honest limits of sharing).
|
||||
|
||||
# LAN & multi-user
|
||||
|
||||
OCP has two roles: **Server** (runs the proxy, needs Claude CLI) and **Client** (connects to a server, zero dependencies).
|
||||
|
||||
```
|
||||
┌─ Server (always-on device) ─────────────────────────────┐
|
||||
│ Mac mini / NAS / Raspberry Pi / Desktop │
|
||||
│ Claude CLI + OCP server → bound to 0.0.0.0:3456 │
|
||||
└───────────────────────┬─────────────────────────────────┘
|
||||
│ LAN
|
||||
┌───────────────────┼───────────────────┐
|
||||
▼ ▼ ▼
|
||||
Laptop Phone/Tablet Pi / Server
|
||||
(client) (browser) (client)
|
||||
```
|
||||
|
||||
## Server Setup
|
||||
|
||||
> **Recommended:** Install OCP on a device that stays powered on — Mac mini, NAS, Raspberry Pi, or a desktop that doesn't sleep. This ensures all clients always have access.
|
||||
|
||||
**Prerequisites:**
|
||||
- macOS or Linux (Windows is not supported — `setup.mjs` installs launchd / systemd auto-start)
|
||||
- Node.js 22.5+ (Node 23+ recommended — `node:sqlite` is fully stable without flags from 23.0; on 22.5–22.x it works behind `--experimental-sqlite`)
|
||||
- `git`
|
||||
- [Claude CLI](https://docs.anthropic.com/en/docs/claude-cli) — install and authenticate:
|
||||
```bash
|
||||
npm install -g @anthropic-ai/claude-code
|
||||
claude auth login # prints a URL + code — open URL on any browser, sign in, paste code back
|
||||
```
|
||||
Headless servers (Pi / NAS / VPS without a desktop browser): see [Headless install notes](#headless-install-notes) below.
|
||||
|
||||
```bash
|
||||
# 1. Clone and run setup
|
||||
git clone https://github.com/dtzp555-max/ocp.git
|
||||
cd ocp
|
||||
node setup.mjs
|
||||
```
|
||||
|
||||
The setup script will:
|
||||
1. Verify Claude CLI is installed and authenticated
|
||||
2. Start the proxy on port 3456
|
||||
3. Install auto-start (launchd on macOS, systemd on Linux)
|
||||
|
||||
After install the `ocp` CLI lives at `~/ocp/ocp`. To put it on your PATH, either symlink it manually (`ln -sf ~/ocp/ocp ~/.local/bin/ocp` if `~/.local/bin` is on your PATH, or `sudo ln -sf ~/ocp/ocp /usr/local/bin/ocp` for a system-wide symlink) or add an alias (`alias ocp=~/ocp/ocp`). Otherwise invoke it as `~/ocp/ocp <subcommand>`. The rest of this document assumes `ocp` is on your PATH.
|
||||
|
||||
> **Cloud/Linux servers:** If `ocp: command not found` after a cloud install, the binary isn't in PATH. Full path in that layout: `~/.openclaw/projects/ocp/ocp`
|
||||
|
||||
**Single-machine use** — just set your IDE to use the proxy:
|
||||
```bash
|
||||
export OPENAI_BASE_URL=http://127.0.0.1:3456/v1
|
||||
```
|
||||
|
||||
**LAN mode** — reach OCP from your own devices on the network (Claude Pro/Max are per-user accounts — see [Sharing with family / a team — honest limits](#deployment-model--security-read-this) before extending access to other people):
|
||||
```bash
|
||||
# Enable LAN access with per-user auth (recommended)
|
||||
node setup.mjs --bind 0.0.0.0 --auth-mode multi
|
||||
```
|
||||
|
||||
Then create API keys for each person/device:
|
||||
```bash
|
||||
# Generate a strong admin key (one-time — save it for later key management):
|
||||
export OCP_ADMIN_KEY=$(openssl rand -base64 32)
|
||||
# Add the same export line to ~/.zshrc or ~/.bashrc so it persists.
|
||||
|
||||
ocp keys add wife-laptop
|
||||
# ✓ Key created for "wife-laptop"
|
||||
# API Key: ocp_example12345abcde...
|
||||
# Copy this key now — you won't see it again.
|
||||
|
||||
ocp keys add son-ipad
|
||||
ocp keys add pi-server
|
||||
```
|
||||
|
||||
Run `ocp lan` to see your IP and ready-to-share instructions.
|
||||
|
||||
**Verify:**
|
||||
```bash
|
||||
curl http://127.0.0.1:3456/v1/models
|
||||
# Returns: claude-opus-4-8, claude-opus-4-7, claude-opus-4-6, claude-sonnet-5, claude-sonnet-4-6, claude-haiku-4-5-20251001
|
||||
```
|
||||
|
||||
### Headless install notes
|
||||
|
||||
OCP is designed for always-on devices that often don't have a desktop browser — Mac mini, NAS, Raspberry Pi, cloud VPS. The Claude CLI auth flow still works headless:
|
||||
|
||||
**Option 1 — interactive OAuth over SSH (one-shot).** `claude auth login` prints a URL + 8-digit code. Open the URL on **any** device with a browser (your laptop, phone), sign in to your Anthropic account, and paste the code back into the SSH session. No browser needed on the server itself.
|
||||
|
||||
**Option 2 — long-lived token (auth once, no re-prompts).**
|
||||
|
||||
```bash
|
||||
claude setup-token # subscription-backed long-lived token
|
||||
```
|
||||
|
||||
Same Claude subscription as Option 1; the token is stored in Claude CLI's normal config location. Useful when you'd rather not redo the OAuth flow when sessions expire.
|
||||
|
||||
If `claude auth login` errors out with something like `cannot open browser`, you've hit the same case — fall back to either option above.
|
||||
|
||||
## AI-assisted install prompts
|
||||
|
||||
If you've got Claude Code, Cursor, or any other AI coding assistant on this machine, you can copy-paste one of these prompts and let the AI walk through the install for you. Each prompt pins the AI to the right README section, names the verification step, and forbids silent retries — so you stay in the loop.
|
||||
|
||||
**Single-machine use** — install OCP for IDEs on this same machine only:
|
||||
|
||||
```text
|
||||
I want to install OCP on this machine to use my Claude Pro/Max subscription
|
||||
as an OpenAI-compatible API for local IDEs.
|
||||
|
||||
Please follow https://github.com/dtzp555-max/ocp/blob/main/README.md
|
||||
§Quickstart (single-machine install):
|
||||
|
||||
1. Verify prerequisites: macOS or Linux, Node.js 22.5+, git, Claude CLI
|
||||
installed and logged in (`claude auth status`). Install missing pieces
|
||||
using my system's package manager.
|
||||
2. git clone the repo, cd in, and run `node setup.mjs`.
|
||||
3. Verify with `curl http://127.0.0.1:3456/v1/models` (should list 6 models).
|
||||
4. Add `export OPENAI_BASE_URL=http://127.0.0.1:3456/v1` to my shell rc.
|
||||
5. Tell me to reload my shell and try a tool like Cline / Continue / Cursor.
|
||||
|
||||
Before each step, tell me what you'll run and wait for confirmation.
|
||||
On any error, diagnose first — don't auto-retry.
|
||||
```
|
||||
|
||||
**LAN mode (server)** — install OCP as a server so your own devices on the LAN can reach it (Claude Pro/Max are per-user accounts — review Anthropic's Usage Policy before extending access to other people):
|
||||
|
||||
```text
|
||||
I want to install OCP on this device as a LAN server so my own devices on the
|
||||
network can reach my Claude Pro/Max subscription through a local
|
||||
OpenAI-compatible endpoint.
|
||||
|
||||
Please follow https://github.com/dtzp555-max/ocp/blob/main/docs/lan-mode.md
|
||||
"Server Setup" → "LAN mode" path:
|
||||
|
||||
1. Verify prerequisites: macOS or Linux (Windows not supported), Node.js
|
||||
22.5+, git, Claude CLI installed and authenticated.
|
||||
2. Generate a strong admin key with `openssl rand -base64 32`. Save it —
|
||||
I'll need it to manage per-user keys later.
|
||||
3. git clone https://github.com/dtzp555-max/ocp.git && cd ocp
|
||||
4. Run `node setup.mjs --bind 0.0.0.0 --auth-mode multi`.
|
||||
5. Add OCP_ADMIN_KEY to my shell rc (~/.zshrc or ~/.bashrc).
|
||||
6. Run `ocp lan` to show me the LAN IP and connect command.
|
||||
7. Optionally create example keys: `ocp keys add laptop`, `ocp keys add tablet`.
|
||||
8. Verify: `curl http://127.0.0.1:3456/v1/models` returns 6 models.
|
||||
|
||||
Tell me each step before running it. On error, diagnose before retrying.
|
||||
```
|
||||
|
||||
**Client connect** — configure this device to use an existing OCP server on your LAN:
|
||||
|
||||
```text
|
||||
There's an OCP server at <SERVER_IP> on my LAN. Configure this machine to
|
||||
use it for any local IDEs (Cursor, Cline, Continue.dev, OpenCode, OpenClaw).
|
||||
|
||||
Server IP: <SERVER_IP>
|
||||
API key (leave blank if the server has anonymous mode enabled): <OPTIONAL_KEY>
|
||||
|
||||
Please follow https://github.com/dtzp555-max/ocp/blob/main/docs/lan-mode.md
|
||||
"Client Setup" path:
|
||||
|
||||
1. Download ocp-connect:
|
||||
curl -fsSL https://raw.githubusercontent.com/dtzp555-max/ocp/main/ocp-connect -o ocp-connect
|
||||
chmod +x ocp-connect
|
||||
2. Run `./ocp-connect <SERVER_IP>` (add `--key <KEY>` if you have one).
|
||||
3. Follow any IDE-specific manual hints it prints.
|
||||
4. Verify: `curl http://<SERVER_IP>:3456/v1/models` returns 6 models.
|
||||
5. Tell me to reload my shell + restart any IDE that was already running.
|
||||
|
||||
Don't auto-retry on error. Tell me the failure mode first.
|
||||
```
|
||||
|
||||
## Client Setup
|
||||
|
||||
> Clients do **not** need to install Node.js, Claude CLI, or the OCP repo. Only `curl` and `python3` are required (pre-installed on most Linux/Mac systems).
|
||||
>
|
||||
> **Find the server's LAN IP** by running `ocp lan` on the server machine — it prints both the IP and a ready-to-share connect command.
|
||||
|
||||
**One-command setup** — download the lightweight `ocp-connect` script:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/dtzp555-max/ocp/main/ocp-connect -o ocp-connect
|
||||
chmod +x ocp-connect
|
||||
./ocp-connect <server-ip>
|
||||
```
|
||||
|
||||
**Zero-config** — when the server admin has set `PROXY_ANONYMOUS_KEY` *and* opted in with `PROXY_ADVERTISE_ANON_KEY=1` (see [Anonymous Access](#anonymous-access-optional) below), just pass the server IP and nothing else. `ocp-connect` reads the anonymous key from `/health` and uses it automatically. Without the opt-in, `/health` does not expose the key (issue #109); pass `--key` or rely on anonymous access instead:
|
||||
|
||||
```bash
|
||||
./ocp-connect <server-ip>
|
||||
```
|
||||
|
||||
If the server requires a key, pass it with `--key`:
|
||||
```bash
|
||||
./ocp-connect <server-ip> --key <your-api-key>
|
||||
```
|
||||
|
||||
Or as a one-liner (no file saved):
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/dtzp555-max/ocp/main/ocp-connect | bash -s -- <server-ip>
|
||||
```
|
||||
|
||||
Example:
|
||||
```
|
||||
$ ./ocp-connect 192.168.1.100
|
||||
|
||||
OCP Connect v1.3.0
|
||||
─────────────────────────────────────
|
||||
Remote: http://192.168.1.100:3456
|
||||
|
||||
Checking connectivity...
|
||||
✓ Connected
|
||||
|
||||
Remote OCP v3.11.0 (auth: multi)
|
||||
|
||||
ⓘ Using server-advertised anonymous key: ocp_publ...n_v1
|
||||
(set by admin via PROXY_ANONYMOUS_KEY; see issue #12 §14 Path A)
|
||||
|
||||
Testing API access...
|
||||
✓ API accessible (6 models available)
|
||||
|
||||
Shell config:
|
||||
✓ .bashrc
|
||||
✓ .zshrc
|
||||
OPENAI_BASE_URL=http://192.168.1.100:3456/v1
|
||||
|
||||
System-level (launchctl):
|
||||
✓ OPENAI_BASE_URL set for GUI apps and daemons
|
||||
|
||||
IDE Configuration
|
||||
─────────────────────────────────────
|
||||
Detected: OpenClaw (~/.openclaw/openclaw.json)
|
||||
|
||||
Configure OpenClaw to use this OCP? [Y/n] y
|
||||
Provider name (models show as <name>/model-id) [ocp]: ocp
|
||||
|
||||
How should OCP models be configured?
|
||||
1) Primary — use OCP by default, keep existing models as backup
|
||||
2) Backup — keep current primary, add OCP as additional option
|
||||
|
||||
Choice [1]: 1
|
||||
|
||||
Writing OpenClaw config...
|
||||
✓ Per-agent auth profile seeded (2):
|
||||
• ~/.openclaw/agents/main/agent/auth-profiles.json
|
||||
• ~/.openclaw/agents/macbook_bot/agent/auth-profiles.json
|
||||
✓ OpenClaw configured
|
||||
Provider: ocp
|
||||
Models:
|
||||
• ocp/claude-opus-4-8
|
||||
• ocp/claude-opus-4-7
|
||||
• ocp/claude-opus-4-6
|
||||
• ocp/claude-sonnet-5
|
||||
• ocp/claude-sonnet-4-6
|
||||
• ocp/claude-haiku-4-5-20251001
|
||||
Priority: PRIMARY (default model)
|
||||
|
||||
Restart OpenClaw to apply: openclaw gateway restart
|
||||
|
||||
Running smoke test...
|
||||
✓ Smoke test passed: OK
|
||||
Note: smoke test only verifies OCP is reachable and the key is valid.
|
||||
It does not verify your IDE/agent end-to-end. To verify OpenClaw works,
|
||||
restart it (`openclaw gateway restart`) and send a test message to your bot.
|
||||
|
||||
Done. Reload your shell to apply:
|
||||
source ~/.zshrc
|
||||
```
|
||||
|
||||
The script automatically:
|
||||
- Writes env vars to all relevant shell rc files (`.bashrc`, `.zshrc`)
|
||||
- Sets system-level env vars (`launchctl setenv` on macOS, `environment.d` on Linux)
|
||||
- **Auto-discovers anonymous key** from `/health.anonymousKey` when no `--key` given (v1.3.0+, requires server v3.10.0+; server must also set `PROXY_ADVERTISE_ANON_KEY=1` — see [Anonymous Access](#anonymous-access-optional))
|
||||
- Configures OpenClaw automatically (including per-agent `auth-profiles.json` for multi-agent setups)
|
||||
- Detects Cline, Continue.dev, Cursor, and opencode, and prints setup hints (manual configuration required for these IDEs)
|
||||
|
||||
On macOS, `launchctl setenv` vars reset on reboot — re-run `ocp-connect` after restart.
|
||||
|
||||
**Manual setup** — if you prefer not to use the script:
|
||||
```bash
|
||||
export OPENAI_BASE_URL=http://<server-ip>:3456/v1
|
||||
export OPENAI_API_KEY=ocp_<your-key>
|
||||
```
|
||||
Add these lines to `~/.bashrc` or `~/.zshrc` to persist across sessions.
|
||||
|
||||
## Monitoring (Server-side)
|
||||
|
||||
```bash
|
||||
# Per-key usage stats
|
||||
ocp usage --by-key
|
||||
# Key Reqs OK Err Avg Time
|
||||
# wife-laptop 5 5 0 8.0s
|
||||
# son-ipad 3 3 0 6.2s
|
||||
|
||||
# Manage keys
|
||||
ocp keys # List all keys
|
||||
ocp keys revoke son-ipad # Revoke a key
|
||||
```
|
||||
|
||||
**Web Dashboard:** Open `http://<server-ip>:3456/dashboard` in any browser for real-time monitoring — per-key usage, request history, plan utilization, and system health.
|
||||
|
||||

|
||||
|
||||
## Auth Modes
|
||||
|
||||
| Mode | Env | Use Case |
|
||||
|------|-----|----------|
|
||||
| `none` | `CLAUDE_AUTH_MODE=none` | Trusted home network, no auth needed |
|
||||
| `shared` | `CLAUDE_AUTH_MODE=shared` + `PROXY_API_KEY=xxx` | Everyone shares one key |
|
||||
| `multi` | `CLAUDE_AUTH_MODE=multi` + `OCP_ADMIN_KEY=xxx` | Per-person keys for usage tracking + quotas (trusted users only — see Deployment model below) |
|
||||
|
||||
> **Usage scope (v3.14.0+):** `/api/usage` returns the caller's own rows by default. Admin callers must pass `?all=true` to retrieve data for all keys; doing so emits an audit log line.
|
||||
|
||||
## Deployment model & security (read this)
|
||||
|
||||
**What OCP is built for today: single-user, multi-IDE.** Run OCP as a server on one machine and point all of *your own* IDEs/devices at it — one Claude Pro/Max subscription, used everywhere. This is the primary, solid use case.
|
||||
|
||||
**Sharing with family / a team — honest limits.** You *can* share OCP on a LAN, but be clear about what the auth modes do and don't give you:
|
||||
|
||||
- The per-key modes (`shared` / `multi`) give per-key **usage tracking, quotas, and cache separation** — useful for seeing who used what and capping budgets.
|
||||
- They do **not** give a **security isolation boundary**. The spawned `claude` runs with the **operator's filesystem access** and is *not* sandboxed per key. **Only share with people you fully trust, on a trusted network.**
|
||||
- For simple trusted family sharing, the easiest setup is a single shared **anonymous key** (see [Anonymous Access](#anonymous-access-optional)) — no per-person separation, same trust assumption.
|
||||
- **Account terms and ToS — read before sharing with others.** Claude Pro/Max are *per-user* accounts. Pooling a single subscription across **multiple distinct people** may violate Anthropic's Consumer Terms of Service and risk account suspension by the abuse classifier. The defensible framing is **"one person, your own devices"** — sharing with friends or a team is not. OCP does not change your account terms, and whether any particular sharing setup complies with the ToS is the account holder's responsibility. Review Anthropic's Usage Policy before extending access to other people.
|
||||
|
||||
**Real per-user isolation (sandboxed, multi-tenant-safe) is planned for after 2026-06-15** — per-key ephemeral home + tool lockdown + an OS sandbox. Until then, treat a multi-user OCP as a *trusted-group convenience*, not a security boundary. (This is also why `CLAUDE_TUI_MODE` is single-user-only — see [Subscription-pool (TUI) mode](tui-mode.md#subscription-pool-tui-mode).)
|
||||
|
||||
## Anonymous Access (optional)
|
||||
|
||||
In `multi` mode, the admin can designate a single well-known "anonymous" key that bypasses `validateKey()` and grants public read/write access. This is useful for letting LAN users (or clients like OpenClaw multi-agent setups) connect without individual per-user keys.
|
||||
|
||||
**Enable**:
|
||||
|
||||
The anonymous key is wired into the service unit (launchd plist on macOS, systemd unit on Linux) at install time. Export `PROXY_ANONYMOUS_KEY` in your shell before running `setup.mjs`, and `setup.mjs` will write it into the service unit env so the auto-started proxy picks it up:
|
||||
|
||||
```bash
|
||||
export PROXY_ANONYMOUS_KEY=ocp_public_anon # or any string of your choice
|
||||
node setup.mjs --bind 0.0.0.0 --auth-mode multi
|
||||
```
|
||||
|
||||
If OCP is already installed without it, re-export the env var and re-run `node setup.mjs` (the installer is idempotent — it refreshes the service unit). Then `ocp restart` so the running proxy picks up the new env. Setting `PROXY_ANONYMOUS_KEY` only in your interactive shell **does not** affect the auto-started proxy — the service unit is the source of truth for its environment.
|
||||
|
||||
**Client side**: the anonymous key value is exposed via `GET /health` as the field `anonymousKey` (null when not set) **only to localhost callers** or when the admin has also set `PROXY_ADVERTISE_ANON_KEY=1` (default off — see issue #109). With that opt-in, clients like `ocp-connect` can auto-discover and use it, so the end user doesn't need to get a personal key from the admin.
|
||||
|
||||
**Security note**: setting this env var is an **opt-in** to public access — anyone who can reach your OCP endpoint can use it, up to any rate limits you configure. Don't enable this on internet-exposed OCP instances without additional protection.
|
||||
|
||||
**Not a secret**: because `/health` is an unauthenticated endpoint, the anonymous key is **publicly readable** by anyone who can reach the server. That is intentional — the key exists so clients can self-configure without out-of-band coordination. Treat it as a convenience handle, not as an access credential.
|
||||
|
||||
## Per-Key Quota (Budget Control)
|
||||
|
||||
Prevent any single user from exhausting your subscription. Set daily, weekly, or monthly request limits per API key:
|
||||
|
||||
```bash
|
||||
# Set a daily limit of 50 requests for a key
|
||||
curl -X PATCH http://127.0.0.1:3456/api/keys/wife-laptop/quota \
|
||||
-H "Authorization: Bearer $OCP_ADMIN_KEY" \
|
||||
-d '{"daily": 50}'
|
||||
|
||||
# Set multiple limits at once
|
||||
curl -X PATCH http://127.0.0.1:3456/api/keys/son-ipad/quota \
|
||||
-H "Authorization: Bearer $OCP_ADMIN_KEY" \
|
||||
-d '{"daily": 20, "weekly": 100}'
|
||||
|
||||
# Check current quota + usage
|
||||
curl http://127.0.0.1:3456/api/keys/wife-laptop/quota
|
||||
# → { "daily": { "limit": 50, "used": 12 }, "weekly": { "limit": null, "used": 34 }, ... }
|
||||
|
||||
# Remove a limit (set to null)
|
||||
curl -X PATCH http://127.0.0.1:3456/api/keys/wife-laptop/quota \
|
||||
-d '{"daily": null}'
|
||||
```
|
||||
|
||||
When a key exceeds its quota, OCP returns HTTP 429 with a structured error:
|
||||
```json
|
||||
{
|
||||
"error": {
|
||||
"message": "Quota exceeded: 50/50 requests (daily). Resets 6h 12m.",
|
||||
"type": "quota_exceeded",
|
||||
"quota": { "period": "daily", "limit": 50, "used": 50, "resetsIn": "6h 12m" }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
- `null` = unlimited (default for all keys)
|
||||
- Only successful requests count toward quota
|
||||
- Admin and anonymous users are never subject to quotas
|
||||
- PATCH is a partial update — omitted fields are left unchanged
|
||||
|
||||
> **Note:** quotas are best-effort. Under concurrent bursts a key can exceed its cap by up to the server's max-concurrency (default 8), and cache hits are not counted toward quota. They cap budgets for cooperative family use, not adversarial abuse.
|
||||
|
||||
## Important Notes
|
||||
|
||||
- All users share your Claude Pro/Max **rate limits** (5h session + 7d weekly)
|
||||
- `ocp usage` shows how much quota remains
|
||||
- Keys are stored in `~/.ocp/ocp.db` (SQLite, zero external dependencies)
|
||||
- Admin key is required for key management API endpoints
|
||||
- The dashboard (`/dashboard`) and health check (`/health`) are always public
|
||||
- File modes for `~/.ocp` (0700), `admin-key` + `ocp.db` (0600) are auto-tightened at server startup as of v3.14.0
|
||||
@@ -0,0 +1,11 @@
|
||||
# OpenAI Compatibility Pin (Class B.1)
|
||||
|
||||
**Status:** Placeholder — populated at first B.1 audit per ADR 0006 §"Class B audit cadence".
|
||||
|
||||
This file is the Class B.1 counterpart to the Class A `cli.js` audit pin in `ALIGNMENT.md` §"Golden Reference". When the first annual alignment audit covers Class B (per `ALIGNMENT.md` §"Annual Alignment Audit"), this file will be populated with:
|
||||
|
||||
- The OpenAI `/v1/chat/completions` specification snapshot date being audited against (and the source URL the snapshot was taken from).
|
||||
- The list of B.1 endpoints (currently `/v1/chat/completions`, `/v1/models`) and, for each one, the specific OpenAI spec fields and behaviours it honors.
|
||||
- Drift detection notes for any OpenAI spec changes since the previous audit, and any OCP code changes required to track those changes.
|
||||
|
||||
Until populated, this file's existence is only a forward reference so that the link in `ALIGNMENT.md` does not 404. The actual audit procedure is defined in `ALIGNMENT.md` §"Annual Alignment Audit" (Class B scope) and ADR 0006 §"Class B audit cadence".
|
||||
@@ -0,0 +1,236 @@
|
||||
# TUI-mode latency: measured floor, and the four things worth fixing
|
||||
|
||||
**Date**: 2026-07-13
|
||||
**Status**: findings + backlog. **Superseded in part** — see the dated update boxes below.
|
||||
Item #1 shipped ([#156](https://github.com/dtzp555-max/ocp/pull/156)); item #2 is **dead**
|
||||
([`streaming-spike.md`](streaming-spike.md)); item #4 measured, **no effect**; item #3 stands.
|
||||
**Measured on**: Mac mini / macOS 26.5.2 / Claude Code **v2.1.207** / Sonnet 5 / Claude Max subscription / **real-home mode** (no `CLAUDE_CODE_OAUTH_TOKEN`, no `OCP_TUI_HOME` in the service env)
|
||||
**Evidence**: [`measurements.jsonl`](measurements.jsonl) — **n=15** (3 configs × 5) · banner captures [`billing-banner.txt`](billing-banner.txt) · harness [`floor.sh`](floor.sh)
|
||||
|
||||
## Why this exists
|
||||
|
||||
An external consumer (the 知音 AI project) benchmarked OCP's prompt path and measured
|
||||
**TTFT p50 ≈ 30–32 s**, and excluded OCP as a backend on that basis. That number is real,
|
||||
but it is *not* the model being slow — this document decomposes where the 30 seconds
|
||||
actually go, and what OCP can do about it.
|
||||
|
||||
**The harness deliberately does not go through OCP.** It spawns `tmux` + `claude` directly
|
||||
(session prefix `zhiyin-floor-`, never `ocp-tui-*`) and polls `tmux capture-pane` for
|
||||
incremental render, so it measures the **true first-token time** of the underlying
|
||||
subscription path — the floor OCP could reach if it were perfect.
|
||||
|
||||
---
|
||||
|
||||
## Measurements
|
||||
|
||||
All rows in [`measurements.jsonl`](measurements.jsonl); every number below is recomputable from it.
|
||||
|
||||
| Config | n | boot→input-ready (median) | **TTFT (median)** | TTFT range | full answer (median) |
|
||||
|---|---|---|---|---|---|
|
||||
| baseline (inherits global `effortLevel: xhigh`) | 5 | 1.07 s | **10.35 s** | 8.32 – 17.19 s | 11.32 s |
|
||||
| **`--effort low`** | 5 | 1.03 s | **6.17 s** | **5.87 – 6.44 s** | 9.98 s |
|
||||
| `--bare` | 5 | 0.44 s | **no answer at all** (5/5 `ttft_ms: -1`) | — | — |
|
||||
|
||||
> **Not from this harness**: the direct Anthropic API reference figure (TTFT 0.84–1.64 s, n=2)
|
||||
> comes from the 知音 AI project's own smoke test, not from `measurements.jsonl`. It is quoted
|
||||
> only to size the gap; do not look for it in the evidence file.
|
||||
|
||||
### Where the 30 seconds go
|
||||
|
||||
```
|
||||
~1.0 s spawn → claude's input bar is ready ← NOT the bottleneck
|
||||
~6-10 s true TTFT (first token rendered in the pane)
|
||||
~20 s ████ waiting for the whole turn to finish ████ ← this is the 30s
|
||||
```
|
||||
|
||||
`runTuiTurn` blocks on the native transcript until a terminal event (`lib/tui/session.mjs`
|
||||
"Block on the native transcript … until terminal"; `readTuiTranscript` in
|
||||
`lib/tui/transcript.mjs`; ADR 0007 step 4) — i.e. it waits for the **entire turn** to complete
|
||||
before returning anything. There is no streaming path. The ~20 s delta between this harness's
|
||||
real TTFT and OCP's reported 30–32 s is exactly that.
|
||||
|
||||
> **⚠️ 2026-07-13 correction — this decomposition attributes the ~20 s to the wrong thing.** It was
|
||||
> inferred from the external 30–32 s report, never measured *through* OCP. It has since been measured
|
||||
> through a real OCP instance (TUI mode, `claude-sonnet-4-6`, the same ~1850-token prompt, n=5):
|
||||
> **median 11.30 s** before [#156](https://github.com/dtzp555-max/ocp/pull/156), **9.55 s** after.
|
||||
> Same-turn decomposition (baseline row `i=5`): **11.563 s** wall through OCP vs `turn_duration:
|
||||
> 7.319 s` of CLI-internal time on that same turn → **OCP's own overhead ≈ 4.2 s** (n=1), **not
|
||||
> ~20 s**. The rest of any larger number is the model *generating a long answer*,
|
||||
> which the blocking wait does not cause and streaming would not shorten — it would only move the
|
||||
> first byte earlier. The 30–32 s figure therefore reflects a much longer output (and/or the
|
||||
> then-inherited `xhigh` effort), not 20 s of OCP dead time. See
|
||||
> [`streaming-spike.md`](streaming-spike.md) § "What streaming would have bought".
|
||||
|
||||
---
|
||||
|
||||
## ⚠️ Blocking constraint: `--bare` silently drops you off the subscription pool
|
||||
|
||||
Captured live ([`billing-banner.txt`](billing-banner.txt)) — the startup banner is the **only**
|
||||
reliable indicator:
|
||||
|
||||
```
|
||||
[] | Sonnet 5 with xhigh effort · Claude Max
|
||||
[--effort low] | Sonnet 5 with low effort · Claude Max
|
||||
[--bare] | Sonnet 5 with xhigh effort · API Usage Billing ← ❌
|
||||
```
|
||||
|
||||
`--bare` ("skip hooks, LSP, plugin…") **also skips the subscription-credential resolution
|
||||
path**. It really does cut boot to 0.43–0.45 s — but you are no longer on the subscription,
|
||||
which defeats the entire purpose of TUI mode (ADR 0007 exists solely to reach the
|
||||
subscription pool).
|
||||
|
||||
**The failure is silent.** All 5 `--bare` samples reached input-ready (boot 0.43–0.45 s), were
|
||||
sent the prompt, and then produced **no answer at all** — 60 s timeout, no error, no crash, the
|
||||
pane simply never rendered a token (the API-billing account had no credit balance). Nothing in
|
||||
the transcript or the exit status reveals this.
|
||||
|
||||
**Anyone changing spawn flags must diff the banner line before and after.**
|
||||
|
||||
---
|
||||
|
||||
## Backlog — four items, ranked by value ÷ effort
|
||||
|
||||
### 1. Pass `--effort` explicitly on spawn — **do this first**
|
||||
|
||||
`buildTuiCmd` (`lib/tui/session.mjs`) does not pass `--effort` — `grep -rn -- "--effort\|effortLevel" lib/ server.mjs`
|
||||
returns zero hits. What the pane's `claude` ends up using therefore depends on **which HOME mode
|
||||
`resolveTuiHome()` picked**:
|
||||
|
||||
| mode | HOME | effort the pane gets |
|
||||
|---|---|---|
|
||||
| **real-home** (legacy default — *current* service config: no `CLAUDE_CODE_OAUTH_TOKEN`, no `OCP_TUI_HOME`) | `~` | **inherits the operator's `~/.claude/settings.json` → `effortLevel: xhigh` on this host** |
|
||||
| env-token scratch (`CLAUDE_CODE_OAUTH_TOKEN` set — the direction #146/#150 pushed) | `~/.ocp-tui/home` | that settings.json contains only `permissions.additionalDirectories`; `prepareTuiHome()` never writes `effortLevel` → **claude's built-in default** |
|
||||
|
||||
**Scope note**: TUI mode is currently *off* on this host (`CLAUDE_TUI_MODE=false`; `/health` →
|
||||
`"tui": {"enabled": false}`), so live traffic takes the `-p` path today. The statement below is
|
||||
about what happens **when TUI mode is enabled**.
|
||||
|
||||
On the current HOME config, **every TUI request would run extended thinking** — pure waste
|
||||
for the typical "generate this JSON" request, and it makes latency depend on an unrelated global
|
||||
setting the operator may have changed for their own interactive use. And the mode split means
|
||||
the effort level silently changes if the operator ever switches to env-token mode.
|
||||
**Passing `--effort` explicitly fixes both problems at once.**
|
||||
|
||||
- **Effect (real-home, measured)**: TTFT p50 **10.35 s → 6.17 s (−40 %)**, and the spread
|
||||
collapses from 8.32–17.19 s to **5.87–6.44 s**. For a proxy, the variance reduction matters
|
||||
more than the median.
|
||||
- **Cost**: one flag. Suggested: a new `OCP_TUI_EFFORT` env var (default `low`), documented in
|
||||
README § "Environment Variables" per `release_kit.new_feature_doc_expectations`.
|
||||
- **Risk**: none — banner confirms it stays on `Claude Max` (see `billing-banner.txt`).
|
||||
- ⚠️ Do **not** reach for `--bare` to shave boot: see above.
|
||||
|
||||
### 2. Real streaming instead of blocking on turn-terminal — **ACHIEVABLE → [`streaming-spike.md`](streaming-spike.md)**
|
||||
|
||||
> **2026-07-13 update — the prereq spike was run. The answer is YES, but not from either source this
|
||||
> item guessed at.** (a) The transcript grows at *event* granularity (the whole answer lands in one
|
||||
> line, ~0.3 s before terminal) — dead. (b) The pane is a **rendered** view whose `capture-pane` text
|
||||
> no longer contains the answer's source bytes (`## `, `**`, code fences are gone) — dead, and worse
|
||||
> than "lossy": it is *not the model's text*. **But there is a third source neither this backlog nor
|
||||
> the first spike considered: `claude` fires a `MessageDisplay` hook carrying incremental,
|
||||
> byte-faithful `delta`s of the raw reply.** Verified live on a plain interactive TUI spawn (no `-p`),
|
||||
> banner `· Claude Max`: 7 fires spread across generation, `concat(deltas) === T` **byte-exactly**
|
||||
> (579 == 579), `T.startsWith(S)` true at every step, `## ` / `**` / ```` ```javascript ```` all
|
||||
> present in the deltas. Granularity is block-level (~5–7 chunks/answer), not token-level — plenty for
|
||||
> SSE. **Build it.**
|
||||
>
|
||||
> ⚠️ Two corrections to this item as written: the **"~20 s" is wrong** (inferred from an external
|
||||
> report, never measured through OCP — the same-turn decomposition puts OCP's own overhead at **~4 s**,
|
||||
> n=1), and **streaming moves the first byte, not the last** — so a consumer needing the *complete*
|
||||
> answer (the JSON-card case that motivated this) gains **nothing** from it. Build it for
|
||||
> progressively-rendering consumers, not as a throughput win.
|
||||
>
|
||||
> Full evidence + implementer caveats (the hook is `forceSyncExecution` — claude BLOCKS on it):
|
||||
> **[`streaming-spike.md`](streaming-spike.md)**. Original framing preserved below.
|
||||
|
||||
Today `runTuiTurn` blocks on the transcript until the turn is *finished*. The pane is already
|
||||
rendering tokens incrementally the whole time — this harness proves you can observe first token
|
||||
at ~6 s by polling `tmux capture-pane`.
|
||||
|
||||
- **Effect**: turns a 30 s wall into a ~6 s TTFT with progressive output; enables SSE streaming
|
||||
on the OCP endpoint instead of a single blob at the end.
|
||||
- **Cost**: real work. Pane capture is ANSI/redraw-based and lossy for exact text (wrapping,
|
||||
scrollback, spinner lines). Two candidate sources: (a) incremental reads of the transcript
|
||||
JSONL, (b) `capture-pane` diffing with a stable start marker. (a) is much cleaner **if it
|
||||
holds**.
|
||||
- **Prereq spike (do this before designing anything)**: does the transcript JSONL grow *during*
|
||||
a turn, or only at the end? If only at the end, (a) is dead and you are stuck with (b).
|
||||
|
||||
### 3. Warm pane pool — ~1 s
|
||||
|
||||
Every request spawns a fresh tmux session + `claude` (`randomUUID()` + `new-session`, then
|
||||
`kill-session` in `finally`; `grep -rn "pool\|warm\|reuse" lib/tui/*.mjs` → zero hits). Boot to
|
||||
input-ready is ~1.0 s, paid on every request. A pool of pre-booted panes (single-use, replaced in
|
||||
the background) amortizes it to zero for any workload below the pool refill rate.
|
||||
|
||||
- **Effect**: −1.0 s.
|
||||
- **Cost**: moderate; interacts with the session reaper and the per-port prefix scoping added in
|
||||
#148 — pooled panes must not look like zombies to the sweep.
|
||||
- Lower priority than #1 and #2: it is the smallest slice.
|
||||
|
||||
### 4. Trim the prefill — ~~probably not worth it~~ **MEASURED: no detectable benefit. Do not adopt.**
|
||||
|
||||
> **2026-07-13 update.** `--exclude-dynamic-system-prompt-sections` was measured with the same
|
||||
> harness (`floor.sh`, n=5, Sonnet 5, on top of `--effort low`): **TTFT median 6.39 s**
|
||||
> (5.87–10.54 s) vs **6.17 s** (5.87–6.44 s) for `--effort low` alone — i.e. **0.22 s worse, inside
|
||||
> the noise band**, with one worse outlier; dropping that outlier does not change the verdict. n=5
|
||||
> cannot prove "zero", only "no benefit detectable above noise" — but there is also a **mechanistic**
|
||||
> reason not to expect one: `--help` says the flag *"Improves cross-user prompt-cache **reuse**"*, and
|
||||
> **OCP is single-user** — there is no cross-user cache to share, so the flag has nothing to buy here.
|
||||
> The banner stayed on `· Claude Max` (no billing-pool drop), but there is no win to bank. The ~6 s
|
||||
> floor stands as stated below. Raw rows: [`prefill-spike-measurements.jsonl`](prefill-spike-measurements.jsonl).
|
||||
|
||||
|
||||
After #1–#3, the floor is **~6 s**, and it does not go lower. `claude` always injects the full
|
||||
Claude Code system prompt + tool definitions (thousands to tens of thousands of prefill tokens)
|
||||
regardless of what you ask it. `--exclude-dynamic-system-prompt-sections` exists and may shave
|
||||
some of it — **unmeasured**; worth one spike, but do not expect to reach the direct API's
|
||||
~1 s.
|
||||
|
||||
**Consequence to accept, and to state in the README**: even fully optimized, TUI mode has a
|
||||
**~6 s TTFT floor**, so it cannot serve real-time / interactive-latency consumers. It remains
|
||||
appropriate for batch, background, and cost-insensitive-latency use. The 知音 AI project
|
||||
excluded it on this basis (their prompt-latency budget is 2–4 s) *independently* of the ToS
|
||||
question already documented in the README.
|
||||
|
||||
---
|
||||
|
||||
## Reproduction
|
||||
|
||||
```bash
|
||||
# harness never touches OCP's :3456 service or ocp-tui-* sessions, and never kill-server
|
||||
bash docs/plans/2026-07-13-tui-latency/floor.sh 5 # baseline
|
||||
TAG=effort-low EXTRA_ARGS="--effort low" bash .../floor.sh 5 # −40 %
|
||||
TAG=bare EXTRA_ARGS="--bare" bash .../floor.sh 5 # the trap
|
||||
|
||||
# billing-pool check for ANY spawn-flag change — the banner is the only source of truth
|
||||
tmux new-session -d -s probe -x 200 -y 50 -c "$HOME" \
|
||||
"claude --model claude-sonnet-5 --session-id $(uuidgen) <your-flags-here>"
|
||||
sleep 6; tmux capture-pane -p -t probe | grep -E "Claude Max|API Usage Billing"
|
||||
tmux kill-session -t probe
|
||||
```
|
||||
|
||||
## Interaction with OCP while the harness runs
|
||||
|
||||
- **Kill direction is safe both ways**: `reapStaleTuiSessions()` only `kill-session`s names
|
||||
matching `ocp-tui-<port>-`, which `zhiyin-floor-*` never matches; and the harness only
|
||||
`kill-session`s its own single session — it contains **no `kill-server`**.
|
||||
- **One benign interaction** (only when TUI mode is enabled — the reap tick is itself gated on
|
||||
`TUI_MODE`): OCP's periodic `kill-server` (zombie reaping) is gated on
|
||||
`othersRemain` — *any* foreign-prefixed tmux session suppresses it. So while the harness is
|
||||
running, that sweep is skipped. This is the coexistence guard working as designed; it resumes
|
||||
on the next tick.
|
||||
|
||||
## Harness caveats (stated so the numbers are not over-trusted)
|
||||
|
||||
- **n=5 per config**, single host, single model (Sonnet 5), single prompt size (~1850 tokens).
|
||||
Enough to separate 6 s from 10 s from 30 s; **not** enough for a p95.
|
||||
- TTFT is "marker visible in `capture-pane`", which includes tmux render latency (small, but
|
||||
nonzero) — it is an upper bound on the true first-token time.
|
||||
- **The harness's readiness marker is not OCP's.** `floor.sh` waits for `│ >|❯|Try "`; OCP's
|
||||
`tuiInputReady()` matches `/\? for shortcuts/`. These are different events, so the ~1.0 s
|
||||
boot figure is **not** directly comparable to OCP's `BOOT_MS` gate (default cap 4000 ms). It
|
||||
does not affect the conclusions (1 s ≪ 6 s TTFT), but it is not apples-to-apples.
|
||||
- The first version of this harness reported TTFT **0.08 s** — a false positive: the prompt
|
||||
literally contained the marker string it was grepping for, so the match fired the instant the
|
||||
prompt was pasted. Fixed by describing the marker instead of spelling it. **The script exited 0
|
||||
and "successfully" produced 5 samples both times** — exit status proves nothing here.
|
||||
@@ -0,0 +1,3 @@
|
||||
[] | ▝▜█████▛▘ Sonnet 5 with xhigh effort · Claude Max
|
||||
[--effort low] | ▝▜█████▛▘ Sonnet 5 with low effort · Claude Max
|
||||
[--bare] | ▝▜█████▛▘ Sonnet 5 with xhigh effort · API Usage Billing
|
||||
Executable
+128
@@ -0,0 +1,128 @@
|
||||
#!/usr/bin/env bash
|
||||
# OCP TUI-mode latency floor harness — see README.md in this directory.
|
||||
#
|
||||
# 目的:回答一个问题——如果把 OCP 现有的两个已知开销砍掉
|
||||
# (a) 每请求 spawn + boot(可用预热进程池消除)
|
||||
# (b) 假流式(等 turn_duration 才返回,可用增量读 pane 消除)
|
||||
# 之后,订阅池路径的**真实 TTFT 地板**是多少?
|
||||
#
|
||||
# 判据:地板 ≤ 4s → OCP 作为"省钱选项"可行;> 8s → 死透,不再讨论。
|
||||
#
|
||||
# 红线:
|
||||
# - 不经过生产 OCP 服务(:3456)—— 直接起 tmux+claude,OCP 进程零干扰
|
||||
# - tmux session 前缀用 zhiyin-floor-(**不是** ocp-tui-),避免被 OCP 的
|
||||
# reaper 当成自己的会话杀掉,也避免我们杀到它的
|
||||
# - 用 real HOME(凭据)—— scratch HOME + symlink 凭据会 fork OAuth 导致 401
|
||||
# (见跨机记忆 tui_scratch_home_credential_fork)
|
||||
set -uo pipefail
|
||||
|
||||
N=${1:-5}
|
||||
MODEL=${MODEL:-claude-sonnet-5}
|
||||
EXTRA_ARGS=${EXTRA_ARGS:-} # 额外 CLI 参数(如 --effort low --bare)
|
||||
TAG=${TAG:-baseline}
|
||||
OUT=${OUT:-$(dirname "$0")/measurements.jsonl}
|
||||
PROMPT_FILE=$(mktemp)
|
||||
PREFIX="zhiyin-floor"
|
||||
|
||||
mkdir -p "$(dirname "$OUT")"
|
||||
|
||||
# ── 构造提示:~2000 token 的假会议转写 + 明确的起始标记 ────────────────
|
||||
# 单行(多行会在 tmux send-keys 时提前触发 Enter)
|
||||
build_prompt() {
|
||||
local seg="Speaker A said the quarterly pipeline is tracking behind plan and the enterprise segment needs a different motion. Speaker B replied that the current onboarding flow loses roughly a third of trial accounts before the first integration is complete. They debated whether the fix belongs in product or in customer success. "
|
||||
local body=""
|
||||
for _ in $(seq 1 22); do body+="$seg"; done
|
||||
printf '%s' "You are a real-time meeting copilot. Meeting transcript so far: $body --- Task: produce ONE prompt card as compact JSON with keys: points (array of 3 short Chinese bullet points), keyline (one English sentence the user can read aloud). IMPORTANT: your reply MUST begin with three hash characters immediately followed by the uppercase word CARD (no space between them), then the JSON. No preamble, no markdown fences." > "$PROMPT_FILE"
|
||||
}
|
||||
build_prompt
|
||||
PROMPT_CHARS=$(wc -c < "$PROMPT_FILE" | tr -d ' ')
|
||||
|
||||
now_ms() { python3 -c 'import time;print(int(time.time()*1000))'; }
|
||||
|
||||
echo "配置: $TAG 参数: [$EXTRA_ARGS]"
|
||||
echo "模型: $MODEL 样本: $N 提示长度: ${PROMPT_CHARS} chars (≈$((PROMPT_CHARS/4)) token)"
|
||||
echo "输出: $OUT"
|
||||
echo
|
||||
|
||||
for i in $(seq 1 "$N"); do
|
||||
SESS="${PREFIX}-$$-$i"
|
||||
SID=$(uuidgen)
|
||||
|
||||
# ── 冷启动:spawn + 等输入框就绪 ─────────────────────────────────
|
||||
T_SPAWN=$(now_ms)
|
||||
tmux new-session -d -s "$SESS" -x 200 -y 50 \
|
||||
-e CLAUDE_CODE_DISABLE_CLAUDE_MDS=1 \
|
||||
-e CLAUDE_CODE_DISABLE_AUTO_MEMORY=1 \
|
||||
-c "$HOME" \
|
||||
"claude --model $MODEL --session-id $SID --strict-mcp-config --disallowedTools 'mcp__*' $EXTRA_ARGS" 2>/dev/null
|
||||
if [ $? -ne 0 ]; then echo "[$i] tmux spawn 失败,跳过"; continue; fi
|
||||
|
||||
# 轮询输入框就绪(claude TUI 的输入提示符)
|
||||
READY=0
|
||||
for _ in $(seq 1 150); do # 上限 15s
|
||||
PANE=$(tmux capture-pane -p -t "$SESS" 2>/dev/null || true)
|
||||
if grep -qE '│ >|❯|Try "' <<<"$PANE"; then READY=1; break; fi
|
||||
sleep 0.1
|
||||
done
|
||||
T_READY=$(now_ms)
|
||||
BOOT_MS=$((T_READY - T_SPAWN))
|
||||
if [ "$READY" -ne 1 ]; then
|
||||
echo "[$i] 启动超时(${BOOT_MS}ms),pane 末 3 行:"
|
||||
tmux capture-pane -p -t "$SESS" 2>/dev/null | tail -3 | sed 's/^/ /'
|
||||
tmux kill-session -t "$SESS" 2>/dev/null
|
||||
continue
|
||||
fi
|
||||
|
||||
# ── 热态:粘提示 → 回车 → 量首 token ─────────────────────────────
|
||||
tmux send-keys -t "$SESS" -l "$(cat "$PROMPT_FILE")" 2>/dev/null
|
||||
sleep 0.4 # 让粘贴落地(OCP 用 400ms 轮询粒度)
|
||||
T0=$(now_ms)
|
||||
tmux send-keys -t "$SESS" Enter 2>/dev/null
|
||||
|
||||
TTFT_MS=-1
|
||||
for _ in $(seq 1 600); do # 上限 60s
|
||||
if tmux capture-pane -p -t "$SESS" 2>/dev/null | grep -q '###CARD'; then
|
||||
TTFT_MS=$(( $(now_ms) - T0 )); break
|
||||
fi
|
||||
sleep 0.1
|
||||
done
|
||||
|
||||
# ── 完整回答:pane 连续 2s 不再变化 ──────────────────────────────
|
||||
COMPLETE_MS=-1
|
||||
if [ "$TTFT_MS" -ge 0 ]; then
|
||||
LAST=""; STABLE=0
|
||||
for _ in $(seq 1 900); do # 上限 90s
|
||||
CUR=$(tmux capture-pane -p -t "$SESS" 2>/dev/null | cksum)
|
||||
if [ "$CUR" = "$LAST" ]; then
|
||||
STABLE=$((STABLE+1))
|
||||
[ "$STABLE" -ge 20 ] && { COMPLETE_MS=$(( $(now_ms) - T0 - 2000 )); break; }
|
||||
else
|
||||
STABLE=0; LAST="$CUR"
|
||||
fi
|
||||
sleep 0.1
|
||||
done
|
||||
fi
|
||||
|
||||
printf '{"i":%d,"tag":"%s","model":"%s","extra_args":"%s","prompt_chars":%s,"boot_ms":%d,"ttft_ms":%d,"complete_ms":%d}\n' \
|
||||
"$i" "$TAG" "$MODEL" "$EXTRA_ARGS" "$PROMPT_CHARS" "$BOOT_MS" "$TTFT_MS" "$COMPLETE_MS" | tee -a "$OUT"
|
||||
|
||||
tmux kill-session -t "$SESS" 2>/dev/null
|
||||
sleep 1
|
||||
done
|
||||
|
||||
rm -f "$PROMPT_FILE"
|
||||
echo
|
||||
echo "=== 汇总 ==="
|
||||
python3 - "$OUT" <<'EOF'
|
||||
import json,sys,statistics
|
||||
rows=[json.loads(l) for l in open(sys.argv[1]) if l.strip()]
|
||||
ok=[r for r in rows if r['ttft_ms']>=0]
|
||||
if not ok: print("无有效样本"); sys.exit()
|
||||
def s(k):
|
||||
v=[r[k] for r in ok if r[k]>=0]
|
||||
return f"n={len(v)} 中位={statistics.median(v)/1000:.2f}s 最小={min(v)/1000:.2f}s 最大={max(v)/1000:.2f}s" if v else "无"
|
||||
print(f" 冷启动 boot : {s('boot_ms')} ← 预热进程池可完全消除")
|
||||
print(f" TTFT(首 token) : {s('ttft_ms')} ★ 这就是地板")
|
||||
print(f" 完整回答 : {s('complete_ms')}")
|
||||
print(f"\n 失败样本: {len(rows)-len(ok)}/{len(rows)}")
|
||||
EOF
|
||||
@@ -0,0 +1,15 @@
|
||||
{"i": 1, "tag": "effort-low", "model": "claude-sonnet-5", "extra_args": "--effort low", "prompt_chars": 7451, "boot_ms": 1077, "ttft_ms": 6172, "complete_ms": 9929}
|
||||
{"i": 2, "tag": "effort-low", "model": "claude-sonnet-5", "extra_args": "--effort low", "prompt_chars": 7451, "boot_ms": 1026, "ttft_ms": 6160, "complete_ms": 9996}
|
||||
{"i": 3, "tag": "effort-low", "model": "claude-sonnet-5", "extra_args": "--effort low", "prompt_chars": 7451, "boot_ms": 1010, "ttft_ms": 6437, "complete_ms": 9977}
|
||||
{"i": 4, "tag": "effort-low", "model": "claude-sonnet-5", "extra_args": "--effort low", "prompt_chars": 7451, "boot_ms": 1033, "ttft_ms": 5872, "complete_ms": 9944}
|
||||
{"i": 5, "tag": "effort-low", "model": "claude-sonnet-5", "extra_args": "--effort low", "prompt_chars": 7451, "boot_ms": 1154, "ttft_ms": 6387, "complete_ms": 9993}
|
||||
{"i":1,"tag":"baseline","model":"claude-sonnet-5","extra_args":"","prompt_chars":7451,"boot_ms":1300,"ttft_ms":8321,"complete_ms":9939}
|
||||
{"i":2,"tag":"baseline","model":"claude-sonnet-5","extra_args":"","prompt_chars":7451,"boot_ms":1070,"ttft_ms":10347,"complete_ms":11320}
|
||||
{"i":3,"tag":"baseline","model":"claude-sonnet-5","extra_args":"","prompt_chars":7451,"boot_ms":911,"ttft_ms":13061,"complete_ms":15163}
|
||||
{"i":4,"tag":"baseline","model":"claude-sonnet-5","extra_args":"","prompt_chars":7451,"boot_ms":1441,"ttft_ms":9981,"complete_ms":11066}
|
||||
{"i":5,"tag":"baseline","model":"claude-sonnet-5","extra_args":"","prompt_chars":7451,"boot_ms":1036,"ttft_ms":17189,"complete_ms":17985}
|
||||
{"i":1,"tag":"bare","model":"claude-sonnet-5","extra_args":"--bare","prompt_chars":7451,"boot_ms":429,"ttft_ms":-1,"complete_ms":-1}
|
||||
{"i":2,"tag":"bare","model":"claude-sonnet-5","extra_args":"--bare","prompt_chars":7451,"boot_ms":437,"ttft_ms":-1,"complete_ms":-1}
|
||||
{"i":3,"tag":"bare","model":"claude-sonnet-5","extra_args":"--bare","prompt_chars":7451,"boot_ms":444,"ttft_ms":-1,"complete_ms":-1}
|
||||
{"i":4,"tag":"bare","model":"claude-sonnet-5","extra_args":"--bare","prompt_chars":7451,"boot_ms":446,"ttft_ms":-1,"complete_ms":-1}
|
||||
{"i":5,"tag":"bare","model":"claude-sonnet-5","extra_args":"--bare","prompt_chars":7451,"boot_ms":441,"ttft_ms":-1,"complete_ms":-1}
|
||||
@@ -0,0 +1,7 @@
|
||||
{"hook_event_name": "MessageDisplay", "index": 0, "final": false, "delta": "## Mutex\n\n"}
|
||||
{"hook_event_name": "MessageDisplay", "index": 1, "final": false, "delta": "A **mutual exclusion lock** prevents concurrent access to a shared resource, ensuring only one thread runs the critical section at a time.\n\n"}
|
||||
{"hook_event_name": "MessageDisplay", "index": 2, "final": false, "delta": "- Acquiring a locked mutex blocks the caller until the current holder releases it.\n"}
|
||||
{"hook_event_name": "MessageDisplay", "index": 3, "final": false, "delta": "- Failing to release a mutex causes a deadlock, freezing all waiting threads.\n\n```javascript\nconst { Mutex } = require('async-mutex');\n\nconst mutex = new Mutex();\n"}
|
||||
{"hook_event_name": "MessageDisplay", "index": 4, "final": false, "delta": "let counter = 0;\n\nasync function increment() {\n const release = await mutex.acquire();\n try {\n"}
|
||||
{"hook_event_name": "MessageDisplay", "index": 5, "final": false, "delta": " counter++; // only one caller here at a time\n } finally {\n release();\n }\n}\n"}
|
||||
{"hook_event_name": "MessageDisplay", "index": 6, "final": true, "delta": "```"}
|
||||
@@ -0,0 +1,5 @@
|
||||
{"i":1,"tag":"effort-low-exclude-dynamic","model":"claude-sonnet-5","extra_args":"--effort low --exclude-dynamic-system-prompt-sections","prompt_chars":7451,"boot_ms":934,"ttft_ms":5867,"complete_ms":9953}
|
||||
{"i":2,"tag":"effort-low-exclude-dynamic","model":"claude-sonnet-5","extra_args":"--effort low --exclude-dynamic-system-prompt-sections","prompt_chars":7451,"boot_ms":1275,"ttft_ms":6388,"complete_ms":9874}
|
||||
{"i":3,"tag":"effort-low-exclude-dynamic","model":"claude-sonnet-5","extra_args":"--effort low --exclude-dynamic-system-prompt-sections","prompt_chars":7451,"boot_ms":874,"ttft_ms":10537,"complete_ms":11782}
|
||||
{"i":4,"tag":"effort-low-exclude-dynamic","model":"claude-sonnet-5","extra_args":"--effort low --exclude-dynamic-system-prompt-sections","prompt_chars":7451,"boot_ms":1170,"ttft_ms":6379,"complete_ms":9947}
|
||||
{"i":5,"tag":"effort-low-exclude-dynamic","model":"claude-sonnet-5","extra_args":"--effort low --exclude-dynamic-system-prompt-sections","prompt_chars":7451,"boot_ms":1329,"ttft_ms":6443,"complete_ms":9884}
|
||||
@@ -0,0 +1,256 @@
|
||||
# Backlog #2 (real streaming): **achievable** — via the `MessageDisplay` hook
|
||||
|
||||
**Date**: 2026-07-13
|
||||
**Status**: prereq-spike result. **Streaming IS achievable on the TUI path**, byte-faithfully, on the
|
||||
subscription pool. Three obvious sources are dead ends; a fourth one works.
|
||||
**Scope**: answers the prereq spike that [`README.md`](README.md) § "Backlog #2" demanded *before* any
|
||||
streaming design:
|
||||
|
||||
> **Prereq spike (do this before designing anything)**: does the transcript JSONL grow *during* a
|
||||
> turn, or only at the end? If only at the end, (a) is dead and you are stuck with (b).
|
||||
|
||||
The answer: **(a) is dead, (b) is dead — and you are not stuck with either.** The CLI exposes its own
|
||||
streaming interface as a **hook**, which the backlog did not consider.
|
||||
|
||||
**Measured on**: Mac mini / Claude Code **v2.1.207** / Sonnet 4.6 + Sonnet 5 / Claude Max /
|
||||
real-home mode. Every claim below is reproducible from the commands given.
|
||||
|
||||
> **Honesty note on how this document was produced.** Its first version concluded the exact opposite —
|
||||
> "streaming is not achievable; the CLI exposes no byte-faithful incremental source" — and was **wrong**.
|
||||
> An adversarial reviewer, commissioned specifically to *refute* it, found `MessageDisplay` on a second
|
||||
> pass; its own first pass had enumerated the hook registry with a truncated grep (it reported 21
|
||||
> events — there are **30**). Both the wrong conclusion and its refutation are preserved here, because
|
||||
> "we checked, it's impossible" is the most expensive kind of claim to get wrong: it closes a door and
|
||||
> nobody re-opens it.
|
||||
|
||||
---
|
||||
|
||||
## ✅ The source that works: the `MessageDisplay` hook
|
||||
|
||||
`claude` fires a **`MessageDisplay`** hook as it renders each block of the assistant's reply. The
|
||||
payload carries the **raw markdown source** of an incremental `delta`, plus a monotonic `index` and a
|
||||
`final` flag:
|
||||
|
||||
```json
|
||||
{ "hook_event_name": "MessageDisplay",
|
||||
"turn_id": "6cb31d21-…", "message_id": "84ab9832-…",
|
||||
"index": 0, "final": false, "delta": "## Mutex\n\n" }
|
||||
```
|
||||
*(payload also carries `session_id`, `transcript_path`, `prompt_id`, `cwd`)*
|
||||
|
||||
Registered as an ordinary command hook via `--settings` on a **plain interactive TUI spawn** (no `-p`,
|
||||
no `--bare`), `claude-sonnet-4-6`, `--effort low`. Banner verified:
|
||||
`▝▜█████▛▘ Sonnet 4.6 with low effort · Claude Max` — **subscription pool, not metered billing**.
|
||||
|
||||
One live turn — 7 fires, spread across generation:
|
||||
|
||||
```
|
||||
index=0 final=false len= 10 '## Mutex\n\n'
|
||||
index=1 final=false len= 140 'A **mutual exclusion lock** prevents concurrent access to a shar…'
|
||||
index=2 final=false len= 83 '- Acquiring a locked mutex blocks the caller until the current h…'
|
||||
index=3 final=false len= 163 '- Failing to release a mutex causes a deadlock, freezing all wai…'
|
||||
index=4 final=false len= 96 'let counter = 0;\n\nasync function increment() {\n const release =…'
|
||||
index=5 final=false len= 84 ' counter++; // only one caller here at a time\n } finally {\n …'
|
||||
index=6 final=true len= 3 '```'
|
||||
```
|
||||
|
||||
**Every invariant a proxy needs — all hold:**
|
||||
|
||||
| requirement | result |
|
||||
|---|---|
|
||||
| **byte-faithful** — deltas are the model's *source*, not the rendered pane | ✅ `## `, `**`, ```` ```javascript ```` all present in the deltas |
|
||||
| **exactness** — `concat(deltas) === T` (the transcript-authoritative text) | ✅ **true**, 579 == 579 bytes |
|
||||
| **prefix-stable** — `T.startsWith(concat(deltas[0..n]))` at every n | ✅ **true at all 7 steps** |
|
||||
| **incremental** — arrives during generation, not at the end | ✅ 7 fires spread across the turn |
|
||||
| **no `-p`** — stays out of the metered `sdk-cli` pool | ✅ plain interactive TUI |
|
||||
| **subscription pool** | ✅ banner `· Claude Max` |
|
||||
|
||||
This is exactly the contract a streaming design needs: deltas forward straight into SSE
|
||||
`delta.content` chunks, and the transcript's final text `T` stays a cheap end-of-turn assertion
|
||||
(`concat === T`) instead of a reconciliation problem.
|
||||
|
||||
### Caveats for the implementer
|
||||
|
||||
- **Block-level granularity, not token-level** — the hook fires **once per rendered block** (roughly one
|
||||
per paragraph / list item / code block), so the chunk count **scales with answer length**: 7 fires for a
|
||||
~600-byte answer, **18 for a ~2 KB one**. Plenty for SSE (`delta.content` has no minimum size), but do
|
||||
not promise token-by-token output, and do not hard-code any assumption about chunk count.
|
||||
- **🔴 The sink MUST be keyed by `session_id` — this is live TODAY, not a future concern.**
|
||||
`OCP_TUI_MAX_CONCURRENT` defaults to **2**, so **two `claude` processes already run concurrently**. One
|
||||
hook command writing to one shared sink would **interleave deltas from two different turns into one
|
||||
stream** — request A's client receiving request B's text, the worst failure a proxy can have, and one a
|
||||
single-request test will never surface. The payload carries `session_id` (and `turn_id` / `message_id`),
|
||||
so demux is easy: derive the sink path from `session_id` (`<dir>/<session_id>.jsonl`) and read only your
|
||||
own turn's file. This *also* keeps the design **warm-pool compatible**, because a pre-booted pane's
|
||||
session-id is fixed at boot — one static hook script serves every pane. **Test it with ≥2 concurrent
|
||||
streaming requests carrying distinguishable prompts and assert zero cross-contamination.**
|
||||
- **⚠️ `forceSyncExecution: true` in the hook's source — `claude` BLOCKS on the hook.** A slow hook
|
||||
adds latency to *every* delta. The hook must write and exit immediately (e.g. write to a FIFO / unix
|
||||
socket that OCP reads; never work inline). **Measure the added per-delta latency.**
|
||||
- **Thinking blocks appear to be excluded — but this is NOT yet stress-tested. Verify before shipping.**
|
||||
The exclusion is inferred from `content.map(c => c.type === "text" ? c.text : "")` — but that snippet is
|
||||
from the **`final:true`** call site, not the incremental one. Four live turns (incl. two at `--effort
|
||||
high`) showed no thinking text in any delta and `concat === T` held — **but each transcript's thinking
|
||||
block was empty (`thinking:""`, 0 chars)**, so the exclusion was never actually stressed. **The failure
|
||||
mode is severe**: if thinking deltas *do* fire on some config (Opus, `xhigh`), `concat(deltas) !== T`
|
||||
**and OCP streams the model's private reasoning to the caller**. The end-of-turn `concat === T` assertion
|
||||
would *detect* that but **cannot prevent** it — SSE deltas cannot be un-sent. **Before shipping, run a
|
||||
turn on a model+effort that produces substantive thinking** (a hard reasoning prompt on Opus / `xhigh`)
|
||||
and confirm both (a) no thinking text in any delta and (b) `concat === T` still holds.
|
||||
- OCP already owns the spawn (isolated HOME, its own flags), so injecting `--settings` with a
|
||||
`MessageDisplay` hook sits inside the existing architecture.
|
||||
- **`ALIGNMENT.md`**: this consumes `claude`'s **own** hook surface as emitted — forwarding, not
|
||||
inventing. Not a new endpoint, not a fabricated protocol. (Class B / ADR 0007 — the TUI spawn is
|
||||
OCP-owned; no `cli.js` citation applies.)
|
||||
|
||||
### Reproduce in 60 seconds
|
||||
|
||||
```bash
|
||||
# hook script: append the payload (arrives on stdin) and exit immediately
|
||||
printf '#!/bin/bash\ncat >> "$MD_LOG"; printf "\\n" >> "$MD_LOG"; exit 0\n' > /tmp/h.sh && chmod +x /tmp/h.sh
|
||||
echo '{"hooks":{"MessageDisplay":[{"hooks":[{"type":"command","command":"MD_LOG=/tmp/deltas.jsonl /tmp/h.sh"}]}]}}' > /tmp/s.json
|
||||
|
||||
# plain interactive claude in tmux (prefix NOT ocp-tui-*, and never kill-server)
|
||||
tmux new-session -d -s md-probe -x 220 -y 50 \
|
||||
"claude --model claude-sonnet-4-6 --effort low --session-id $(uuidgen) --settings /tmp/s.json"
|
||||
# …wait for '? for shortcuts', paste a markdown-producing prompt, press Enter…
|
||||
|
||||
jq -r '"\(.index) \(.final) \(.delta|@json)"' /tmp/deltas.jsonl # incremental raw-markdown deltas
|
||||
# then assert: concat(deltas) == extractLatestAssistantText(<transcript>.jsonl)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## The three dead ends (still worth knowing — they say what NOT to build)
|
||||
|
||||
### (a) Incremental transcript reads — **dead: event granularity, not token granularity**
|
||||
|
||||
The transcript JSONL *does* grow during a turn, but one **whole event at a time**; the assistant's text
|
||||
event is written as **one complete line**, appearing only ~0.3 s before the terminal `turn_duration`.
|
||||
|
||||
Observed (session `efd5b161`, `turn_duration: 7319 ms`):
|
||||
|
||||
```
|
||||
#6 t+0.0s type=user (the prompt)
|
||||
#15 t+4.7s type=assistant blocks=thinking
|
||||
#16 t+7.0s type=assistant blocks=text ← the ENTIRE answer, in one line
|
||||
#21 t+7.3s type=system subtype=turn_duration ← terminal
|
||||
```
|
||||
|
||||
Cross-checked at **20 ms polling + `fs.watch`** (25× finer): a partial line **never touches disk** —
|
||||
one write, `+1` line, carrying the complete answer. Also forced with the undocumented
|
||||
`CLAUDE_CODE_INCLUDE_PARTIAL_MESSAGES=1`: still 1 assistant event, 0 partials (interactive mode has no
|
||||
stream-json *sink* for it to write to).
|
||||
|
||||
**The transcript is still needed** — as the terminal-turn signal, as the authoritative `concat === T`
|
||||
check, and as the input to the existing honesty gates (auth-banner detection, `truncated`). It is just
|
||||
not the *streaming* source.
|
||||
|
||||
### (b) `tmux capture-pane` diffing — **dead: the pane is a RENDERED view, not the text**
|
||||
|
||||
The backlog expected to fall back to this, calling it "lossy … (wrapping, scrollback, spinner lines)".
|
||||
The loss is far worse than formatting noise: **the pane does not contain the answer's source bytes at
|
||||
all.** The TUI *renders* markdown, and `capture-pane -p` strips the ANSI that rendering produced.
|
||||
|
||||
Same turn, same lines:
|
||||
|
||||
```
|
||||
TRANSCRIPT (authoritative T): PANE (capture-pane -p -J -S -500):
|
||||
'## Semaphore' '⏺ Semaphore' ← heading marker gone
|
||||
'' ''
|
||||
'A **semaphore** is a synchro…' ' A semaphore is a synchro…' ← bold markers gone, indented
|
||||
```
|
||||
|
||||
| token in the answer | in `T` | in the pane's answer region |
|
||||
|---|---|---|
|
||||
| `## ` (ATX heading) | yes | **no** — rendered as `⏺` |
|
||||
| `**` (bold markers) | yes | **no** — rendered to ANSI bold, then stripped by `-p` |
|
||||
| ` ```javascript ` (fence + language) | yes | **no** — fence and language tag both gone |
|
||||
| `- ` (list item) | yes | yes |
|
||||
|
||||
*(A literal `**` does appear elsewhere in the pane — in the **prompt echo**, because the prompt asked
|
||||
for bold. Not in the answer.)*
|
||||
|
||||
**`capture-pane -e` (keeping the ANSI) does not rescue it — the inverse is provably non-unique.**
|
||||
With `T` = ``"## Alpha\n\n**bravo**\n\n```javascript\nlet x=1;\n```"``:
|
||||
|
||||
```
|
||||
⏺\e[39m \e[1mAlpha\n\n\e[0m \e[1mbravo\n\n\e[0m \e[34mlet\e[39m x=\e[32m1\e[39m;
|
||||
```
|
||||
|
||||
`## Alpha` → **SGR 1 (bold)**. `**bravo**` → **SGR 1 (bold)**. *Identical ANSI* — an H2 and a bold span
|
||||
are indistinguishable, never mind `**` vs `__`. The fence and its `javascript` tag are consumed by the
|
||||
syntax highlighter into colours; recovering the tag would mean inverting a highlighter, and
|
||||
`let x=1;` is valid in several languages.
|
||||
|
||||
So `T.startsWith(paneText)` is **false** — raw and indent-stripped, on essentially every markdown
|
||||
answer. A proxy streaming pane text would be streaming **something the model did not say**. With
|
||||
`MessageDisplay` available there is no reason to go near it.
|
||||
|
||||
### (c) `--debug-file` — **dead: it logs stream *timing*, never stream *content***
|
||||
|
||||
Worth stating precisely, because a casual check misleads in **both** directions here.
|
||||
|
||||
The default log level is `debug`, which **suppresses every `verbose` site**. Raise it and per-chunk
|
||||
lines *do* appear, spread across generation:
|
||||
|
||||
```bash
|
||||
CLAUDE_CODE_DEBUG_LOG_LEVEL=verbose claude --debug-file /tmp/d.log …
|
||||
```
|
||||
```
|
||||
05:51:11.088 [VERBOSE] [shoji-engine] yield stream_event/- ← 16 of these, mid-turn,
|
||||
05:51:11.537 [VERBOSE] [shoji-engine] yield stream_event/- over ~3.9 s of generation
|
||||
05:51:15.192 [DEBUG] [shoji-engine] turn 1 end (usage in=575 out=255 api=6736ms stop=end_turn resultLen=857)
|
||||
```
|
||||
|
||||
**But they carry no payload** — the format is `yield <type>/<subtype>`, a bare presence marker. Run with
|
||||
no category filter (i.e. all categories) at verbose level: `content_block_delta` = **0**, `text_delta` =
|
||||
**0**, `content_block_start` / `message_start` = **0**. The only byte-exact text in the log is the
|
||||
end-of-turn `Stop` hook payload (`"last_assistant_message":"## Title\n\n**alpha bravo charlie**"`) —
|
||||
transcript granularity. The log tells you **when** tokens arrive, never **what** they are. It is also
|
||||
~2.7 MB per turn.
|
||||
|
||||
### Also checked, also not the answer
|
||||
|
||||
| candidate | outcome |
|
||||
|---|---|
|
||||
| `--output-format stream-json` (the one interface that emits `text_delta`) | **requires `--print`/`-p`** → `cc_entrypoint=sdk-cli` → the **metered** credit pool, which is exactly what TUI mode exists to avoid. Reproduced live. |
|
||||
| `--input-format stream-json` | `Error: --input-format=stream-json requires output-format=stream-json` → same gate. |
|
||||
| `CLAUDE_CODE_INCLUDE_PARTIAL_MESSAGES=1` (undocumented) | No stream-json sink in interactive mode → no partials. Banner stayed `· Claude Max`. |
|
||||
| `sessionMirror` (undocumented) | Gated on `outputFormat === "stream-json"` → the `-p` family. |
|
||||
| `--sdk-url` (hidden) | Forces stream-json + non-interactive → `sdk-cli`. *(inferred from the minified bundle; not banner-tested)* |
|
||||
| `~/.claude/sessions/<pid>.json` | Registry metadata only (`{pid, sessionId, cwd, status, version, entrypoint:"cli", kind:"interactive"}`). No assistant text. *(Its `entrypoint:"cli"` incidentally confirms the TUI path stays on the subscription pool.)* |
|
||||
| `~/.claude/history.jsonl` | User prompts only; the answer text is absent. |
|
||||
| Asking the model to emit plain text (so the pane renders faithfully) | Would mean **mutating the caller's prompt** — a correctness violation for a proxy, and still not byte-faithful (wrapping + indent remain). Rejected. |
|
||||
|
||||
---
|
||||
|
||||
## Value: what streaming actually buys (read before building)
|
||||
|
||||
Streaming is *possible*. Whether it is *worth it* depends on the consumer, and the honest answer is
|
||||
uncomfortable:
|
||||
|
||||
- **Streaming never makes the answer arrive sooner. It moves the *first* byte, not the *last*.** The
|
||||
final token lands at the same wall-clock moment either way.
|
||||
- So a consumer that must have the **complete** answer before it can act — e.g. one parsing a structured
|
||||
JSON reply, **which is exactly the 知音 AI use case that motivated this entire investigation** — gains
|
||||
**nothing at all**. Only a **progressively-rendering** consumer (a chat UI) gains.
|
||||
|
||||
And the number the backlog attached to this item was wrong:
|
||||
|
||||
- The backlog's "~20 s" was inferred from an external 30–32 s report, **never measured through OCP**.
|
||||
Measured through a real OCP instance (TUI mode, `claude-sonnet-4-6`, ~1850-token prompt, n=5):
|
||||
**median 11.30 s** before [#156](https://github.com/dtzp555-max/ocp/pull/156), **9.55 s** after.
|
||||
- **Same-turn decomposition** (baseline row `i=5`): **11.563 s** wall through OCP vs `turn_duration:
|
||||
7.319 s` of CLI-internal time on that same turn → **OCP's own overhead ≈ 4.2 s** (n=1, baseline
|
||||
`effort=high` config). *Caveats*: n=1; and `turn_duration` is the CLI's internal duration of an
|
||||
**OCP-driven** turn, not a separate "native" baseline. Do **not** subtract this `effort=high` 7.3 s
|
||||
from the `effort=low` 9.55 s median — a low-effort turn generates faster, so mixing them
|
||||
*understates* the overhead.
|
||||
- So OCP's own overhead is **single-digit seconds**, not ~20 s. The rest of any large number is the
|
||||
model generating a long answer — which streaming hides but does not shorten.
|
||||
|
||||
**Recommendation**: build it — the contract is clean and the cost is small — but size the expectation
|
||||
honestly. It is a *perceived-latency* feature for progressively-rendering consumers, not a throughput
|
||||
win, and it does not move the **~6 s TTFT floor** ([`README.md`](README.md)) that rules TUI mode out for
|
||||
interactive-latency consumers regardless.
|
||||
@@ -0,0 +1,268 @@
|
||||
# OCP Anthropic-Only Sandbox Strategy — Handoff Document
|
||||
|
||||
**Status:** Forward-looking planning doc (not yet a decision)
|
||||
**Date:** 2026-05-29
|
||||
**Audience:** future OCP maintainer / session picking up multi-tenant security work
|
||||
**Provenance:** authored during OLP Phase 7 PR-B re-evaluation; OLP's parallel analysis (multi-provider) lives at `dtzp555-max/olp` `docs/adr/0014-sandbox-runtime-integration.md` Amendment 1 (pending). This OCP-side doc strips the multi-LLM generalization and keeps only what applies to OCP's single-provider (anthropic) deployment.
|
||||
|
||||
---
|
||||
|
||||
## 1. Why this doc exists
|
||||
|
||||
OCP is in maintenance mode (per OLP ADR 0001 supersession of OCP ADR 0005). It is not under active development for new features. However, two things may eventually drive sandbox work in OCP:
|
||||
|
||||
1. **Multi-key OCP deployments.** `OCP_OWNER_TOKEN` + per-key cache namespace already shipped (OCP `lib/keys.mjs`). If multiple human users share an OCP instance, the same multi-tenant filesystem-isolation gap that motivated OLP Phase 7 also exists here.
|
||||
2. **Cloud or shared-host OCP deployments.** Any deployment beyond "single user on their own machine" inherits the threat surface.
|
||||
|
||||
If/when that work starts, this doc is the prior-art capture so the maintainer doesn't repeat OLP's PR-B path (which has a documented dead-end — see § 3.2 below).
|
||||
|
||||
This doc is anthropic-only by design — codex/mistral/etc. multi-LLM concerns are out of scope per OCP ADR 0005.
|
||||
|
||||
---
|
||||
|
||||
## 2. The multi-tenant gap (OCP-specific)
|
||||
|
||||
OCP spawns `claude -p` as the OCP-process user. Every spawned claude instance runs with the OCP user's filesystem permissions. Consequences for a multi-key OCP deployment:
|
||||
|
||||
1. **Cross-key lateral read.** A prompt-injected `cat ~/.ocp/keys/<other-key>.json` reads any other key's manifest (token hash, owner_tier, providers_enabled — not catastrophic since it's only the *hash*, but still identity-attribution surface).
|
||||
2. **OAuth credential exposure.** `~/.claude/.credentials.json` is the Anthropic OAuth refresh token. A prompt-injected read of this file = stealing the subscription that OCP exists to pool.
|
||||
3. **SSH identity exposure.** `~/.ssh/id_*` reachable for lateral movement to other hosts the OCP user can reach.
|
||||
4. **Other host secrets.** Anything else under the OCP user's home is reachable.
|
||||
|
||||
OCP's `ALIGNMENT.md` Class A/B endpoint discipline does not address this — that discipline is wire-level honesty (`cli.js` mirror), not host-level isolation.
|
||||
|
||||
The threat model assumes prompt-injection capability — any caller with a valid OCP key + ability to craft a prompt that elicits a tool call. Default `claude -p` mode includes Read/Bash/etc. tool descriptions in the system prompt; the model is **eager** to use them.
|
||||
|
||||
---
|
||||
|
||||
## 3. Why OLP Phase 7 PR-B is the wrong path to copy
|
||||
|
||||
OLP attempted to wrap `claude -p` spawn in `@anthropic-ai/sandbox-runtime` (outer bubblewrap on Linux, sandbox-exec on macOS). This produced four binding problems documented during OLP's re-evaluation:
|
||||
|
||||
### 3.1 Anthropic's design doesn't expect external sandboxing
|
||||
|
||||
Per Anthropic's [engineering blog on Claude Code sandboxing](https://www.anthropic.com/engineering/claude-code-sandboxing), `sandbox-runtime` is designed to be invoked **by claude code itself** to sandbox **its own** Bash tool / MCP servers / spawn children. It is **not** designed to sandbox claude code as an externally-wrapped process.
|
||||
|
||||
Concretely: claude CLI assumes it can freely read+write its own `$HOME`-derived paths (`~/.claude.json`, `~/.claude/.credentials.json`, `~/.config/claude/`, future state files). When wrapped in `bwrap --ro-bind / /`, those writes hit `EROFS` and claude silently exits with no stdout.
|
||||
|
||||
### 3.2 `~/.claude.json` upstream status is "closed not planned"
|
||||
|
||||
claude CLI writes `~/.claude.json` non-atomically at startup. Upstream issues #28842, #29162, #29217, #28837, #29051, #29250, #7243 all document this. **#29250 is closed as "not planned / duplicate"** — Anthropic is not going to make this file atomic-write because their mental model is that claude runs in an environment that can write its `$HOME`.
|
||||
|
||||
For OCP, this means: any outer-sandbox approach that uses `--ro-bind` on `$HOME` will be a **permanent maintenance treadmill** — every new claude CLI version that adds a state file outside the patched mount paths breaks OCP. OLP's PR-B fold-in tried to patch this by promoting `~/.claude/` to rw, which was insufficient (the actual file is `~/.claude.json` at $HOME root, not inside `~/.claude/`).
|
||||
|
||||
### 3.3 The threat model doesn't justify the cost
|
||||
|
||||
OCP is, per ADR 0005, a personal-and-family-scale tool. The realistic threat surface is misbehaving prompts from family members or self-injected via dependent agents, not adversarial external attackers. The blast radius of a successful cross-key read is bounded (token *hash*, OAuth that's pooled-by-design across all OCP keys).
|
||||
|
||||
A maintenance-mode project investing weeks into outer-sandboxing for a hypothetical threat is a poor cost/benefit. There are cheaper architectures (§ 4 below) that get most of the protection.
|
||||
|
||||
### 3.4 OLP-specific reason that does NOT apply to OCP
|
||||
|
||||
OLP also hit a multi-provider conflict: codex CLI has its own inner bubblewrap that breaks when wrapped in an outer bwrap (openai/codex#16018). **This is not an OCP concern** — OCP only spawns claude. So the multi-provider forcing function for OLP doesn't apply here. The other three reasons (§ 3.1–3.3) are sufficient on their own.
|
||||
|
||||
---
|
||||
|
||||
## 4. Three viable approaches for OCP
|
||||
|
||||
Ranked by "engineering cost vs isolation strength" — pick by deployment context.
|
||||
|
||||
### 4.1 Approach A — Ephemeral `$HOME` via env var (recommended starting point)
|
||||
|
||||
Per-spawn setup:
|
||||
|
||||
```
|
||||
ephemeralRoot=/tmp/ocp-spawn/<keyId>/<reqId>/home
|
||||
mkdir -p $ephemeralRoot/.claude
|
||||
ln -s ~/.claude/.credentials.json $ephemeralRoot/.claude/.credentials.json
|
||||
HOME=$ephemeralRoot claude -p --output-format stream-json ...
|
||||
```
|
||||
|
||||
Mechanics:
|
||||
- claude CLI uses Node's `os.homedir()` which reads `$HOME` env first.
|
||||
- `~/.claude.json` written by claude on startup → lands in `/tmp/ocp-spawn/<keyId>/<reqId>/home/.claude.json` (tmpfs, discarded after spawn).
|
||||
- `~/.claude/.credentials.json` is the OAuth file claude needs — symlinked in read-only from the real one.
|
||||
- Any new state file claude CLI introduces in a future version → also lands in the ephemeral home, no patch needed.
|
||||
|
||||
Threat coverage:
|
||||
- ✅ Solves EROFS upgrade tax permanently — any claude state-file location works because they all land in tmpfs.
|
||||
- ✅ Cross-key OAuth credential isolation — keyA's ephemeral home has only keyA's symlink, but here the symlink target is the SAME real file because OCP shares OAuth (this is fine: shared OAuth is OCP's design, the symlink just keeps the file inaccessible via `cat ~/.claude/.credentials.json` from a different keyId's ephemeral root).
|
||||
- ❌ Does NOT solve cross-key lateral filesystem read via absolute paths. A prompt-injected `cat /home/<ocp-user>/.ocp/keys/<otherKey>.json` still works — `os.homedir()` override doesn't affect absolute-path reads.
|
||||
|
||||
5-minute spike before adopting:
|
||||
|
||||
```bash
|
||||
HOME=/tmp/fake-home-spike claude --print "echo PONG" --no-session-persistence 2>&1
|
||||
ls -la /tmp/fake-home-spike # expect: .claude.json + .claude/ created here
|
||||
find ~/.claude ~/.claude.json -newer /tmp/spike-marker 2>/dev/null # expect: empty
|
||||
```
|
||||
|
||||
If claude falls back to `os.userInfo().homedir` (uses getpwuid_r, ignores HOME env), this approach degrades — fall back to Approach B.
|
||||
|
||||
**Engineering cost:** ~50 LOC in OCP's spawn pipeline (mkdir + symlink + env merge + cleanup-on-exit). No new dependencies.
|
||||
|
||||
### 4.2 Approach B — Outer bubblewrap with `--tmpfs $HOME` + `--ro-bind` credentials
|
||||
|
||||
```
|
||||
bwrap \
|
||||
--ro-bind / / \
|
||||
--tmpfs /home/<ocp-user> \
|
||||
--ro-bind /home/<ocp-user>/.claude/.credentials.json /home/<ocp-user>/.claude/.credentials.json \
|
||||
--ro-bind /home/<ocp-user>/.ocp/keys/<thisKeyId>.json /home/<ocp-user>/.ocp/keys/<thisKeyId>.json \
|
||||
--dev /dev --proc /proc --tmpfs /tmp \
|
||||
claude -p ...
|
||||
```
|
||||
|
||||
This is the canonical bwrap pattern (Flatpak uses exactly this for every sandboxed app — see [Bubblewrap ArchWiki Examples](https://wiki.archlinux.org/title/Bubblewrap/Examples)).
|
||||
|
||||
Threat coverage:
|
||||
- ✅ Solves EROFS upgrade tax (tmpfs accepts any write path).
|
||||
- ✅ Cross-key lateral read prevention — only the current key's manifest is bind-mounted in, others are simply absent from the sandbox view.
|
||||
- ✅ `~/.ssh` and similar identity material absent from sandbox.
|
||||
|
||||
Trade-offs:
|
||||
- bwrap dependency: install `bubblewrap` apt package on host.
|
||||
- Bypasses `@anthropic-ai/sandbox-runtime` library — direct bwrap arg composition. Worth it because sandbox-runtime's outer-wrap design is for short-lived claude-internal subprocesses, not long-running claude CLI itself (per § 3.1).
|
||||
- macOS: not supported by bwrap (macOS would need separate `sandbox-exec` profile, ~50-100 LOC additional work). OCP cross-machine maintainer deploys mostly on Mac mini + Oracle ARM VM — both Linux on the cloud side, Mac mini side may remain unsandboxed if family-trust-zone.
|
||||
|
||||
**Engineering cost:** ~150 LOC for the spawn wrapper + deployment doc updates to require `apt install bubblewrap`. macOS support is a separate ~100 LOC if/when needed.
|
||||
|
||||
### 4.3 Approach C — OverlayFS lowerdir (read-only) + tmpfs upperdir (writable)
|
||||
|
||||
```
|
||||
mount -t overlay overlay \
|
||||
-o lowerdir=/home/<ocp-user>/.claude,upperdir=/tmp/ocp-spawn/<reqId>/upper,workdir=/tmp/ocp-spawn/<reqId>/work \
|
||||
/tmp/ocp-spawn/<reqId>/merged-claude
|
||||
HOME=/tmp/ocp-spawn/<reqId>/home claude -p ...
|
||||
# After spawn: umount + rm -rf
|
||||
```
|
||||
|
||||
Most elegant — claude sees a view identical to its real `~/.claude/`, all writes go to tmpfs upperdir, real `~/.claude/` is never touched.
|
||||
|
||||
Trade-offs:
|
||||
- Requires `CAP_SYS_ADMIN` or rootless-overlayfs (kernel ≥5.11 + user-ns enabled). OCP currently runs as the maintainer's user — no SYS_ADMIN — so this would require either running OCP as root (bad) or rootless-overlayfs setup.
|
||||
- More moving parts (mount/umount per spawn, work-dir lifetime, cleanup-on-crash).
|
||||
|
||||
Better fit if OCP ever moves to a dedicated `ocp` system user with `CAP_SYS_ADMIN` capability via systemd.
|
||||
|
||||
**Engineering cost:** ~120 LOC + kernel/permission preflight check.
|
||||
|
||||
---
|
||||
|
||||
## 5. Cross-key isolation orthogonal layer
|
||||
|
||||
The three approaches above all solve `~/.claude.json` EROFS + state-write isolation. None of them alone solve **cross-key lateral filesystem read via absolute paths** (e.g. prompt-injected `cat /home/<user>/.ocp/keys/<otherKey>.json`).
|
||||
|
||||
For that, two options compose with any of A/B/C:
|
||||
|
||||
### 5.1 Per-spawn `sandbox-runtime` customConfig with `denyRead`
|
||||
|
||||
`@anthropic-ai/sandbox-runtime`'s `wrapWithSandbox(command, binShell?, customConfig?, abortSignal?)` accepts per-call override:
|
||||
|
||||
```
|
||||
const otherKeysWorkspaces = listAllKeyManifestsExcept(thisKeyId)
|
||||
const wrapped = await SandboxManager.wrapWithSandbox(claudeCommand, undefined, {
|
||||
filesystem: {
|
||||
denyRead: [
|
||||
...otherKeysWorkspaces, // all keys except current
|
||||
'/home/<ocp-user>/.ssh',
|
||||
'/home/<ocp-user>/.gnupg',
|
||||
'/home/<ocp-user>/.aws',
|
||||
],
|
||||
allowWrite: [ephemeralRoot, '/tmp'],
|
||||
},
|
||||
})
|
||||
```
|
||||
|
||||
This adds bwrap deny-paths per-spawn (after sandbox-runtime singleton init). Works in combination with Approach A (the `HOME` env-var override is independent of sandbox-runtime's restrictions).
|
||||
|
||||
Caveat: this re-introduces the outer-bwrap concern from § 3.1 — claude CLI is now wrapped after all. Mitigation: use this only for **cross-key isolation**, not for `$HOME` restriction. The `denyRead` paths are all outside `$HOME`, so claude's `~/.claude.json` write is unaffected.
|
||||
|
||||
### 5.2 Per-OS-user OCP spawning
|
||||
|
||||
Each OCP key gets a dedicated Linux user (`ocp-<keyId>`). Spawn claude as that user via `runuser` or `sudo -u`. OAuth credential shared via Linux group permissions or bind-mount.
|
||||
|
||||
True kernel-level uid isolation. Most robust answer for OCP-as-shared-host scenarios.
|
||||
|
||||
Trade-offs:
|
||||
- Setup script complexity (one-time per key).
|
||||
- Linux-only.
|
||||
- Doesn't fit Mac mini deployment.
|
||||
|
||||
Best fit for a cloud OCP deployment where per-tenant trust isolation matters.
|
||||
|
||||
---
|
||||
|
||||
## 6. Trust model framing
|
||||
|
||||
OCP's authentication layer (`lib/keys.mjs`) provides **attribution** (per-key audit, per-key cache namespace). It does NOT, by itself, provide **isolation** (per-key trust boundary against prompt-injection lateral reads).
|
||||
|
||||
This distinction is worth making explicit in OCP's README "Security" section (it currently isn't). The three tiers:
|
||||
|
||||
| Tier | Trust Model | Sandbox requirement |
|
||||
|---|---|---|
|
||||
| **Single-user** | maintainer's own machine, single OCP token | None — system-user permissions are sufficient |
|
||||
| **Family-trust-zone** | maintainer + family members on shared OCP instance, all parties trusted not to attack each other | Optional — Approach A (ephemeral $HOME) gives cleanup hygiene without changing trust assumptions |
|
||||
| **Shared-host / cloud / external callers** | OCP keys handed to potentially-adversarial callers (CI runners, third-party agents, public demo) | Required — Approach B or C + § 5 cross-key isolation |
|
||||
|
||||
The current OCP deployment fits tier 1 or 2. The work in this doc applies only when promoting to tier 3.
|
||||
|
||||
---
|
||||
|
||||
## 7. Recommendation if/when this work starts
|
||||
|
||||
**Phase 1 — Approach A (ephemeral `$HOME`) only.**
|
||||
- ~50 LOC, no apt deps, works on Mac mini + Linux
|
||||
- Solves the EROFS upgrade tax structurally
|
||||
- Closes cross-key OAuth-credential-file lateral read
|
||||
- Cost-effective hygiene improvement
|
||||
|
||||
**Phase 2 — Approach B (outer bwrap) gated by deployment config.**
|
||||
- Add `~/.ocp/config.json` field `security.sandbox: 'off' | 'tmpfs-home'`
|
||||
- Default off (preserves Mac mini family deployment)
|
||||
- Operator opts in on Linux cloud deployments
|
||||
- Apt prereq documented in deployment guide
|
||||
|
||||
**Phase 3 — § 5 cross-key isolation (only if tier 3 deployment is planned).**
|
||||
- Layer per-spawn customConfig denyRead OR per-OS-user spawning
|
||||
- Treat as separate ADR amendment with its own threat-model evidence
|
||||
|
||||
**Skip Approach C** unless a future requirement forces overlay (low likelihood for OCP scope).
|
||||
|
||||
---
|
||||
|
||||
## 8. Authority citations
|
||||
|
||||
This doc claims findings about claude CLI / `@anthropic-ai/sandbox-runtime` behavior. Sources for verification:
|
||||
|
||||
- [Anthropic engineering — Claude Code sandboxing](https://www.anthropic.com/engineering/claude-code-sandboxing) (sandbox-runtime design intent)
|
||||
- [Anthropic sandbox-runtime GitHub](https://github.com/anthropic-experimental/sandbox-runtime) (wrapWithSandbox API + customConfig per-call signature)
|
||||
- [claude-code#29250 — `.claude.json` non-atomic-write closed-not-planned](https://github.com/anthropics/claude-code/issues/29250)
|
||||
- [claude-code#29162 — read-only `~/.claude.json` startup hang](https://github.com/anthropics/claude-code/issues/29162)
|
||||
- [claude-code#29217 — concurrent-write corruption](https://github.com/anthropics/claude-code/issues/29217)
|
||||
- [claude-code#28842 — Windows startup race](https://github.com/anthropics/claude-code/issues/28842)
|
||||
- [claude-code#7243 — "the .claude.json elephant in the room"](https://github.com/anthropics/claude-code/issues/7243)
|
||||
- [Bubblewrap README](https://github.com/containers/bubblewrap)
|
||||
- [Bubblewrap ArchWiki — Examples section, --tmpfs HOME pattern](https://wiki.archlinux.org/title/Bubblewrap/Examples)
|
||||
- [Sandboxing CLI tools with Bubblewrap — botmonster](https://botmonster.com/self-hosting/sandbox-linux-apps-cli-tools-bubblewrap/)
|
||||
- [OverlayFS kernel documentation](https://docs.kernel.org/filesystems/overlayfs.html)
|
||||
- [OverlayFS ArchWiki](https://wiki.archlinux.org/title/Overlay_filesystem)
|
||||
|
||||
OLP's parallel work (multi-provider generalization of this strategy, including the codex inner-bwrap conflict that does not apply to OCP):
|
||||
|
||||
- `dtzp555-max/olp` `docs/adr/0014-sandbox-runtime-integration.md` (PR-B as-shipped) + Amendment 1 (pending — Solution 1 architecture)
|
||||
- `dtzp555-max/olp` `docs/plans/cloud-deployment-family.md` § 5 (deployment-side trust tier mapping)
|
||||
- archive branch `dtzp555-max/olp:phase-7-pr-b-outer-bwrap-snapshot` captures the outer-bwrap approach as snapshot if anyone wants to revisit it
|
||||
|
||||
---
|
||||
|
||||
## 9. What this doc is NOT
|
||||
|
||||
- Not an ADR. ADRs are decisions; this is a forward-facing strategy doc that becomes an ADR only when work starts and a decision is made.
|
||||
- Not a binding spec. The three approaches are alternatives; the recommendation in § 7 is the maintainer's lean from prior-art analysis, not a constitution.
|
||||
- Not authority for any code change. OCP `ALIGNMENT.md` still requires citation per Class A/B; no sandbox code lands without proper authority pinning when the work eventually starts.
|
||||
- Not a security audit. The threat model is informal — based on prior-art search + incident memory from OLP's parallel session. A real cloud deployment should commission an independent threat model.
|
||||
|
||||
---
|
||||
|
||||
**Authors:** project maintainer (handoff prepared with AI drafting assistance during OLP Phase 7 PR-B re-evaluation, 2026-05-29).
|
||||
@@ -0,0 +1,151 @@
|
||||
# 2026-06-15 Canary Runbook
|
||||
|
||||
**Purpose:** Confirm that a TUI-mode turn is billed to the **Pro/Max subscription pool** (not the Agent SDK credit pool) after Anthropic's 2026-06-15 billing split activates.
|
||||
|
||||
The billing classifier reading `cli` is **necessary but NOT sufficient** proof. (Note the naming: the value is stored in the JSONL transcript under the field name `entrypoint`, and sent to Anthropic on the wire as the `cc_entrypoint` header — they carry the same value after claude's startup classification. The commands below grep the transcript, so they match `entrypoint`.) A `cli` label tells you OCP sent the right classification; it does not tell you Anthropic billed the right pool. The only authoritative test is to observe whether the **Agent SDK credit balance** moves or not before and after the canary turn.
|
||||
|
||||
---
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- `CLAUDE_TUI_MODE=true` already set and OCP restarted (see [TUI-mode setup](../tui-mode.md#enabling-tui-mode-opt-in))
|
||||
- `tmux` installed on the host
|
||||
- No other OCP traffic during the canary (quiesce — see below)
|
||||
- Access to your Anthropic account billing page (manual step — see below)
|
||||
|
||||
---
|
||||
|
||||
## Step 1 — Quiesce the host
|
||||
|
||||
Stop any IDE or client that is actively sending requests through this OCP instance.
|
||||
|
||||
Confirm the proxy is idle:
|
||||
|
||||
```bash
|
||||
curl -s http://127.0.0.1:3456/health | python3 -m json.tool | grep activeRequests
|
||||
# Expected: "activeRequests": 0
|
||||
```
|
||||
|
||||
Wait until `activeRequests` is `0` before proceeding. If you cannot quiesce (e.g. family members are actively using it), run the canary on a separate OCP instance or during a quiet window.
|
||||
|
||||
---
|
||||
|
||||
## Step 2 — Read the Agent SDK credit balance BEFORE the canary
|
||||
|
||||
> **Manual step — no programmatic API available.**
|
||||
>
|
||||
> OCP's `/usage` endpoint reads `anthropic-ratelimit-unified-*` response headers from the Pro/Max plan quota (5-hour and 7-day subscription windows). These headers report **subscription usage**, not the Agent SDK credit pool balance. There is no known programmatic API to query the Agent SDK credit pool balance from outside the Anthropic web app.
|
||||
|
||||
To read the balance:
|
||||
|
||||
1. Open [https://claude.ai/settings/billing](https://claude.ai/settings/billing) (or your Anthropic Console billing page) in a browser.
|
||||
2. Find the **Agent SDK Credits** section (sometimes labeled "API Credits" or "Agent SDK usage").
|
||||
3. Note the current balance (e.g. `$18.43 remaining of $20.00`).
|
||||
|
||||
Write the value down — you will compare it after the canary turn.
|
||||
|
||||
---
|
||||
|
||||
## Step 3 — Send the canary turn
|
||||
|
||||
With TUI-mode on and the host quiesced, send exactly one small request:
|
||||
|
||||
```bash
|
||||
curl -s -X POST http://127.0.0.1:3456/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "claude-haiku-4-5-20251001",
|
||||
"messages": [{"role": "user", "content": "Reply with the single word: pong"}],
|
||||
"max_tokens": 10
|
||||
}' | python3 -m json.tool
|
||||
```
|
||||
|
||||
Use Haiku (the cheapest model) to minimize any hypothetical impact if the canary turns red.
|
||||
|
||||
Wait for the response to arrive completely (TUI-mode buffers the full response before returning — you will see a delay of several seconds, then the full reply).
|
||||
|
||||
---
|
||||
|
||||
## Step 4 — Confirm the transcript shows `entrypoint:"cli"`
|
||||
|
||||
After the canary turn completes, inspect the most recent JSONL transcript for the billing-classifier label:
|
||||
|
||||
```bash
|
||||
# The canary was run quiesced (Step 1), so the most recent JSONL across ALL project
|
||||
# dirs IS the canary turn. We glob every projects subdir instead of recomputing
|
||||
# claude's cwd-encoding rule (it maps every "/" AND "." to "-", e.g. ~/.ocp-tui/work
|
||||
# => projects/-home-<user>--ocp-tui-work/; see lib/tui/transcript.mjs encodeCwd) —
|
||||
# a glob is robust even if that encoding changes in a future claude build.
|
||||
LATEST=$(ls -t "$HOME"/.claude/projects/*/*.jsonl 2>/dev/null | head -1)
|
||||
echo "Transcript: $LATEST"
|
||||
grep -o '"entrypoint":"[^"]*"' "$LATEST" | tail -1
|
||||
# Expected: "entrypoint":"cli"
|
||||
```
|
||||
|
||||
If the output shows `"entrypoint":"cli"`, the billing-classifier label is correct. If it shows `"entrypoint":"sdk-cli"`, the spawn did not get a real PTY — stop immediately and do not re-enable TUI-mode without investigation. Check `tmux new-session` manually and review ADR 0007 § spawn/PTY gate. (If the grep returns nothing, the transcript may not yet be flushed — re-run after a second, or confirm the turn completed.)
|
||||
|
||||
**Reminder: an `entrypoint:cli` label (the `cc_entrypoint=cli` wire header) is necessary but not sufficient.** It tells you OCP sent the right label to Anthropic. You must still check the credit balance in Step 5.
|
||||
|
||||
---
|
||||
|
||||
## Step 5 — Re-read the Agent SDK credit balance AFTER the canary
|
||||
|
||||
Return to [https://claude.ai/settings/billing](https://claude.ai/settings/billing) and reload the page. Note the current balance again.
|
||||
|
||||
---
|
||||
|
||||
## Step 6 — Green/Red decision
|
||||
|
||||
### Green (balance unchanged)
|
||||
|
||||
The Agent SDK credit balance did not decrease. The turn billed against the Pro/Max subscription pool as expected. TUI-mode is working correctly.
|
||||
|
||||
**Actions:**
|
||||
- Keep `CLAUDE_TUI_MODE=true` on this host.
|
||||
- Monitor the balance periodically for the first week to catch any delayed attribution.
|
||||
- Resume normal traffic.
|
||||
|
||||
### Red (Agent SDK credit balance decreased)
|
||||
|
||||
The Agent SDK credit balance decreased. The subscription pool is not being used for TUI-mode turns on this host, despite `cc_entrypoint=cli` being set. This may indicate a backend routing change on Anthropic's side, a TTY detection failure, or a policy change.
|
||||
|
||||
**Actions — immediate:**
|
||||
1. Unset `CLAUDE_TUI_MODE` (or set to any value other than `"true"`) in the service unit:
|
||||
- systemd: edit `/etc/ocp/ocp.env` (or the unit's `Environment=` line), then `sudo systemctl daemon-reload && sudo systemctl restart ocp.service`
|
||||
- launchd: edit the plist `EnvironmentVariables` section, then `launchctl bootout gui/$(id -u)/dev.ocp.proxy && launchctl bootstrap gui/$(id -u) <plist-path>`
|
||||
2. Restart OCP and confirm the `/health` response no longer shows TUI-mode active.
|
||||
3. If you share this OCP with family or other Max users: freeze their access temporarily until you understand the billing impact.
|
||||
4. Consider pivoting to OLP multi-provider (see [OLP](https://github.com/dtzp555-max/olp)) which can spread load across other providers to avoid the Agent SDK credit drain.
|
||||
|
||||
Per ALIGNMENT.md Rule 2 / ADR 0007 § Kill-switch: "Per the constitution, the response is to drop the Anthropic provider rather than escalate spoofing."
|
||||
|
||||
---
|
||||
|
||||
## Ongoing monitoring — self-classification mini-canary
|
||||
|
||||
To detect future drift (e.g. a claude CLI upgrade that changes TTY-detection behavior), you can run a periodic one-liner that sends a tiny TUI turn with `OCP_TUI_ENTRYPOINT=auto` (so claude self-classifies rather than having OCP pin the value) and alerts if the transcript self-classification is not `cli`:
|
||||
|
||||
```bash
|
||||
# Run with OCP temporarily configured OCP_TUI_ENTRYPOINT=auto
|
||||
# Then check the most recent transcript:
|
||||
# Glob the most recent transcript across all project dirs (robust to claude's
|
||||
# cwd-encoding rule; run this right after the auto-mode mini-canary turn).
|
||||
LATEST=$(ls -t "$HOME"/.claude/projects/*/*.jsonl 2>/dev/null | head -1)
|
||||
RESULT=$(grep -o '"entrypoint":"[^"]*"' "$LATEST" | tail -1)
|
||||
echo "Self-classified entrypoint: $RESULT"
|
||||
if echo "$RESULT" | grep -q '"entrypoint":"cli"'; then
|
||||
echo "OK — subscription pool"
|
||||
else
|
||||
echo "ALERT — not cli; check TTY and billing"
|
||||
fi
|
||||
```
|
||||
|
||||
Run this after any major `claude` CLI upgrade. The `auto` mode lets the CLI's own `t$A` startup function determine the value from the actual TTY state (see ADR 0007 § Billing-classifier labeling).
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Flip/rollback runbook](./tui-flip-rollback.md) — how to set and unset `CLAUDE_TUI_MODE` on systemd and launchd hosts
|
||||
- [ADR 0007](../adr/0007-tui-interactive-mode.md) — TUI-mode architecture and governing rules
|
||||
- [Subscription-pool (TUI) mode](../tui-mode.md#subscription-pool-tui-mode)
|
||||
@@ -0,0 +1,180 @@
|
||||
# TUI-Mode Flip and Rollback Runbook
|
||||
|
||||
**Purpose:** Step-by-step instructions for enabling (`CLAUDE_TUI_MODE=true`) or disabling TUI-mode on real OCP deployments managed by **systemd** (Linux) or **launchd** (macOS).
|
||||
|
||||
Run the [615-canary](./615-canary.md) runbook after any flip to confirm billing pool routing is correct.
|
||||
|
||||
---
|
||||
|
||||
## Critical pitfalls — read first
|
||||
|
||||
### systemd: `daemon-reload` is required after editing the unit
|
||||
|
||||
Editing the unit file (or EnvironmentFile) and then doing `systemctl restart ocp.service` **without** `daemon-reload` will restart the process with the **old** environment from the cached unit. Always run `daemon-reload` after editing any unit file.
|
||||
|
||||
### launchd: `launchctl kickstart -k` does NOT reload plist env
|
||||
|
||||
`launchctl kickstart -k gui/$(id -u)/dev.ocp.proxy` kills the running process and re-launches it, but it **re-uses the launchd-cached environment** — not the current plist file. If you edited the plist's `EnvironmentVariables` section, you must do a full `bootout` + `bootstrap` cycle for the change to take effect. `kickstart` is not sufficient.
|
||||
|
||||
---
|
||||
|
||||
## Flip — enable TUI-mode
|
||||
|
||||
### systemd (Linux, e.g. Raspberry Pi, VPS)
|
||||
|
||||
**Option A — EnvironmentFile (recommended for clean separation)**
|
||||
|
||||
If your unit uses `EnvironmentFile=/etc/ocp/ocp.env` (or similar):
|
||||
|
||||
```bash
|
||||
# 1. Edit the environment file
|
||||
sudo nano /etc/ocp/ocp.env
|
||||
# Add or update:
|
||||
# CLAUDE_TUI_MODE=true
|
||||
#
|
||||
# If OCP binds to 0.0.0.0 AND you trust the network:
|
||||
# OCP_TUI_ALLOW_LAN=1
|
||||
# (WARNING: TUI-mode is single-user only — only enable OCP_TUI_ALLOW_LAN=1
|
||||
# if you fully trust every caller that can reach the OCP port on your network)
|
||||
|
||||
# 2. Reload the unit definition and restart
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl restart ocp.service
|
||||
|
||||
# 3. Verify
|
||||
curl -s http://127.0.0.1:3456/health | python3 -m json.tool | grep -E "tui|version"
|
||||
# Expected: "tuiMode": true (or similar TUI indicator in the health response)
|
||||
```
|
||||
|
||||
**Option B — inline Environment= in the unit file**
|
||||
|
||||
```bash
|
||||
# 1. Edit the unit file
|
||||
sudo systemctl edit --full ocp.service
|
||||
# Add or update in the [Service] section:
|
||||
# Environment=CLAUDE_TUI_MODE=true
|
||||
|
||||
# 2. Reload and restart
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl restart ocp.service
|
||||
|
||||
# 3. Verify
|
||||
systemctl show ocp.service --property=Environment
|
||||
# Expected: Environment=CLAUDE_TUI_MODE=true ...
|
||||
```
|
||||
|
||||
### launchd (macOS)
|
||||
|
||||
Locate the OCP plist. The standard label is `dev.ocp.proxy`:
|
||||
|
||||
```bash
|
||||
# Find the plist path
|
||||
ls ~/Library/LaunchAgents/dev.ocp.proxy.plist
|
||||
```
|
||||
|
||||
**Edit the plist:**
|
||||
|
||||
```bash
|
||||
# 1. Stop the service first (bootout)
|
||||
launchctl bootout gui/$(id -u)/dev.ocp.proxy
|
||||
|
||||
# 2. Edit the plist — add CLAUDE_TUI_MODE to EnvironmentVariables
|
||||
# Use your editor of choice:
|
||||
nano ~/Library/LaunchAgents/dev.ocp.proxy.plist
|
||||
```
|
||||
|
||||
Inside the plist, in the `<key>EnvironmentVariables</key>` `<dict>` block, add:
|
||||
|
||||
```xml
|
||||
<key>CLAUDE_TUI_MODE</key>
|
||||
<string>true</string>
|
||||
```
|
||||
|
||||
If `OCP_TUI_ALLOW_LAN=1` is also needed (only if OCP binds to `0.0.0.0` and you trust the network):
|
||||
|
||||
```xml
|
||||
<key>OCP_TUI_ALLOW_LAN</key>
|
||||
<string>1</string>
|
||||
```
|
||||
|
||||
```bash
|
||||
# 3. Bootstrap (reload from disk + start)
|
||||
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/dev.ocp.proxy.plist
|
||||
|
||||
# 4. Verify
|
||||
curl -s http://127.0.0.1:3456/health | python3 -m json.tool | grep -E "tui|version"
|
||||
```
|
||||
|
||||
**Confirm env was actually loaded** (not just set in your shell):
|
||||
|
||||
```bash
|
||||
ps aux | grep server.mjs | grep -v grep
|
||||
# Get the PID, then:
|
||||
# macOS: ps -E -p <PID> | tr ' ' '\n' | grep CLAUDE_TUI_MODE
|
||||
# Expected: CLAUDE_TUI_MODE=true
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Rollback — disable TUI-mode
|
||||
|
||||
Rollback is the same procedure as flip, but you **remove** `CLAUDE_TUI_MODE` or set it to any value other than `"true"` (e.g. `false`, or simply omit it).
|
||||
|
||||
After rollback, OCP returns to the default `callClaude` / `callClaudeStreaming` stream-json path — byte-for-byte identical to the pre-TUI code path. No other change is required.
|
||||
|
||||
### systemd rollback
|
||||
|
||||
```bash
|
||||
# Option A — EnvironmentFile
|
||||
sudo nano /etc/ocp/ocp.env
|
||||
# Remove or comment out:
|
||||
# CLAUDE_TUI_MODE=true
|
||||
# OCP_TUI_ALLOW_LAN=1 (if set)
|
||||
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl restart ocp.service
|
||||
|
||||
# Verify
|
||||
curl -s http://127.0.0.1:3456/health | python3 -m json.tool | grep tui
|
||||
# Expected: "tuiMode": false (or the field absent)
|
||||
```
|
||||
|
||||
### launchd rollback
|
||||
|
||||
```bash
|
||||
# 1. Stop
|
||||
launchctl bootout gui/$(id -u)/dev.ocp.proxy
|
||||
|
||||
# 2. Edit plist — remove the CLAUDE_TUI_MODE and OCP_TUI_ALLOW_LAN entries from EnvironmentVariables
|
||||
|
||||
# 3. Bootstrap
|
||||
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/dev.ocp.proxy.plist
|
||||
|
||||
# 4. Verify
|
||||
curl -s http://127.0.0.1:3456/health | python3 -m json.tool | grep tui
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Billing impact of staying on the default (non-TUI) path after 2026-06-15
|
||||
|
||||
If you do NOT flip to TUI-mode and keep `CLAUDE_TUI_MODE` unset (the default), OCP continues using `claude -p --output-format stream-json`, which sets `cc_entrypoint=sdk-cli`. After 2026-06-15, every OCP request on the default path will draw from the Agent SDK credit pool (approximately $20/month on a Pro plan, or $100/month on a Max plan) rather than the Pro/Max subscription. The subscription pool usage (5-hour and 7-day windows) will be unaffected, but the Agent SDK credit balance will drain with each request.
|
||||
|
||||
If you want to continue using OCP without TUI-mode after 2026-06-15, budget for the Agent SDK credit cost accordingly — or switch to [OLP](https://github.com/dtzp555-max/olp) for multi-provider fallback.
|
||||
|
||||
---
|
||||
|
||||
## Verify after any flip
|
||||
|
||||
1. Check `/health` shows the expected `tuiMode` state.
|
||||
2. Run the [615-canary](./615-canary.md) to confirm billing pool routing.
|
||||
3. If TUI-mode is ON: check `ocp logs 10` for any TUI spawn errors (`tui_spawn_failed`, tmux errors).
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [615-canary runbook](./615-canary.md) — how to verify billing pool routing after a flip
|
||||
- [ADR 0007](../adr/0007-tui-interactive-mode.md) — TUI-mode architecture; Kill-switch section
|
||||
- [Subscription-pool (TUI) mode](../tui-mode.md#subscription-pool-tui-mode)
|
||||
- README § [Environment Variables](../../README.md#environment-variables) — `CLAUDE_TUI_MODE`, `OCP_TUI_ALLOW_LAN=1`
|
||||
@@ -0,0 +1,737 @@
|
||||
# TUI-mode (OCP-first) Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Add an opt-in `CLAUDE_TUI_MODE` to OCP that serves `/v1/chat/completions` by driving a *real interactive* `claude` session (no `-p`, no `--output-format`) so the request bills as `cc_entrypoint=cli` (subscription pool), reading the answer from claude's native JSONL transcript — while the default stream-json path stays byte-for-byte unchanged.
|
||||
|
||||
**Architecture:** Two new pure-ish modules under `lib/tui/` — a transcript **reader** (`transcript.mjs`, provider-agnostic, the shareable core) and a tmux **session driver** (`session.mjs`, OCP-specific). `server.mjs` gains a `callClaudeTui()` that returns `Promise<string>` and is gated into the existing dispatch by a single env flag; because OCP's entire downstream (singleflight → `setCachedResponse` → `completionResponse` / chunked-SSE-replay → `recordUsage`) already consumes a string from `callClaude`, TUI-mode is a drop-in. Streaming is buffered then replayed as chunked SSE (no token streaming — deliberately, "don't build fragile features").
|
||||
|
||||
**Tech Stack:** Node.js ESM (`.mjs`), `tmux` (interactive PTY host), `child_process` (`spawnSync`), `node:fs` polling (no `fs.watch`, no terminal-screen parsing). Test harness: `node test-features.mjs`.
|
||||
|
||||
**Source of truth for the TUI mechanism:** the OLP design spec `docs/superpowers/specs/2026-05-30-tui-mode-production-design.md` (CLI-level, applies to both projects) + its 6 validation spikes (S1–S6, T1–T6) run on PI231 against `claude v2.1.158`. This plan is the OCP-grounded execution of that spec.
|
||||
|
||||
---
|
||||
|
||||
## Why OCP-first / scope decisions (read before coding)
|
||||
|
||||
- **OCP-first** because OCP has the users and its compute path is `callClaude → Promise<string>`, a near-perfect impedance match for a reader that also returns a string. OLP would additionally need a string→IR-chunk-array adapter. OLP-sync is **deferred entirely until the post-2026-06-15 fork decision** — do not spend cycles keeping OLP's TUI in lockstep.
|
||||
- **A-path only.** Single-user / multi-device on one subscription. No per-key ephemeral isolation, no multi-tenant. (That is the OLP B-path, deferred.)
|
||||
- **A-path isolation = real `$HOME` + dedicated scratch cwd + `--strict-mcp-config`.** OCP has *no* ISOLATION contract and we do not build one. We run interactive `claude` in the operator's real home (OAuth + onboarding already valid) but in a **dedicated scratch working directory** (`OCP_TUI_CWD`, default `$HOME/.ocp-tui/work`) so transcripts land under one stable `projects/<cwd>` folder instead of polluting the operator's genuine project histories, and the trust-folder dialog is granted once.
|
||||
- **One `claude` session per request.** OCP is stateless (full conversation re-serialized each request via `messagesToPrompt`). TUI-mode mirrors this: per request, start a fresh interactive session with a fresh `--session-id`, submit one serialized prompt, await turn completion, read the transcript, extract the latest assistant text, tear the session down. Warm-pool / large-paste optimizations are explicitly out of v1 scope.
|
||||
- **Billing is unmeasurable until 2026-06-15.** Spike S1 proved the `cc_entrypoint=cli` *signal*, not the billed pool. The pre-6/15 deliverable is "a tested, working transport that emits `cli`"; 6/16 we flip the flag and measure with a documented kill-switch.
|
||||
- **Coexistence rule (PI231 runs an OLP test instance too).** All tmux sessions use the prefix `ocp-tui-`; the reaper kills **only** `ocp-tui-*`, never `olp-tui-*`. Never run two TUI proxies on the same OAuth concurrently — stop the OLP test instance during OCP integration.
|
||||
- **Provenance.** TUI-mode originated in OCP PR #101 (author courtesy: jaekwon-park <insainty21@gmail.com>). The PR #101 author should be credited + notified on the shipping PR.
|
||||
|
||||
---
|
||||
|
||||
## File Structure
|
||||
|
||||
| File | Responsibility | New/Modified |
|
||||
|------|----------------|--------------|
|
||||
| `lib/tui/transcript.mjs` | Pure transcript parsing + the polling reader. Returns the latest assistant text once the turn is terminal or the wall-clock cap elapses. Provider-agnostic — the shareable core. | **Create** |
|
||||
| `lib/tui/session.mjs` | tmux session lifecycle: boot interactive `claude`, answer the trust dialog, submit the prompt (file → `"$(cat)"` paste → separate Enter), await the reader, tear down. Plus the prefix-scoped reaper. OCP-specific. | **Create** |
|
||||
| `lib/tui/fixtures/` | Real transcript JSONL harvested from PI231 + a few hand-crafted edge cases, for the reader's unit tests. | **Create** |
|
||||
| `server.mjs` | `callClaudeTui()` (`Promise<string>`); `streamStringAsSSE()` helper (DRY refactor of the cache-replay block); single-flag dispatch gates; reaper hook at boot; env consts. | **Modify** (`:258` env consts, `:1018`–`:1023` helpers, `:1467` dispatch, boot block) |
|
||||
| `test-features.mjs` | Suite for the reader (fixtures, runs in CI) + a live-only guarded suite for the driver (`OCP_TUI_LIVE=1`, skipped in CI). | **Modify** |
|
||||
| `docs/adr/0007-tui-interactive-mode.md` | OCP ADR 0007 (OCP's next number) — TUI mode rationale, billing-signal authority, scope, kill-switch. | **Create** |
|
||||
| `README.md` | New env vars (`CLAUDE_TUI_MODE`, `CLAUDE_TUI_WALLCLOCK_MS`, `OCP_TUI_CWD`), a "Subscription-pool (TUI) mode" section, troubleshooting + kill-switch. | **Modify** |
|
||||
| `CHANGELOG.md` | Unreleased entry. | **Modify** |
|
||||
|
||||
---
|
||||
|
||||
## PR-1 — Transcript reader (`lib/tui/transcript.mjs`)
|
||||
|
||||
The shareable core. Pure functions + a polling reader. Fully unit-testable from committed fixtures; needs PI231 only once, to harvest realistic fixtures.
|
||||
|
||||
### Task 0: Harvest real fixtures from PI231
|
||||
|
||||
**Files:**
|
||||
- Create: `lib/tui/fixtures/complete-haiku.jsonl` (real, has `turn_duration`)
|
||||
- Create: `lib/tui/fixtures/complete-sonnet-multiblock.jsonl` (real, multi content-block answer)
|
||||
|
||||
- [ ] **Step 1: Drive one real interactive turn on PI231 and copy its transcript**
|
||||
|
||||
On PI231 (the only box with an authenticated interactive `claude`), run a single interactive turn in a scratch cwd, then locate its transcript:
|
||||
|
||||
Run (on PI231):
|
||||
```bash
|
||||
SID=$(uuidgen)
|
||||
mkdir -p ~/.ocp-tui/work
|
||||
# drive one turn by hand in tmux OR reuse a transcript already produced by the S-spikes:
|
||||
ls -t ~/.claude/projects/-home-*-.ocp-tui-work/*.jsonl 2>/dev/null | head
|
||||
# pick one complete transcript (must contain a line with "subtype":"turn_duration")
|
||||
```
|
||||
Expected: at least one `.jsonl` file whose tail contains `{"type":"system","subtype":"turn_duration",...}`.
|
||||
|
||||
- [ ] **Step 2: Copy 2 real transcripts into the repo as fixtures, scrubbed**
|
||||
|
||||
Run (from the workstation):
|
||||
```bash
|
||||
scp pi231:'~/.claude/projects/<encoded-cwd>/<sid>.jsonl' lib/tui/fixtures/complete-haiku.jsonl
|
||||
# Scrub: the transcript may contain the prompt/answer text only (no OAuth token — tokens
|
||||
# live in ~/.claude/.credentials.json, NOT in projects/*.jsonl). Confirm no credential
|
||||
# material before committing:
|
||||
grep -iE "sk-ant|oat01|bearer|authorization" lib/tui/fixtures/*.jsonl && echo "STOP: scrub" || echo "clean"
|
||||
```
|
||||
Expected: `clean`. (Transcripts hold conversation content + metadata, never the bearer token. If a fixture's prompt text is sensitive, replace it with a benign hand-edited turn that keeps the JSON shape.)
|
||||
|
||||
- [ ] **Step 3: Commit the fixtures**
|
||||
|
||||
```bash
|
||||
git add lib/tui/fixtures/complete-haiku.jsonl lib/tui/fixtures/complete-sonnet-multiblock.jsonl
|
||||
git commit -m "test(tui): real claude transcript fixtures harvested from PI231 (v2.1.158)"
|
||||
```
|
||||
|
||||
### Task 1: `encodeCwd` + `transcriptPath` (the path formula)
|
||||
|
||||
**Files:**
|
||||
- Create: `lib/tui/transcript.mjs`
|
||||
- Test: `test-features.mjs` (new Suite "TUI transcript")
|
||||
|
||||
- [ ] **Step 1: Write the failing test**
|
||||
|
||||
Add to `test-features.mjs`:
|
||||
```js
|
||||
// ── Suite: TUI transcript reader ────────────────────────────────────────
|
||||
import { encodeCwd, transcriptPath } from "./lib/tui/transcript.mjs";
|
||||
|
||||
test("encodeCwd replaces every slash incl. leading", () => {
|
||||
assertEqual(encodeCwd("/home/u/.ocp-tui/work"), "-home-u-.ocp-tui-work");
|
||||
});
|
||||
test("transcriptPath composes EHOME/.claude/projects/<enc>/<sid>.jsonl", () => {
|
||||
assertEqual(
|
||||
transcriptPath("/home/u", "/home/u/.ocp-tui/work", "abc-123"),
|
||||
"/home/u/.claude/projects/-home-u-.ocp-tui-work/abc-123.jsonl"
|
||||
);
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run to verify it fails**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -i "tui transcript\|Cannot find module"`
|
||||
Expected: FAIL — `Cannot find module './lib/tui/transcript.mjs'`.
|
||||
|
||||
- [ ] **Step 3: Minimal implementation**
|
||||
|
||||
Create `lib/tui/transcript.mjs`:
|
||||
```js
|
||||
// Transcript reader for TUI-mode. Reads claude's native JSONL session transcript
|
||||
// and returns the latest assistant turn's text once the turn is terminal.
|
||||
//
|
||||
// Authority: claude CLI v2.1.158 — interactive session transcript at
|
||||
// <HOME>/.claude/projects/<CWD with every "/" -> "-">/<--session-id>.jsonl
|
||||
// Completion marker: a line {"type":"system","subtype":"turn_duration",...}.
|
||||
// See docs/superpowers/specs/2026-05-30-tui-mode-production-design.md §4.
|
||||
import { readFileSync, existsSync } from "node:fs";
|
||||
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
// Project-dir encoding: every "/" -> "-" (including the leading slash).
|
||||
export function encodeCwd(cwd) {
|
||||
return cwd.replace(/\//g, "-");
|
||||
}
|
||||
|
||||
export function transcriptPath(home, cwd, sessionId) {
|
||||
return `${home}/.claude/projects/${encodeCwd(cwd)}/${sessionId}.jsonl`;
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run to verify it passes**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -i "tui transcript"`
|
||||
Expected: PASS for both cases.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add lib/tui/transcript.mjs test-features.mjs
|
||||
git commit -m "feat(tui): transcript path formula (encodeCwd + transcriptPath)"
|
||||
```
|
||||
|
||||
### Task 2: `parseTranscriptLines` + `isTerminalLine` + `extractLatestAssistantText`
|
||||
|
||||
**Files:**
|
||||
- Modify: `lib/tui/transcript.mjs`
|
||||
- Test: `test-features.mjs`
|
||||
|
||||
- [ ] **Step 1: Write the failing tests**
|
||||
|
||||
```js
|
||||
import { parseTranscriptLines, isTerminalLine, extractLatestAssistantText } from "./lib/tui/transcript.mjs";
|
||||
import { readFileSync } from "node:fs";
|
||||
|
||||
test("parseTranscriptLines skips blank + malformed/partial lines", () => {
|
||||
const evs = parseTranscriptLines('{"a":1}\n\n{bad json\n{"b":2}\n');
|
||||
assertEqual(evs.length, 2);
|
||||
assertEqual(evs[1].b, 2);
|
||||
});
|
||||
test("isTerminalLine true on turn_duration", () => {
|
||||
assertEqual(isTerminalLine({ type: "system", subtype: "turn_duration" }), true);
|
||||
});
|
||||
test("isTerminalLine true on stop_reason tool_use (message-wrapped + flat)", () => {
|
||||
assertEqual(isTerminalLine({ type: "assistant", message: { stop_reason: "tool_use" } }), true);
|
||||
assertEqual(isTerminalLine({ stop_reason: "tool_use" }), true);
|
||||
});
|
||||
test("isTerminalLine false on ordinary assistant/text lines", () => {
|
||||
assertEqual(isTerminalLine({ type: "assistant", message: { content: [{ type: "text", text: "hi" }] } }), false);
|
||||
});
|
||||
test("extractLatestAssistantText concatenates text blocks of the LAST assistant turn", () => {
|
||||
const evs = [
|
||||
{ type: "assistant", message: { content: [{ type: "text", text: "first" }] } },
|
||||
{ type: "user", message: { content: "..." } },
|
||||
{ type: "assistant", message: { content: [{ type: "text", text: "A" }, { type: "thinking", thinking: "x" }, { type: "text", text: "B" }] } },
|
||||
];
|
||||
assertEqual(extractLatestAssistantText(evs), "AB");
|
||||
});
|
||||
test("real complete fixture yields non-empty text and is terminal", () => {
|
||||
const evs = parseTranscriptLines(readFileSync("./lib/tui/fixtures/complete-haiku.jsonl", "utf8"));
|
||||
assert(evs.some(isTerminalLine), "fixture must contain a terminal line");
|
||||
assert(extractLatestAssistantText(evs).length > 0, "fixture must yield assistant text");
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run to verify it fails**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -i "parseTranscript\|isTerminal\|extractLatest\|real complete fixture"`
|
||||
Expected: FAIL — exports not defined.
|
||||
|
||||
- [ ] **Step 3: Minimal implementation** (append to `lib/tui/transcript.mjs`)
|
||||
|
||||
```js
|
||||
// Parse NDJSON text into objects; skip blank lines and partial/forming lines
|
||||
// (the live transcript is read mid-write, so the last line may be incomplete).
|
||||
export function parseTranscriptLines(text) {
|
||||
const out = [];
|
||||
for (const line of text.split("\n")) {
|
||||
const t = line.trim();
|
||||
if (!t) continue;
|
||||
try { out.push(JSON.parse(t)); } catch { /* partial line being written */ }
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// A line marks the assistant turn complete when it is the turn_duration system
|
||||
// event, or an assistant message that stopped to hand off to a tool.
|
||||
export function isTerminalLine(obj) {
|
||||
if (!obj || typeof obj !== "object") return false;
|
||||
if (obj.type === "system" && obj.subtype === "turn_duration") return true;
|
||||
const sr = (obj.message && obj.message.stop_reason) || obj.stop_reason;
|
||||
return sr === "tool_use";
|
||||
}
|
||||
|
||||
// Text of the LAST assistant turn: concatenate its text content blocks
|
||||
// (ignore thinking/tool_use blocks). Later assistant entries overwrite earlier.
|
||||
export function extractLatestAssistantText(events) {
|
||||
let text = "";
|
||||
for (const ev of events) {
|
||||
if (!ev || ev.type !== "assistant") continue;
|
||||
const content = ev.message && ev.message.content;
|
||||
if (!Array.isArray(content)) continue;
|
||||
const parts = content
|
||||
.filter((b) => b && b.type === "text" && typeof b.text === "string")
|
||||
.map((b) => b.text);
|
||||
if (parts.length) text = parts.join("");
|
||||
}
|
||||
return text;
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run to verify it passes**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -iE "parseTranscript|isTerminal|extractLatest|real complete fixture"`
|
||||
Expected: all PASS.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add lib/tui/transcript.mjs test-features.mjs
|
||||
git commit -m "feat(tui): transcript parsing + terminal detection + assistant-text extraction"
|
||||
```
|
||||
|
||||
### Task 3: `readTuiTranscript` (the polling reader with wall-clock cap)
|
||||
|
||||
**Files:**
|
||||
- Modify: `lib/tui/transcript.mjs`
|
||||
- Test: `test-features.mjs`
|
||||
|
||||
- [ ] **Step 1: Write the failing tests**
|
||||
|
||||
```js
|
||||
import { readTuiTranscript } from "./lib/tui/transcript.mjs";
|
||||
import { mkdtempSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
|
||||
test("readTuiTranscript returns assistant text when terminal marker present", async () => {
|
||||
const dir = mkdtempSync(`${tmpdir()}/tui-`);
|
||||
const p = `${dir}/s.jsonl`;
|
||||
writeFileSync(p, [
|
||||
JSON.stringify({ type: "assistant", message: { content: [{ type: "text", text: "hello world" }] } }),
|
||||
JSON.stringify({ type: "system", subtype: "turn_duration", durationMs: 1200 }),
|
||||
].join("\n") + "\n");
|
||||
const out = await readTuiTranscript({ transcriptPath: p, wallclockMs: 2000, pollMs: 50 });
|
||||
assertEqual(out, "hello world");
|
||||
});
|
||||
test("readTuiTranscript honours wall-clock cap and returns partial text", async () => {
|
||||
const dir = mkdtempSync(`${tmpdir()}/tui-`);
|
||||
const p = `${dir}/s.jsonl`;
|
||||
writeFileSync(p, JSON.stringify({ type: "assistant", message: { content: [{ type: "text", text: "partial" }] } }) + "\n");
|
||||
const out = await readTuiTranscript({ transcriptPath: p, wallclockMs: 300, pollMs: 50 }); // never terminal
|
||||
assertEqual(out, "partial");
|
||||
});
|
||||
test("readTuiTranscript throws when no text and cap elapses", async () => {
|
||||
const dir = mkdtempSync(`${tmpdir()}/tui-`);
|
||||
const p = `${dir}/missing.jsonl`; // file never appears
|
||||
let threw = false;
|
||||
try { await readTuiTranscript({ transcriptPath: p, wallclockMs: 200, pollMs: 50 }); }
|
||||
catch { threw = true; }
|
||||
assert(threw, "must throw on empty timeout");
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run to verify it fails**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -i "readTuiTranscript"`
|
||||
Expected: FAIL — export not defined.
|
||||
|
||||
- [ ] **Step 3: Minimal implementation** (append)
|
||||
|
||||
```js
|
||||
// Block until the session transcript is terminal (turn_duration / tool_use) or
|
||||
// the wall-clock cap elapses, polling the file (no fs.watch — robust over NFS /
|
||||
// editors). Returns the latest assistant text. On cap with text, returns the
|
||||
// partial text; on cap with no text at all, throws.
|
||||
//
|
||||
// No quiescence heuristic by design: a long Opus thinking turn stalls transcript
|
||||
// growth and a "file stable for N s" rule would false-abort it (spec §4.3).
|
||||
export async function readTuiTranscript({ transcriptPath: p, wallclockMs = 120000, pollMs = 250 }) {
|
||||
const deadline = Date.now() + wallclockMs;
|
||||
let lastText = "";
|
||||
while (Date.now() < deadline) {
|
||||
if (existsSync(p)) {
|
||||
const events = parseTranscriptLines(readFileSync(p, "utf8"));
|
||||
lastText = extractLatestAssistantText(events) || lastText;
|
||||
if (events.some(isTerminalLine)) return lastText;
|
||||
}
|
||||
await sleep(pollMs);
|
||||
}
|
||||
if (lastText) return lastText;
|
||||
throw new Error("tui_transcript_timeout: no assistant text within wallclock cap");
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run to verify it passes**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -i "readTuiTranscript"`
|
||||
Expected: all 3 PASS.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add lib/tui/transcript.mjs test-features.mjs
|
||||
git commit -m "feat(tui): polling transcript reader with wall-clock cap (no quiescence)"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## PR-2 — Session driver (`lib/tui/session.mjs`)
|
||||
|
||||
tmux lifecycle + the validated submission recipe. Cannot be unit-tested without a live authenticated `claude`; tested by a live-only guarded suite that runs on PI231.
|
||||
|
||||
### Task 4: `reapStaleTuiSessions` (prefix-scoped reaper)
|
||||
|
||||
**Files:**
|
||||
- Create: `lib/tui/session.mjs`
|
||||
- Test: `test-features.mjs`
|
||||
|
||||
- [ ] **Step 1: Write the failing test** (pure — no live claude; inject a fake tmux runner)
|
||||
|
||||
```js
|
||||
import { reapStaleTuiSessions, SESSION_PREFIX } from "./lib/tui/session.mjs";
|
||||
|
||||
test("reaper kills ONLY ocp-tui- sessions, never olp-tui-", () => {
|
||||
const killed = [];
|
||||
const fakeTmux = (args) => {
|
||||
if (args[0] === "list-sessions") return { status: 0, stdout: "ocp-tui-aaaa\nolp-tui-bbbb\nmisc\nocp-tui-cccc\n" };
|
||||
if (args[0] === "kill-session") { killed.push(args[args.indexOf("-t") + 1]); return { status: 0 }; }
|
||||
return { status: 0, stdout: "" };
|
||||
};
|
||||
const n = reapStaleTuiSessions({ tmux: fakeTmux });
|
||||
assertEqual(SESSION_PREFIX, "ocp-tui-");
|
||||
assertEqual(n, 2);
|
||||
assertEqual(killed.join(","), "ocp-tui-aaaa,ocp-tui-cccc");
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run to verify it fails**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -i "reaper kills"`
|
||||
Expected: FAIL — module/export missing.
|
||||
|
||||
- [ ] **Step 3: Minimal implementation**
|
||||
|
||||
Create `lib/tui/session.mjs`:
|
||||
```js
|
||||
// TUI-mode session driver: hosts an interactive `claude` in tmux, submits one
|
||||
// serialized prompt, awaits the transcript reader, tears down. OCP-specific.
|
||||
//
|
||||
// Authority: claude CLI v2.1.158 interactive mode (no -p / no --output-format
|
||||
// => cc_entrypoint=cli). Submission recipe + dialog handling validated by spikes
|
||||
// T3/T6 on PI231. See docs/superpowers/specs/2026-05-30-tui-mode-production-design.md.
|
||||
import { spawnSync } from "node:child_process";
|
||||
import { mkdtempSync, writeFileSync, mkdirSync, existsSync, rmSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { transcriptPath, readTuiTranscript } from "./transcript.mjs";
|
||||
|
||||
export const SESSION_PREFIX = "ocp-tui-"; // per-proxy namespace (coexistence rule)
|
||||
const TMUX = process.env.OCP_TUI_TMUX_BIN || "tmux";
|
||||
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
const defaultTmux = (args, opts = {}) => spawnSync(TMUX, args, { encoding: "utf8", ...opts });
|
||||
|
||||
// Kill ONLY our own stale sessions. Scoped to SESSION_PREFIX so a co-hosted
|
||||
// OLP test instance's `olp-tui-*` sessions are never touched.
|
||||
export function reapStaleTuiSessions({ tmux = defaultTmux } = {}) {
|
||||
const r = tmux(["list-sessions", "-F", "#{session_name}"]);
|
||||
if (!r || r.status !== 0) return 0; // no tmux server / no sessions
|
||||
let killed = 0;
|
||||
for (const name of String(r.stdout || "").split("\n").map((s) => s.trim()).filter(Boolean)) {
|
||||
if (name.startsWith(SESSION_PREFIX)) { tmux(["kill-session", "-t", name]); killed++; }
|
||||
}
|
||||
return killed;
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run to verify it passes**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | grep -i "reaper kills"`
|
||||
Expected: PASS.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add lib/tui/session.mjs test-features.mjs
|
||||
git commit -m "feat(tui): prefix-scoped session reaper (ocp-tui-* only)"
|
||||
```
|
||||
|
||||
### Task 5: `runTuiTurn` (boot → trust dialog → paste → Enter → read → teardown)
|
||||
|
||||
**Files:**
|
||||
- Modify: `lib/tui/session.mjs`
|
||||
- Test: `test-features.mjs` (live-only, guarded by `OCP_TUI_LIVE=1`)
|
||||
|
||||
- [ ] **Step 1: Write the live-only guarded test** (skipped in CI; run on PI231)
|
||||
|
||||
```js
|
||||
// Live-only: requires an authenticated interactive `claude`. Skipped unless OCP_TUI_LIVE=1.
|
||||
if (process.env.OCP_TUI_LIVE === "1") {
|
||||
test("runTuiTurn drives a real interactive turn and returns text", async () => {
|
||||
const { runTuiTurn } = await import("./lib/tui/session.mjs");
|
||||
const out = await runTuiTurn({
|
||||
prompt: "Reply with exactly the word PONG and nothing else.",
|
||||
model: "claude-haiku-4-5-20251001",
|
||||
claudeBin: process.env.OCP_TUI_CLAUDE_BIN || "claude",
|
||||
home: process.env.HOME,
|
||||
cwd: `${process.env.HOME}/.ocp-tui/work`,
|
||||
wallclockMs: 120000,
|
||||
});
|
||||
assert(/PONG/i.test(out), `expected PONG, got: ${out.slice(0, 200)}`);
|
||||
});
|
||||
} else {
|
||||
test("runTuiTurn (live) — SKIPPED (set OCP_TUI_LIVE=1 on PI231 to run)", () => { assert(true); });
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run to verify it fails** (on a box, with the flag)
|
||||
|
||||
Run (PI231): `OCP_TUI_LIVE=1 node test-features.mjs 2>&1 | grep -i "runTuiTurn"`
|
||||
Expected: FAIL — `runTuiTurn` not exported yet.
|
||||
|
||||
- [ ] **Step 3: Implementation** (append to `lib/tui/session.mjs`)
|
||||
|
||||
```js
|
||||
// Boot wait + dialog timing. Conservative defaults validated on PI231; env-tunable.
|
||||
const BOOT_MS = parseInt(process.env.OCP_TUI_BOOT_MS || "3500", 10);
|
||||
const DIALOG_MS = parseInt(process.env.OCP_TUI_DIALOG_MS || "1200", 10);
|
||||
const PASTE_SETTLE_MS = parseInt(process.env.OCP_TUI_PASTE_MS || "1800", 10);
|
||||
|
||||
const shq = (s) => `'${String(s).replace(/'/g, "'\\''")}'`; // single-quote for sh -c
|
||||
|
||||
// Build interactive claude argv: NO -p, NO --output-format (=> cc_entrypoint=cli).
|
||||
// MCP hard-disabled: --strict-mcp-config (no --mcp-config) is the only mechanism
|
||||
// that stops account-attached managed MCP from connecting (spec §5.2 / T6),
|
||||
// belt-and-braces with --disallowedTools "mcp__*".
|
||||
function buildTuiCmd(claudeBin, model, sessionId) {
|
||||
return [
|
||||
shq(claudeBin),
|
||||
"--model", shq(model),
|
||||
"--session-id", sessionId,
|
||||
"--strict-mcp-config",
|
||||
"--disallowedTools", shq("mcp__*"),
|
||||
].join(" ");
|
||||
}
|
||||
|
||||
export async function runTuiTurn({
|
||||
prompt, model, claudeBin, home, cwd,
|
||||
wallclockMs = 120000, tmux = defaultTmux,
|
||||
}) {
|
||||
const sessionId = randomUUID();
|
||||
const tmuxName = SESSION_PREFIX + sessionId.slice(0, 8);
|
||||
if (!existsSync(cwd)) mkdirSync(cwd, { recursive: true });
|
||||
|
||||
const tmpDir = mkdtempSync(`${tmpdir()}/ocp-tui-`);
|
||||
const promptFile = `${tmpDir}/prompt.txt`;
|
||||
writeFileSync(promptFile, prompt, { mode: 0o600 });
|
||||
|
||||
const env = { ...process.env, CLAUDE_CODE_DISABLE_OFFICIAL_MARKETPLACE_AUTOINSTALL: "1" };
|
||||
delete env.CLAUDECODE; delete env.ANTHROPIC_API_KEY; delete env.ANTHROPIC_BASE_URL; delete env.ANTHROPIC_AUTH_TOKEN;
|
||||
if (home) env.HOME = home;
|
||||
|
||||
try {
|
||||
// 1. Boot the interactive session inside tmux, in the dedicated scratch cwd.
|
||||
tmux(["new-session", "-d", "-s", tmuxName, "-x", "220", "-y", "50", "-c", cwd,
|
||||
buildTuiCmd(claudeBin, model, sessionId)], { env });
|
||||
await sleep(BOOT_MS);
|
||||
|
||||
// 2. Answer the trust-folder dialog defensively. The seeded bypass flag (if any)
|
||||
// suppresses the *bypass-permissions* dialog but NOT the trust-folder dialog;
|
||||
// "1" = "Yes, proceed". Harmless if the dialog is absent (cwd already trusted).
|
||||
tmux(["send-keys", "-t", tmuxName, "1"]);
|
||||
tmux(["send-keys", "-t", tmuxName, "Enter"]);
|
||||
await sleep(DIALOG_MS);
|
||||
|
||||
// 3. Submit the prompt. Body is pasted via `"$(cat file)"` so the content never
|
||||
// touches the command line (no shell injection from prompt text), then a
|
||||
// SEPARATE Enter key event submits it (Ink #15553: literal "\n" in a paste
|
||||
// does not submit; the Enter key event does).
|
||||
spawnSync("sh", ["-c",
|
||||
`${shq(TMUX)} send-keys -t ${shq(tmuxName)} -- "$(cat ${shq(promptFile)})"`],
|
||||
{ env, encoding: "utf8" });
|
||||
await sleep(PASTE_SETTLE_MS);
|
||||
tmux(["send-keys", "-t", tmuxName, "Enter"]);
|
||||
|
||||
// 4. Read the answer from the native transcript.
|
||||
const tpath = transcriptPath(home || process.env.HOME, cwd, sessionId);
|
||||
return await readTuiTranscript({ transcriptPath: tpath, wallclockMs });
|
||||
} finally {
|
||||
// 5. Teardown — always. Kill the session, remove the temp prompt dir.
|
||||
try { tmux(["kill-session", "-t", tmuxName]); } catch { /* already gone */ }
|
||||
try { rmSync(tmpDir, { recursive: true, force: true }); } catch { /* best effort */ }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run to verify it passes** (PI231, live)
|
||||
|
||||
Run (PI231): `OCP_TUI_LIVE=1 node test-features.mjs 2>&1 | grep -i "runTuiTurn"`
|
||||
Expected: PASS — output contains `PONG`. Also confirm no orphan sessions: `tmux ls 2>/dev/null | grep ocp-tui- || echo "clean"` → `clean`.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add lib/tui/session.mjs test-features.mjs
|
||||
git commit -m "feat(tui): runTuiTurn — interactive session driver (boot/trust/paste/Enter/read/teardown)"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## PR-3 — Wiring into `server.mjs`
|
||||
|
||||
Gate TUI-mode behind one env flag. Default path (`CLAUDE_TUI_MODE` unset) stays byte-for-byte identical.
|
||||
|
||||
### Task 6: env consts + `streamStringAsSSE` DRY refactor
|
||||
|
||||
**Files:**
|
||||
- Modify: `server.mjs` (env consts near `:275`; refactor cache-replay block `:1524`–`:1539` into a helper near `:1023`)
|
||||
|
||||
- [ ] **Step 1: Add TUI env consts + import** (near the other `const ... = process.env...` at `server.mjs:258`–`:275`)
|
||||
|
||||
```js
|
||||
import { runTuiTurn, reapStaleTuiSessions } from "./lib/tui/session.mjs";
|
||||
|
||||
// TUI-mode (subscription-pool bridge). Opt-in; default OFF keeps stream-json path.
|
||||
// Authority: docs/adr/0007-tui-interactive-mode.md.
|
||||
const TUI_MODE = process.env.CLAUDE_TUI_MODE === "true";
|
||||
const TUI_WALLCLOCK_MS = parseInt(process.env.CLAUDE_TUI_WALLCLOCK_MS || "120000", 10);
|
||||
const TUI_CWD = process.env.OCP_TUI_CWD || `${process.env.HOME}/.ocp-tui/work`;
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Extract the chunked-SSE-replay into a reusable helper** (near `completionResponse` at `:1023`)
|
||||
|
||||
```js
|
||||
// Replay a complete string as a chunked SSE stream (80 codepoints/chunk).
|
||||
// Extracted from the cache-hit replay block so TUI-mode streaming reuses it.
|
||||
function streamStringAsSSE(res, id, model, content) {
|
||||
const created = Math.floor(Date.now() / 1000);
|
||||
res.writeHead(200, { "Content-Type": "text/event-stream", "Cache-Control": "no-cache", "Connection": "keep-alive", "X-Accel-Buffering": "no" });
|
||||
sendSSE(res, { id, object: "chat.completion.chunk", created, model, choices: [{ index: 0, delta: { role: "assistant" }, finish_reason: null }] });
|
||||
const CHUNK = 80;
|
||||
const codepoints = Array.from(content);
|
||||
for (let i = 0; i < codepoints.length; i += CHUNK) {
|
||||
sendSSE(res, { id, object: "chat.completion.chunk", created, model, choices: [{ index: 0, delta: { content: codepoints.slice(i, i + CHUNK).join("") }, finish_reason: null }] });
|
||||
}
|
||||
sendSSE(res, { id, object: "chat.completion.chunk", created, model, choices: [{ index: 0, delta: {}, finish_reason: "stop" }] });
|
||||
res.write("data: [DONE]\n\n");
|
||||
res.end();
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Point the cache-hit streaming replay (`:1524`–`:1539`) at the helper** (DRY — behavior identical)
|
||||
|
||||
Replace the inline block inside `if (stream) { ... }` of the cache hit with:
|
||||
```js
|
||||
if (stream) {
|
||||
const id = `chatcmpl-${randomUUID()}`;
|
||||
streamStringAsSSE(res, id, model, cached.response);
|
||||
return;
|
||||
} else {
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run the full suite to verify no regression**
|
||||
|
||||
Run: `node test-features.mjs 2>&1 | tail -3`
|
||||
Expected: all existing tests PASS (the refactor is behavior-preserving; cache-replay covered by existing D3 tests).
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add server.mjs
|
||||
git commit -m "refactor(server): extract streamStringAsSSE helper + add TUI env consts"
|
||||
```
|
||||
|
||||
### Task 7: `callClaudeTui` + dispatch gates
|
||||
|
||||
**Files:**
|
||||
- Modify: `server.mjs` (new `callClaudeTui` near `callClaude:735`; gates at the buffered dispatch `:1563`/`:1594` and streaming dispatch `:1551`)
|
||||
|
||||
- [ ] **Step 1: Add `callClaudeTui`** (near `callClaude`, after `:800`)
|
||||
|
||||
```js
|
||||
// TUI-mode upstream: drive an interactive claude session, return the assistant
|
||||
// text as a string — same contract as callClaude(), so all downstream
|
||||
// (singleflight, cache write-back, completionResponse) is unchanged.
|
||||
// System messages are rendered inline as [System] blocks by messagesToPrompt;
|
||||
// we deliberately do NOT pass --system-prompt in interactive mode to avoid any
|
||||
// flag that could perturb cc_entrypoint classification.
|
||||
function callClaudeTui(model, messages, conversationId, keyName) {
|
||||
const cliModel = MODEL_MAP[model] || model;
|
||||
const prompt = messagesToPrompt(messages); // includes system as [System] inline
|
||||
recordModelRequest(cliModel, prompt.length);
|
||||
return runTuiTurn({
|
||||
prompt, model: cliModel, claudeBin: CLAUDE,
|
||||
home: process.env.HOME, cwd: TUI_CWD, wallclockMs: TUI_WALLCLOCK_MS,
|
||||
}).then((text) => {
|
||||
recordModelSuccess(cliModel, 0);
|
||||
return text;
|
||||
}).catch((err) => {
|
||||
recordModelError(cliModel, false);
|
||||
throw err;
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Gate the buffered dispatch** — at `server.mjs:1563`–`:1597`, replace the two `callClaude(...)` call sites (inside the singleflight closure and the cache-disabled fallback) with a selected upstream:
|
||||
|
||||
Add once, just before the `if (CACHE_TTL > 0 && req._cacheHash)` block (~`:1563`):
|
||||
```js
|
||||
const upstreamCall = TUI_MODE ? callClaudeTui : callClaude;
|
||||
```
|
||||
Then change `await callClaude(model, messages, conversationId, req._authKeyName)` → `await upstreamCall(model, messages, conversationId, req._authKeyName)` at **both** sites (`:1572` and `:1594`).
|
||||
|
||||
- [ ] **Step 3: Gate the streaming dispatch** — at `server.mjs:1551`–`:1553`, branch TUI streaming to buffer-then-replay:
|
||||
|
||||
```js
|
||||
if (stream) {
|
||||
if (TUI_MODE) {
|
||||
// TUI has no token stream; buffer the turn, write-back to cache, replay as chunked SSE.
|
||||
const t0Usage = Date.now();
|
||||
try {
|
||||
const content = await callClaudeTui(model, messages, conversationId, req._authKeyName);
|
||||
if (CACHE_TTL > 0 && req._cacheHash) {
|
||||
try { setCachedResponse(req._cacheHash, model, content); } catch (e) { logEvent("error", "cache_write_failed", { error: e.message }); }
|
||||
}
|
||||
const id = `chatcmpl-${randomUUID()}`;
|
||||
streamStringAsSSE(res, id, model, content);
|
||||
try { recordUsage({ keyId: req._authKeyId, keyName: req._authKeyName, model, promptChars: messages.reduce((a, m) => a + (typeof m.content === "string" ? m.content.length : JSON.stringify(m.content).length), 0), responseChars: content.length, elapsedMs: Date.now() - t0Usage, success: true }); } catch {}
|
||||
return;
|
||||
} catch (err) {
|
||||
if (res.headersSent || res.writableEnded || res.destroyed) { try { res.end(); } catch {}; return; }
|
||||
const safeMessage = (err.message || "Internal error").replace(/\/[\w/.\-]+/g, "[path]");
|
||||
return jsonResponse(res, 500, { error: { message: safeMessage, type: "proxy_error" } });
|
||||
}
|
||||
}
|
||||
// Default: real stream-json streaming, unchanged.
|
||||
return callClaudeStreaming(model, messages, conversationId, res, { keyId: req._authKeyId, keyName: req._authKeyName, cacheHash: req._cacheHash });
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Verify default path is untouched + TUI path selected only by flag**
|
||||
|
||||
Run: `CLAUDE_TUI_MODE= node -e "process.env.CLAUDE_TUI_MODE; import('./server.mjs')" 2>&1 | head -1 || true`
|
||||
Then the regression suite: `node test-features.mjs 2>&1 | tail -3`
|
||||
Expected: all PASS (no test sets `CLAUDE_TUI_MODE`, so `upstreamCall === callClaude` and streaming uses `callClaudeStreaming` — identical to today).
|
||||
|
||||
Live end-to-end (PI231, after Task 8 setup): with `CLAUDE_TUI_MODE=true` start OCP and `curl` both `stream:false` and `stream:true`:
|
||||
```bash
|
||||
curl -s localhost:3456/v1/chat/completions -H "Authorization: Bearer <key>" \
|
||||
-d '{"model":"claude-haiku-4-5-20251001","messages":[{"role":"user","content":"say PONG"}]}' | head
|
||||
```
|
||||
Expected: a normal OpenAI completion whose content contains `PONG`. Cross-check on PI231 that the spawned `claude` had no `-p`/`--output-format` (`ps -ef | grep claude`).
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add server.mjs
|
||||
git commit -m "feat(tui): gate interactive TUI upstream behind CLAUDE_TUI_MODE (buffered + streaming)"
|
||||
```
|
||||
|
||||
### Task 8: reaper hook at boot + ADR + README + CHANGELOG
|
||||
|
||||
**Files:**
|
||||
- Modify: `server.mjs` (boot block — call `reapStaleTuiSessions()` once on startup when `TUI_MODE`)
|
||||
- Create: `docs/adr/0007-tui-interactive-mode.md`
|
||||
- Modify: `README.md`, `CHANGELOG.md`
|
||||
|
||||
- [ ] **Step 1: Reaper on boot** (in the server start/`listen` block)
|
||||
|
||||
```js
|
||||
if (TUI_MODE) {
|
||||
try { const n = reapStaleTuiSessions(); if (n) logEvent("info", "tui_reaped_stale_sessions", { count: n }); } catch {}
|
||||
console.log(` TUI-mode: ON (interactive claude → cc_entrypoint=cli). cwd=${TUI_CWD} wallclock=${TUI_WALLCLOCK_MS}ms`);
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Write ADR 0007** — `docs/adr/0007-tui-interactive-mode.md`
|
||||
|
||||
Context: 2026-06-15 billing split routes by `cc_entrypoint`; `-p`/`--output-format` ⇒ `sdk-cli` (Agent SDK credit pool, ~$20 on Pro = unusable). Decision: opt-in interactive driver ⇒ `cli` (subscription pool). Authority: spec §1/§4, claude v2.1.158. Scope: A-path single-user; MCP hard-disabled via `--strict-mcp-config`. Kill-switch: unset `CLAUDE_TUI_MODE` → stream-json path restored. Consequences: no token streaming (buffered+replayed); grey-area, billing unmeasurable until 6/15; reaper + tmux-prefix coexistence rules.
|
||||
|
||||
- [ ] **Step 3: README** — add `CLAUDE_TUI_MODE`, `CLAUDE_TUI_WALLCLOCK_MS`, `OCP_TUI_CWD` to the env-var table; add a "Subscription-pool (TUI) mode" section (what it is, opt-in, the 6/15 rationale, no-streaming caveat, the one-time `mkdir -p ~/.ocp-tui/work` + tmux dependency, and the `CLAUDE_TUI_MODE` unset kill-switch).
|
||||
|
||||
- [ ] **Step 4: CHANGELOG** — Unreleased: `feat(tui): opt-in CLAUDE_TUI_MODE — serve via interactive claude (cc_entrypoint=cli / subscription pool); default stream-json path unchanged.`
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add server.mjs docs/adr/0007-tui-interactive-mode.md README.md CHANGELOG.md
|
||||
git commit -m "feat(tui): boot reaper + ADR 0007 + README + CHANGELOG (TUI-mode docs)"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Integration & canary (post-implementation, on PI231)
|
||||
|
||||
1. Stop the OLP test instance (`:4567`) — clean shared OAuth + no tmux collision.
|
||||
2. `git clone`/checkout this branch on PI231, `mkdir -p ~/.ocp-tui/work`, start OCP on `:3456` with `CLAUDE_TUI_MODE=true`.
|
||||
3. Run the live driver suite: `OCP_TUI_LIVE=1 node test-features.mjs`.
|
||||
4. End-to-end `curl` (buffered + streaming) through OCP; confirm spawned `claude` carries no `-p`/`--output-format`.
|
||||
5. **Pre-6/15 deliverable = here.** Billing measurement waits for 6/15; document the kill-switch (unset `CLAUDE_TUI_MODE`).
|
||||
|
||||
---
|
||||
|
||||
## Self-Review (against spec + the OCP-first execution review)
|
||||
|
||||
- **Spec coverage:** transcript path formula (§4 → Task 1), parsing/terminal/extract (§4 → Task 2), polling reader + wall-clock cap + no-quiescence (§4.3 → Task 3), submission recipe file→paste→Enter (§5/T3 → Task 5), trust-dialog handling (§5.2 → Task 5), MCP disable `--strict-mcp-config` (§5.2/T6 → Tasks 5 & buildTuiCmd), string-contract drop-in (→ Tasks 6–7), kill-switch + default-path-sacred (→ Task 7 Step 4), coexistence prefix + reaper (→ Tasks 4 & 8). ✅
|
||||
- **Review findings folded:** OCP-first string match (Task 7); no ephemeral-home, real-home + scratch cwd (scope §); reader-only sharing, driver forked (file table); tmux prefix + scoped reaper + never-both-on-OAuth (Task 4, Integration §1); `TIMEOUT=600000 > 120s` cap verified (no SIGKILL-mid-turn); `--strict-mcp-config` added (Task 5); provenance jaekwon-park (Why §). OLP-sync deferred. ✅
|
||||
- **Placeholder scan:** none — every code step carries real code; every run step an exact command + expected output. ✅
|
||||
- **Type consistency:** `runTuiTurn`/`reapStaleTuiSessions`/`SESSION_PREFIX` exported in Task 4–5 match imports in Task 6–8; `streamStringAsSSE(res, id, model, content)` defined Task 6, used Tasks 6–7; `callClaudeTui(model, messages, conversationId, keyName)` mirrors `callClaude`'s signature. ✅
|
||||
- **Open item for integration:** confirm on PI231 that the seeded `~/.claude.json` is unnecessary for real-home A (onboarding already complete); if a bypass-permissions dialog *does* appear in real home, add a one-line seed step (`bypassPermissionsModeAccepted:true`) — but the driver already answers the trust dialog defensively, so the turn still completes.
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,193 @@
|
||||
# Design: SSE Heartbeat on Streaming Path
|
||||
|
||||
**Issue:** [#47](https://github.com/dtzp555-max/ocp/issues/47)
|
||||
**Date:** 2026-04-25
|
||||
**Status:** Draft (awaiting maintainer approval)
|
||||
**Target version:** v3.12.0 (minor — new opt-in feature + env var)
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Add an opt-in idle-watchdog heartbeat to OCP's streaming response. When enabled, OCP emits an SSE comment frame (`: keepalive\n\n`) whenever the stream has been idle for a configurable interval. Timer resets on every real frame. Covers both pre-first-byte and mid-stream silent windows. Default **disabled**. Zero behavior change for existing deployments on upgrade.
|
||||
|
||||
Companion tweak: `X-Accel-Buffering: no` response header added to both SSE header sites so heartbeats survive nginx-default proxy buffering.
|
||||
|
||||
## Motivation
|
||||
|
||||
Per [#47](https://github.com/dtzp555-max/ocp/issues/47): when `claude -p` takes a long time to respond (processing large contexts, or executing long tool calls that pause the token stream for 30s–5min), OCP emits no bytes to the downstream client for up to 600s. The caller cannot distinguish "slow but alive" from "hung." A recent incident reported 15 consecutive 600s silent waits cascading into a 2-hour downstream gateway outage.
|
||||
|
||||
SSE heartbeats at the application layer let a caller observe liveness without OCP introducing any new client-killing timer.
|
||||
|
||||
## Key decisions (with rationale)
|
||||
|
||||
Six decisions were fixed during brainstorming. Each is presented as "decision → rationale" so future readers can judge whether a decision is still load-bearing.
|
||||
|
||||
### D1. Coverage: whole-stream with idle-watchdog reset-on-byte
|
||||
|
||||
A per-request timer starts when SSE headers are written and resets on every real `sendSSE()` call. Heartbeat fires only during genuine idle windows — never during healthy token bursts.
|
||||
|
||||
**Rationale.** The `server.mjs:8-10` comment documents Claude tool-use pauses as "30s-5min pauses in the token stream." This means silent windows happen both pre-first-byte AND mid-stream. Covering only one of the two windows means re-opening this issue in three months when a different user reports the uncovered case. The reset-on-byte discipline is a ~2-LOC discipline (`clearTimeout` + `setTimeout` inside `sendSSE`) and is the standard "idle watchdog" pattern.
|
||||
|
||||
### D2. Frame format: SSE comment (`: keepalive\n\n`)
|
||||
|
||||
**Rationale.** Per SSE spec / MDN, lines starting with `:` are comments and MUST be ignored by conforming parsers. This is the maximally inert shape we can emit. Alternatives considered:
|
||||
- `event: ping` named event — Anthropic's own Messages API uses this, but on an OpenAI-compatible surface, downstream clients don't recognize that event name, so risk of client confusion is higher.
|
||||
- Empty-delta JSON chunk — parser-safe on OpenAI-compatible clients but burns an event id and is less observably "a heartbeat" in logs.
|
||||
|
||||
The known risk with SSE comments is that some SDKs crash on empty comment frames (`openai-go` issue #556). The default-disabled posture (D3) mitigates this: users who opt in can also verify their client tolerates comments. If the comment format turns out to be broken in the wild for common OCP callers, we can add a second format behind `CLAUDE_HEARTBEAT_FORMAT` in a follow-up — **not in this PR**.
|
||||
|
||||
### D3. Default disabled
|
||||
|
||||
`CLAUDE_HEARTBEAT_INTERVAL=0` (meaning disabled) is the default when the env var is unset. Any positive integer enables at that ms interval.
|
||||
|
||||
**Rationale.** Existing deployments see zero byte-shape change on upgrade. Users who have pingvvino's problem set the env var and get the fix. This is a reversible posture: once field evidence shows comment frames are safe across current OCP callers, a future minor can flip the default to `30000`.
|
||||
|
||||
### D4. Header relocation: `ensureHeaders()` moves earlier
|
||||
|
||||
Currently `ensureHeaders()` is invoked inside the `proc.stdout` handler, so SSE headers are written only on first byte from claude CLI. This PR moves the `ensureHeaders()` call to immediately after successful `spawn()` return.
|
||||
|
||||
**Rationale.** You cannot emit SSE frames before sending SSE headers. For heartbeats to cover the pre-first-byte silent window (pingvvino's "processing large contexts" case), headers must be sent earlier. Behavioral consequence: the narrow "spawn succeeded but subprocess erroneously died before any byte" branch — currently a JSON error response — becomes an SSE error event + `[DONE]` + `res.end()`. Pre-spawn errors (before `spawn()`) still return JSON, unchanged. The affected path is rare (claude CLI either spawns or it doesn't).
|
||||
|
||||
### D5. Transport-layer buffering hint: `X-Accel-Buffering: no`
|
||||
|
||||
Added to both SSE response header sites (real-streaming `ensureHeaders()` and the cache-hit simulated-streaming header write).
|
||||
|
||||
**Rationale.** nginx (and many LBs / Cloudflare) default to buffering proxied responses. Without this header, heartbeat bytes may accumulate in an upstream buffer and never reach the client, defeating the feature silently. This header is a nginx-specific hint; other stacks ignore it. 1 line per site, 2 sites total, indistinguishable-from-no-op for stacks that don't use it.
|
||||
|
||||
### D6. Observability: single log line per affected request
|
||||
|
||||
On the first heartbeat fire within a request, emit a structured log entry (`logEvent("info", "heartbeat_active", { session, intervalMs })`). No log spam for subsequent fires in the same request.
|
||||
|
||||
**Rationale.** For a first-mover feature the question "did the heartbeat actually work for that hung request?" needs to be answerable from the existing `/logs` endpoint alone, without external tooling. One line per affected request gives proof-of-life without polluting the log stream during healthy traffic (where the timer resets and never fires).
|
||||
|
||||
## Architecture
|
||||
|
||||
Single-file change in `server.mjs`. One new helper plus small patches to existing sites.
|
||||
|
||||
```
|
||||
startHeartbeat(res, intervalMs, sessionId) → { reset(), stop() }
|
||||
if intervalMs <= 0: return no-op handle
|
||||
internal: handle = setTimeout(intervalMs, onFire)
|
||||
onFire():
|
||||
res.write(": keepalive\n\n")
|
||||
if !hasFired: logEvent("info", "heartbeat_active", { session, intervalMs }); hasFired = true
|
||||
handle = setTimeout(intervalMs, onFire)
|
||||
reset(): clearTimeout(handle); handle = setTimeout(intervalMs, onFire)
|
||||
stop(): clearTimeout(handle); handle = null
|
||||
```
|
||||
|
||||
Handle created once per streaming request. `sendSSE()` calls `heartbeat.reset()` before its `res.write()`. All exit paths call `heartbeat.stop()`.
|
||||
|
||||
## Components and LOC budget
|
||||
|
||||
| Location | Change | Est LOC |
|
||||
|---|---|---|
|
||||
| `server.mjs` env block | Parse `CLAUDE_HEARTBEAT_INTERVAL` | 1 |
|
||||
| `server.mjs` new `startHeartbeat()` function | Per spec above | 12 |
|
||||
| `server.mjs:565-579` `ensureHeaders()` | Add `"X-Accel-Buffering": "no"` to header object | 1 |
|
||||
| `server.mjs:~548-554` streaming entry | Move `ensureHeaders()` call to post-spawn; create heartbeat handle | 3 (net) |
|
||||
| `server.mjs:669` `sendSSE()` | Accept optional `hb` param; call `hb?.reset()` before `res.write()` | 2 |
|
||||
| `server.mjs` streaming exit hooks (proc 'close', proc 'error', req 'close') | Call `hb.stop()` | 3 |
|
||||
| `server.mjs:~610-611` pre-first-byte error branch | If headers sent, SSE error + `[DONE]` + `res.end()` instead of JSON | 3 |
|
||||
| `server.mjs:1171` cache-hit header write | Add `"X-Accel-Buffering": "no"` | 1 |
|
||||
| `README.md` env var table | One new row | 1 |
|
||||
| `README.md` new short section | "Streaming heartbeat" paragraph + nginx note | 5 |
|
||||
| `CHANGELOG.md` new v3.12.0 entry | Features + Config additions | 4 |
|
||||
| `package.json` + `ocp-plugin/package.json` + `ocp-plugin/openclaw.plugin.json` | Version bump 3.11.1 → 3.12.0 | 3 |
|
||||
|
||||
**Estimated total: ~40 lines.** This is 5–10 lines over the 25–35 budget set in brainstorming. The overshoot is accounted for by D4 (header relocation + SSE error branch) and D5 (X-Accel-Buffering at two sites). Reviewer may reject if actual code lands materially larger than ~45 lines of server.mjs code excluding docs and version files.
|
||||
|
||||
## Data flow (streaming request, heartbeat enabled)
|
||||
|
||||
1. Client sends `POST /v1/chat/completions` with `stream=true`.
|
||||
2. Cache miss → `callClaudeStreaming()` invoked.
|
||||
3. `spawn()` claude subprocess succeeds → `ensureHeaders(res)` writes SSE headers including `X-Accel-Buffering: no`.
|
||||
4. `const hb = startHeartbeat(res, HEARTBEAT_INTERVAL, sessionId)` arms the watchdog (no-op if interval is 0).
|
||||
5. Watchdog ticks after `HEARTBEAT_INTERVAL` ms of idle. On fire: `: keepalive\n\n` out; first-fire logs once; re-arm.
|
||||
6. Every real `sendSSE()` write calls `hb.reset()` — cancels and re-arms timer.
|
||||
7. Healthy token bursts → heartbeat never fires.
|
||||
8. Tool-use pause → timer elapses → heartbeat fires → client stays alive → re-arm → repeat until next chunk arrives.
|
||||
9. On proc 'close' (success / `[DONE]`) / proc 'error' / req 'close' / `CLAUDE_TIMEOUT` kill → `hb.stop()`.
|
||||
|
||||
## Error handling
|
||||
|
||||
- **Client disconnect mid-heartbeat.** `req.on('close')` fires → `hb.stop()`. Any in-flight write becomes a no-op / emits `'error'` on `res`; existing code already tolerates this.
|
||||
- **`CLAUDE_TIMEOUT` (600s) fires mid-request.** Existing timeout handler SIGTERM's subprocess → proc 'close' → `hb.stop()`. This PR does **not** fix the separate issue that the current timeout path does not `res.end()` or emit an SSE error frame; that is documented as a separate issue.
|
||||
- **`spawn()` throws synchronously.** Heartbeat never started. Existing JSON error response unchanged.
|
||||
- **`spawn()` succeeds, subprocess errors before first byte.** Headers have already been written (per D4). Branch emits an SSE error event + `[DONE]` + `res.end()` instead of a JSON error. Documented behavior change.
|
||||
|
||||
## Testing plan
|
||||
|
||||
OCP has no unit test framework beyond `test-features.mjs`. Verification is manual + cloud-backed.
|
||||
|
||||
### Manual local smoke test
|
||||
|
||||
1. Set `CLAUDE_HEARTBEAT_INTERVAL=5000` (5s for easy observation) and start OCP.
|
||||
2. Issue a streaming completion with a prompt that triggers a tool-use pause (e.g., ask for a large file read or long reasoning):
|
||||
```
|
||||
curl -N http://localhost:3456/v1/chat/completions \
|
||||
-H "Authorization: Bearer $OCP_KEY" \
|
||||
-d '{"model":"claude-opus-4-7","stream":true,
|
||||
"messages":[{"role":"user","content":"read the attached 200KB text and summarize"}]}'
|
||||
```
|
||||
3. Confirm `: keepalive` comment lines appear in the raw response during the pause, at ~5s cadence.
|
||||
4. Confirm `/logs` shows exactly one `heartbeat_active` entry for the request.
|
||||
5. With `CLAUDE_HEARTBEAT_INTERVAL=0` (default), confirm no heartbeats and no log line.
|
||||
|
||||
### Cloud-backed test run (pre-push, required)
|
||||
|
||||
Per project feedback, tests must pass before any push to the public repo. Options, in preference order:
|
||||
|
||||
1. **GitHub Actions** — add a temporary smoke workflow (or piggyback on an existing one) that runs `test-features.mjs` against a sandboxed claude mock. Not viable if `test-features.mjs` requires a real claude CLI auth.
|
||||
2. **Remote Linux test host via cc-chat handoff** — push feature branch, instruct a cloud machine with claude CLI installed to run the manual steps above, capture output, return verdict via cc-chat.
|
||||
3. **Docker/compose locally** — if the maintainer has Docker available, `docker-compose.yml` is present and can be extended.
|
||||
|
||||
The implementation subagent and the reviewer subagent MUST include the chosen verification evidence in the PR body (command + output excerpt, sanitized of any identifiers) before the PR is opened for merge review.
|
||||
|
||||
### Downstream-parser compatibility
|
||||
|
||||
Before merging, verify at least one real downstream client (OCP's own `ocp-connect`, plus — if feasible — the current OpenClaw gateway) does not crash on comment frames. If a target client crashes, either (a) adjust the default to 0 and document the incompatibility, or (b) scope a follow-up PR for `CLAUDE_HEARTBEAT_FORMAT=empty-delta` as an alternative.
|
||||
|
||||
## Privacy preflight (for public-repo push)
|
||||
|
||||
Before `gh pr create` or any `git push` to `dtzp555-max/ocp`, run the following scan on the full diff (`git diff origin/main...HEAD`):
|
||||
|
||||
1. **Use OCP's PR template privacy self-check** — the `.github/PULL_REQUEST_TEMPLATE.md` Privacy self-check section is the canonical list. Fill every checkbox.
|
||||
2. **Run `.gitleaks.toml` via gitleaks if available.**
|
||||
3. **Manual grep on the diff** for these patterns (sanitize any hits before commit):
|
||||
- Personal names in any language (check commit-author trailers especially — `Co-Authored-By` lines have leaked names before).
|
||||
- Email addresses beyond automated placeholders (`noreply@*`).
|
||||
- Local paths like `/Users/<name>/`, `/home/<name>/`, `C:\Users\<name>\` — replace with `$HOME/` or `~/`.
|
||||
- Machine hostnames — use role-based names or generic descriptors.
|
||||
- IPs, internal URLs, tailnet names.
|
||||
4. **Log samples and test output** pasted into spec / README / CHANGELOG / PR body must be sanitized pre-paste, not pre-push. This spec doc itself was drafted under this discipline.
|
||||
|
||||
Historical reference: PR #43 / postmortem #44 (2026-04-22) scrubbed a prior leak and established the current apparatus. The user has explicitly flagged this as a scar to avoid re-treading for this PR.
|
||||
|
||||
## Scope lock (out of scope)
|
||||
|
||||
- `server.mjs:480-489` `CLAUDE_TIMEOUT` dangling-client behavior (no `res.end()` / no SSE error frame on timeout kill) — **will be filed as a separate issue** before this PR opens.
|
||||
- Issue #41 (handleSessionFailure deletes on resume only) — separate.
|
||||
- Issue #42 (SESSION_TTL and `lastUsed` interaction) — separate.
|
||||
- Any new first-byte / idle / adaptive-tier timeout logic — explicitly forbidden per the v3.3 lesson (`server.mjs:8-11`, commit 3843ec8).
|
||||
- Non-streaming chat path — no HTTP-level fix possible per [#47](https://github.com/dtzp555-max/ocp/issues/47)'s own conclusion.
|
||||
- Named-event format (`event: ping`) or empty-delta JSON chunk variants — possible follow-up PR if field evidence shows comment frames break real clients.
|
||||
- Changes to `CLAUDE_TIMEOUT` default — unchanged (stays 600s).
|
||||
- Circuit breaker revival — explicitly forbidden.
|
||||
|
||||
## ALIGNMENT.md disposition
|
||||
|
||||
`cli.js` does not itself emit SSE heartbeat frames — claude CLI speaks newline-delimited JSON to stdout, not SSE. SSE is an OCP-owned translation layer. Per `ALIGNMENT.md` Rule 2 / `AGENTS.md` ("OCP forwards, observes, and multiplexes traffic that cli.js already emits"): heartbeats are a translation-layer response-shaping concern, not a new endpoint and not a behavior mimicry. The PR body will state this explicitly in the `cli.js` citation checkbox and reference this design doc.
|
||||
|
||||
## IDR (Iron Rule 11) disposition
|
||||
|
||||
Single PR. Scope is one feature (SSE heartbeat) × one layer (streaming response formatting) × one severity (minor opt-in addition). Release-kit companion files (version bump, CHANGELOG, README) are bundled with the code change per the explicit Iron Rule 11 example ("版本 bump 相关的小改动 + README + CHANGELOG 可以同 PR"). The separate filings for the dangling-client bug (`server.mjs:480-489`) and any follow-up heartbeat format variants are IDR-compliant — each lands as its own PR.
|
||||
|
||||
## Related
|
||||
|
||||
- Issue: [#47](https://github.com/dtzp555-max/ocp/issues/47)
|
||||
- Prior timeout scar: commit 3843ec8 (v3.3.0 "simplify timeout to single CLAUDE_TIMEOUT")
|
||||
- Prior privacy scar: PR [#43](https://github.com/dtzp555-max/ocp/pull/43) / postmortem [#44](https://github.com/dtzp555-max/ocp/issues/44)
|
||||
- Constitution: `ALIGNMENT.md`
|
||||
- Project instructions: `AGENTS.md`, `CLAUDE.md`
|
||||
@@ -0,0 +1,147 @@
|
||||
# Design: Response Cache Upgrade (Per-Key Isolation, cache_control Bypass, Chunked Stream Replay, Singleflight)
|
||||
|
||||
**Date:** 2026-05-07
|
||||
**Status:** Draft (awaiting maintainer approval)
|
||||
**Target version:** v3.13.0 (minor — internal correctness/concurrency improvements; no new public env vars or endpoints)
|
||||
**Driving ADR:** [ADR 0005 — No Multi-Provider](../../adr/0005-no-multi-provider.md), decision §3 ("Cache improvements are in scope")
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
OCP already has a response cache (`keys.mjs:296` `cacheHash` / `keys.mjs:311` `getCachedResponse` / `keys.mjs:324` `setCachedResponse`), wired into the proxy core at `server.mjs:1220` (non-streaming path read), `server.mjs:1227` (cache-hit-on-streaming-request replay), and `server.mjs:683` (streaming write-back). Today it has four functional gaps. This PR pair closes all four, in two minimum-reviewable units, **without changing the public API surface**.
|
||||
|
||||
| Gap | Impact today | Fix lands in |
|
||||
|---|---|---|
|
||||
| All keys share one cache pool | Key A's cache hit can leak Key B's prompt response | PR-A |
|
||||
| Anthropic `cache_control` markers not detected | OCP cache may interfere with Anthropic prompt caching that the user explicitly requested | PR-A |
|
||||
| Stream cache hit replays whole content in one SSE chunk | Downstream renders all-at-once; some SDKs misbehave on huge single deltas | PR-A |
|
||||
| Concurrent identical cache misses all spawn `cli.js` independently | Cache stampede: N requests → N spawns → N billable calls | PR-B |
|
||||
|
||||
---
|
||||
|
||||
## Constitutional alignment (ALIGNMENT.md)
|
||||
|
||||
**`cli.js` does not perform response caching at the proxy layer.** The OCP response cache is a value-add operation that exists only inside OCP, between the wire (clients ↔ OCP) and the spawn (OCP ↔ `cli.js`). It does not introduce, rename, or alter any endpoint, header, request field, or response field that `cli.js` emits or expects. Cache hits return content byte-identical to what `cli.js` returned on the original miss, with the same `chat.completion` / `chat.completion.chunk` shape — **no client-observable wire shape change**.
|
||||
|
||||
This PR pair extends the existing cache (introduced in earlier commits) without expanding its surface. No new endpoints. No new headers. No new env vars exposed publicly (we add internal counters readable via the existing `/cache/stats` endpoint, but the response shape only gains numeric fields, not new structural fields).
|
||||
|
||||
Per Rule 1 / Rule 5: every commit body in this PR pair will state the absence of `cli.js` reference explicitly and justify scope under Rule 2's value-add carve-out for non-wire-affecting proxy operations.
|
||||
|
||||
---
|
||||
|
||||
## Key decisions (with rationale)
|
||||
|
||||
### D1. Per-key isolation via hash input, not schema column
|
||||
|
||||
`cacheHash` gains an optional `keyId` input. Distinct `keyId` values produce distinct hashes for the same prompt, so SQLite-level isolation falls out for free without a schema change.
|
||||
|
||||
**Rationale.** Adding a `key_id` column to `response_cache` requires either (a) dropping the existing `hash UNIQUE` index and replacing with a composite `(hash, key_id) UNIQUE`, which SQLite cannot do via plain `ALTER TABLE` and would require a table-rebuild migration, or (b) tolerating duplicate `hash` rows, which contradicts the existing schema comment and breaks `setCachedResponse`'s `ON CONFLICT(hash)` upsert clause.
|
||||
|
||||
The hash-input approach is reversible (we can switch to a schema column later if analytics across keys becomes a real need) and zero-risk on the SQL plane. The trade-off — losing the ability to query "which keys have cached this prompt?" — has no current consumer.
|
||||
|
||||
**Hash input format.** `cacheHash` prepends a version tag and key tag before the existing inputs:
|
||||
|
||||
```
|
||||
v2|k:<keyId or "anon">|<model>|...rest as today
|
||||
```
|
||||
|
||||
The `v2` prefix means existing v1-format rows in the cache table no longer hash-match any new request. They are abandoned, not deleted; the existing TTL-based `clearCache(CACHE_TTL)` cleanup interval at `server.mjs:185` reaps them within one TTL window. **No migration step is needed.** This is acceptable because the cache is by definition ephemeral and best-effort.
|
||||
|
||||
**Anonymous fallback.** When the request has no authenticated key (`req._authKeyId === undefined`), `keyId` is `"anon"`. Anonymous-mode users (PROXY_ANONYMOUS_KEY or no auth) share one anonymous pool, which preserves the only legitimate today-multi-user use case (a household running OCP without per-user keys). If this becomes a problem we can add per-IP scoping later, but anonymous-pool sharing is acceptable for v1 because anonymous mode is fundamentally a trust-everyone-on-LAN posture.
|
||||
|
||||
### D2. `cache_control` bypass: detect anywhere, skip OCP cache entirely
|
||||
|
||||
If any element in `messages` (top-level or nested in `content` arrays) carries a `cache_control` field, OCP sets `req._cacheHash = null` and skips both lookup and write-back.
|
||||
|
||||
**Rationale.** Anthropic's [prompt caching](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching) is opt-in by client-side annotation. A user who annotates `cache_control: { type: "ephemeral" }` is explicitly requesting that *Anthropic's* cache serve the call (and is paying the reduced cache-read pricing). Layering OCP's response cache on top in this case is wrong on two counts:
|
||||
|
||||
1. The user's intent is "cache at provider, not at proxy." OCP overruling that intent silently is the same drift family as the 2026-04-11 incident — proxy invents behavior the upstream surface doesn't request.
|
||||
2. OCP cache hits would make `usage.cache_read_input_tokens` (the client-observable signal that prompt caching worked) appear inconsistent — sometimes present, sometimes absent — depending on whether OCP cached.
|
||||
|
||||
Detection is purely structural: walk `messages`, for each `m` check `m.cache_control` (rare top-level form) and if `m.content` is an array, check each part. No semantic interpretation; if the field is present, we bypass.
|
||||
|
||||
**Implementation site.** A small helper `hasCacheControl(messages)` exported from `keys.mjs`, called in `handleChatCompletions` immediately before the existing `cacheHash` call. If it returns true, we skip the cache-lookup branch entirely.
|
||||
|
||||
### D3. Chunked stream replay (80 chars/chunk, no artificial delay)
|
||||
|
||||
Today's cache-hit-on-streaming-request branch (`server.mjs:1227–1237`) sends the entire cached content in a single `delta.content` chunk. This works for spec-compliant SSE clients but visibly degrades the UX (no incremental render) and has tripped at least one buggy SDK in the wild that assumes deltas are small.
|
||||
|
||||
The fix splits cached content into ~80-character substrings, each sent as a separate `chat.completion.chunk` SSE event. **No artificial delay between chunks** — they ship as fast as `res.write` accepts. This preserves OCP's "ship as fast as possible" disposition; we are simulating *the chunk shape* of streaming, not the *latency*.
|
||||
|
||||
**Why 80 chars?** Compromise: small enough that even a multi-paragraph cached response yields >5 chunks (visible incremental render), large enough that even a 4 KB response only produces 50 chunks (not 4000 single-char events). Tunable later via internal constant; not exposed as env var per scope-creep avoidance.
|
||||
|
||||
**Boundary safety.** UTF-8 multibyte characters: we slice by `Array.from(content)` (so each iteration step is a full code point) and group every 80 code points. This avoids producing invalid UTF-8 mid-character.
|
||||
|
||||
### D4. Singleflight stampede protection: in-process Map, all-or-nothing failure
|
||||
|
||||
`keys.mjs` exports `singleflight(hash, fn)`. An in-memory `Map<hash, Promise>` deduplicates concurrent identical cache-miss flows. The first request executes `fn()`; concurrent requests with the same hash receive the same promise. When the promise settles (resolve or reject), the map entry is deleted.
|
||||
|
||||
**Rationale (single-process scope).** OCP runs as a single Node.js process per host. A `Map` is sufficient. Adding Redis or another shared store would be the start of a multi-instance evolution, which is out of scope per ADR 0005 (OCP is a personal power tool, not a horizontally-scaled SaaS).
|
||||
|
||||
**All-or-nothing failure semantics.** When the leader's `fn()` rejects, all followers receive the same rejection. The alternative — letting followers retry independently after a leader failure — risks N retries of an already-broken upstream, which is exactly what stampede protection was meant to prevent. Followers can retry at the *next* request, with idle backoff handled by the client. This matches Go's `golang.org/x/sync/singleflight` reference behavior.
|
||||
|
||||
**Streaming caveat.** Singleflight wraps the *non-streaming* code path only in PR-B. For streaming, deduplicating concurrent identical streaming requests is materially harder (we'd need to fan out one upstream stream to N downstream connections in real time, with backpressure). It's also a less common case (cache stampedes typically come from non-streaming batch jobs hitting the proxy in parallel). Streaming dedup is **explicitly out of scope** for this PR pair; leave a TODO comment in `callClaudeStreaming` for a future ticket.
|
||||
|
||||
**Map size unboundedness.** In normal operation the map is empty most of the time (entries delete on Promise settlement). Pathological case: an upstream call that hangs forever leaks one Map entry per stuck request. The existing `TIMEOUT` guard on `callClaude` (server.mjs spawn timeout) bounds this — the Promise will reject (timeout) within `TIMEOUT` ms, and the entry clears. No additional sweep needed.
|
||||
|
||||
---
|
||||
|
||||
## PR boundaries
|
||||
|
||||
### PR-A — Foundation (D1 + D2 + D3)
|
||||
|
||||
**Files touched:**
|
||||
- `keys.mjs`: extend `cacheHash` with optional `keyId`/version prefix; add `hasCacheControl(messages)` helper
|
||||
- `server.mjs`: pass `req._authKeyId` to `cacheHash`; check `hasCacheControl` and bypass; chunk cache-hit replay at line 1227–1237
|
||||
- `test-features.mjs`: add cases for keyId isolation, cache_control bypass, chunked replay shape
|
||||
|
||||
**LOC budget:** ~80 production + ~50 test
|
||||
**Risk:** Low — all changes are additive or guard-clause; existing cache behavior preserved when `keyId` defaults to "anon" and no `cache_control` present.
|
||||
**Backward compat:** v1-format hashes naturally orphan; TTL cleanup reaps within one window; no migration script.
|
||||
|
||||
### PR-B — Concurrency (D4)
|
||||
|
||||
**Files touched:**
|
||||
- `keys.mjs`: add `singleflight(hash, fn)` and `getInflightStats()` exports
|
||||
- `server.mjs`: wrap non-streaming cache-miss path through `singleflight`; add inflight count to `/cache/stats` response
|
||||
- `test-features.mjs`: add concurrent-request test that asserts only 1 spawn occurs for N=10 simultaneous identical requests
|
||||
|
||||
**LOC budget:** ~70 production + ~40 test
|
||||
**Risk:** Medium — concurrency code is harder to reason about; mitigation is an explicit test case for the dedup behavior.
|
||||
**Streaming explicitly out of scope:** TODO comment placed in `callClaudeStreaming` for follow-up ticket.
|
||||
|
||||
---
|
||||
|
||||
## Testing strategy
|
||||
|
||||
**Unit-ish (in `test-features.mjs`):**
|
||||
1. `cacheHash` with two different `keyId` values → different hashes
|
||||
2. `cacheHash` v2 prefix present in output (sanity check)
|
||||
3. `hasCacheControl` returns true for top-level `cache_control` and for nested in `content[]`
|
||||
4. `hasCacheControl` returns false for benign messages
|
||||
5. Chunked replay: cached "abcdefgh..." (160 chars) produces 2 deltas
|
||||
|
||||
**Integration (manual smoke before merge):**
|
||||
1. Set `CLAUDE_CACHE_TTL=60000`; create key A and key B; identical prompt from each → both spawn fresh; second-call from same key → cache hit
|
||||
2. Send a message with `cache_control` annotation → OCP logs `cache_skipped: cache_control_present`; no cache write
|
||||
3. Streaming cache hit visibly produces multiple SSE deltas (`curl -N | grep "data: "` shows >1 lines)
|
||||
|
||||
**Concurrent (PR-B only):**
|
||||
1. Spawn 10 simultaneous identical non-streaming requests; assert (via `/cache/stats` inflight peak or via a process spawn counter) only 1 `cli.js` spawn occurred
|
||||
|
||||
---
|
||||
|
||||
## Out of scope (deliberately deferred)
|
||||
|
||||
- **Streaming singleflight** — see D4 streaming caveat. TODO in code.
|
||||
- **Semantic cache** (embedding-based near-match) — needs an embedding provider + vector index. Punt to v3.14+ if there's user demand.
|
||||
- **Cross-process cache** (Redis backend) — violates ADR 0005's "personal power tool" posture.
|
||||
- **Cache versioning by model ID hash** — model upgrades currently invalidate cache organically because model is in the hash; if Anthropic ever silently changes a model's behavior without a model ID bump, that's a separate alignment problem.
|
||||
- **Per-key cache TTL override** — single global TTL (existing `CLAUDE_CACHE_TTL`) is fine; per-key TTL is a knob no one has asked for.
|
||||
|
||||
---
|
||||
|
||||
## Rollback plan
|
||||
|
||||
If either PR introduces a regression, the rollback is a clean git revert. The cache layer is opt-in (default `CLAUDE_CACHE_TTL=0` = disabled), so users who never enabled the cache are unaffected by any cache-layer regression. Users who *had* enabled the cache lose only ephemeral state on revert. No persistent on-disk state is reshaped by this PR pair (we explicitly avoid schema migrations per D1 rationale).
|
||||
@@ -0,0 +1,136 @@
|
||||
Part of [OCP](../README.md) — full troubleshooting manual. The README keeps a slim version with the most common issues and the one-time bootstrap quirks; everything else lives here.
|
||||
|
||||
# Troubleshooting
|
||||
|
||||
The simplest path: ask your AI.
|
||||
|
||||
Paste this prompt:
|
||||
|
||||
```
|
||||
Run `ocp doctor` and follow its `next_action`. Tell me if you hit
|
||||
anything that needs human input.
|
||||
```
|
||||
|
||||
The doctor produces a JSON `next_action` with `ai_executable[]` (commands
|
||||
the agent runs verbatim) and `human_required[]` (steps that need you,
|
||||
typically just OAuth).
|
||||
|
||||
## Manual debugging
|
||||
|
||||
### Setup fails with "claude: command not found"
|
||||
|
||||
`setup.mjs` requires the Claude CLI to be on `PATH`. Install it via the [official guide](https://docs.anthropic.com/en/docs/claude-cli), confirm with `which claude`, then run `claude auth login` before re-running `node setup.mjs`.
|
||||
|
||||
### Setup fails with "EADDRINUSE: port 3456 already in use"
|
||||
|
||||
Something else is already bound to port 3456 — usually an old OCP instance. Check what:
|
||||
|
||||
```bash
|
||||
lsof -nP -iTCP:3456 -sTCP:LISTEN
|
||||
```
|
||||
|
||||
If it's an old OCP process, stop it before re-running setup:
|
||||
|
||||
```bash
|
||||
launchctl bootout gui/$(id -u)/dev.ocp.proxy # macOS launchd
|
||||
systemctl --user stop ocp-proxy # Linux systemd (installed as a --user unit)
|
||||
```
|
||||
|
||||
(There is no `ocp stop` subcommand — the proxy runs as a service, so stopping it goes through the service manager above. `ocp restart` exists for the bounce case.)
|
||||
|
||||
### Setup fails with "node: command not found" or version error
|
||||
|
||||
OCP requires Node.js 22.5+. Install:
|
||||
|
||||
```bash
|
||||
brew install node # macOS
|
||||
# Linux: see https://nodejs.org/en/download for current install commands
|
||||
```
|
||||
|
||||
Confirm with `node --version` (should be ≥ v22.5).
|
||||
|
||||
### Requests fail or agents stuck
|
||||
|
||||
```bash
|
||||
# Clear sessions and restart
|
||||
ocp clear
|
||||
ocp restart
|
||||
|
||||
# If using OpenClaw gateway
|
||||
openclaw gateway restart
|
||||
```
|
||||
|
||||
### Env var change (e.g. `CLAUDE_BIND`, `CLAUDE_CODE_OAUTH_TOKEN`) doesn't take effect after restart
|
||||
|
||||
On **macOS**, `ocp restart` does a full `launchctl bootout` + `bootstrap` of the agent, which **re-reads the plist `EnvironmentVariables`** — so an env change you made (in `~/Library/LaunchAgents/dev.ocp.proxy.plist`) actually takes effect:
|
||||
|
||||
```bash
|
||||
ocp restart
|
||||
```
|
||||
|
||||
This is deliberate: the older `launchctl kickstart -k` only re-execs the process and **reuses launchd's cached environment**, so plist env edits would be silently ignored. If you ever restart the agent by hand, use bootout+bootstrap, not `kickstart -k`:
|
||||
|
||||
```bash
|
||||
launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null
|
||||
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/dev.ocp.proxy.plist
|
||||
```
|
||||
|
||||
Verify the new value reached the running process:
|
||||
|
||||
```bash
|
||||
ps -E -p "$(launchctl print gui/$(id -u)/dev.ocp.proxy 2>/dev/null | awk '/pid =/{print $3}')" | tr ' ' '\n' | grep CLAUDE_
|
||||
```
|
||||
|
||||
On **Linux**, `systemctl --user restart` already re-reads the unit's `EnvironmentFile`, so no special handling is needed.
|
||||
|
||||
### Usage shows "unknown"
|
||||
|
||||
Usually caused by an expired Claude CLI session. Fix:
|
||||
```bash
|
||||
claude auth login
|
||||
ocp restart
|
||||
```
|
||||
|
||||
### Startup log warns "OpenClaw registry out of sync"
|
||||
|
||||
On boot, OCP compares OpenClaw's registered models against [`models.json`](../models.json) and warns if they drift. Cause: someone (or an OpenClaw upgrade) modified `~/.openclaw/openclaw.json` and removed entries OCP expects. Fix:
|
||||
|
||||
```bash
|
||||
node ~/ocp/scripts/sync-openclaw.mjs
|
||||
```
|
||||
|
||||
This is read-only at startup; the warning never blocks the gateway from running.
|
||||
|
||||
### A TUI session vanished right after upgrading OCP
|
||||
|
||||
If you ran a pre-3.21.1 OCP instance and a post-3.21.1 instance on the same host at the same time during an upgrade, the new instance's one-time boot reap can, once, kill an old-format (`ocp-tui-<8hex>`) live TUI session belonging to the still-running old instance — restart the affected session (`ocp restart` or re-run your TUI turn) and it will come back under the new instance's port-scoped naming.
|
||||
|
||||
### OpenClaw shows old models after `ocp update` (v3.10→v3.11 only)
|
||||
|
||||
One-time bootstrap quirk for the v3.10.0 → v3.11.0 jump only — the running shell had the old `cmd_update` cached. Run once manually:
|
||||
|
||||
```bash
|
||||
node ~/ocp/scripts/sync-openclaw.mjs
|
||||
openclaw gateway restart # so OpenClaw re-reads the config
|
||||
```
|
||||
|
||||
Future `ocp update` invocations sync automatically.
|
||||
|
||||
<a id="tui-401"></a>
|
||||
### TUI-mode returns a permanent `Please run /login` 401 (re-login doesn't stick)
|
||||
|
||||
A long-running TUI-mode host can get stuck returning a permanent 401 (`Please run /login · API Error: 401`) that re-login cannot fix.
|
||||
|
||||
**Root cause (two layers):** interactive `claude` **prefers `~/.claude/.credentials.json` over the `CLAUDE_CODE_OAUTH_TOKEN` env var** (this is *unlike* the `-p` path, where the env token wins). So (a) a stale/corrupt `credentials.json` **shadows** the env token — passing the token is not enough on its own; and (b) when claude does use `credentials.json`, its single-use OAuth refresh token can be corrupted (ending up an empty string) by the per-request spawn + `kill-session` teardown racing claude's token rotation. Re-login writes a fresh token, but the next spawn re-corrupts it. Proven live on PI231: *env token passed + broken `credentials.json` present → 401; env token passed + `credentials.json` moved aside → works.*
|
||||
|
||||
**Fix:** set `CLAUDE_CODE_OAUTH_TOKEN` on the OCP host and leave `OCP_TUI_HOME` **unset**. OCP then runs the TUI `claude` in a **credential-isolated home** (`$HOME/.ocp-tui/home`) that has **no `credentials.json`** at all, so the env token is the only credential (authoritative — nothing shadows it) and claude never runs the refresh path (so the single-use token can't be corrupted). Then restart — on systemd `daemon-reload`, on launchd `bootout`+`bootstrap`; `kickstart -k` does **not** reload env. Verify the env reached the process and the boot log shows the isolated home:
|
||||
|
||||
```bash
|
||||
# Linux (systemd): confirm the token is in the service env
|
||||
tr '\0' '\n' < /proc/$(pgrep -f server.mjs | head -1)/environ | grep CLAUDE_CODE_OAUTH_TOKEN
|
||||
# Boot log should read: TUI-mode: ON home=$HOME/.ocp-tui/home ... auth=env-token (credential-isolated home — no credentials.json)
|
||||
```
|
||||
|
||||
> If you previously set `OCP_TUI_HOME` to the real home (or any home that contains a `credentials.json`), **unset it** so the credential-isolated default takes effect — otherwise the shadowing `credentials.json` remains in play.
|
||||
|
||||
See [Subscription-pool (TUI) mode](tui-mode.md#subscription-pool-tui-mode) and ADR 0007 PR-C / PR-D amendments.
|
||||
@@ -0,0 +1,196 @@
|
||||
Part of [OCP](../README.md) — subscription-pool (TUI) mode: serve requests through interactive `claude` so they bill the Pro/Max subscription pool instead of the metered Agent SDK path.
|
||||
|
||||
# Subscription-pool (TUI) mode
|
||||
|
||||
> **SECURITY — read before enabling.**
|
||||
> TUI-mode is **single-user / single-operator only**. `claude` runs with the OCP process owner's filesystem access regardless of `HOME` setting. If OCP serves multiple users or guest API keys, a guest prompt could exfiltrate files or exhaust the subscription. **Never enable `CLAUDE_TUI_MODE=true` on a multi-user OCP.**
|
||||
|
||||
## What it is and why
|
||||
|
||||
> **⚠️ Status (as of 2026-07): the billing split below is PAUSED.** Anthropic announced it for 2026-06-15, then paused it on the effective date — *"For now, nothing has changed: Claude Agent SDK, `claude -p`, and third-party app usage still draw from your subscription's usage limits"* ([official help article](https://support.claude.com/en/articles/15036540-use-the-claude-agent-sdk-with-your-claude-plan)). While the pause holds, OCP's default `-p` path bills the subscription and **TUI-mode is a hedge, not a necessity**. The table describes the *announced* regime, kept here because Anthropic says a reworked change will return (with advance notice) — everything in this section is ready to flip on that day.
|
||||
|
||||
The announced routing keys `claude` invocations by `cc_entrypoint`:
|
||||
|
||||
| Launch method | `cc_entrypoint` | Billing pool (announced regime, currently paused) |
|
||||
|---------------|-----------------|-------------|
|
||||
| `claude -p` / `--output-format` (OCP default) | `sdk-cli` | Agent SDK credit pool (~$20/mo on Pro) |
|
||||
| Interactive `claude` (no flags) | `cli` | Pro/Max subscription pool |
|
||||
|
||||
TUI-mode lets OCP serve requests via the interactive path so they bill against the subscription pool under that regime. The response is read from claude's native JSONL session transcript once the turn is complete, then replayed to the caller as a normal OpenAI completion or chunked SSE response.
|
||||
|
||||
<a id="tui-entrypoint"></a>
|
||||
## Billing-classifier labeling (`OCP_TUI_ENTRYPOINT`)
|
||||
|
||||
`OCP_TUI_ENTRYPOINT` (default `cli`) controls how `CLAUDE_CODE_ENTRYPOINT` is set on the spawn
|
||||
environment. The default (`cli`) pins the value deterministically — immune to a stray inherited
|
||||
env var or a future stdout-redirect bug silently flipping it to `sdk-cli`. This label is honest
|
||||
**only** when the spawn is a genuine interactive PTY (tmux pane, no `-p`, stdout not redirected,
|
||||
and `tmux new-session` verified to succeed). If you need to observe the raw TTY-derived value, set
|
||||
`OCP_TUI_ENTRYPOINT=auto`. See ADR 0007 for the full rationale and governing rule.
|
||||
|
||||
## Enabling TUI-mode (opt-in)
|
||||
|
||||
```bash
|
||||
# Prerequisites
|
||||
mkdir -p ~/.ocp-tui/work # one-time scratch cwd setup
|
||||
# tmux must be installed: brew install tmux / apt install tmux
|
||||
|
||||
# Enable
|
||||
export CLAUDE_TUI_MODE=true
|
||||
# STRONGLY RECOMMENDED on a TUI host — authenticate via the long-lived OAuth token.
|
||||
# With this set (and OCP_TUI_HOME left UNSET), OCP runs the interactive claude in a
|
||||
# credential-isolated home ($HOME/.ocp-tui/home, no credentials.json), so the env token
|
||||
# is the only credential and is authoritative. This both stops a stale credentials.json
|
||||
# from shadowing the token AND ends the refresh-token corruption that caused a permanent
|
||||
# "Please run /login" 401 (no credentials file → claude never runs the refresh path).
|
||||
# See the auth note below + ADR 0007 PR-D.
|
||||
export CLAUDE_CODE_OAUTH_TOKEN=sk-ant-oat01-...
|
||||
# Optionally tune:
|
||||
export CLAUDE_TUI_WALLCLOCK_MS=180000 # 3 min cap for long Opus turns
|
||||
export OCP_TUI_CWD=$HOME/.ocp-tui/work # default; override if needed
|
||||
export OCP_TUI_ENTRYPOINT=cli # default; use 'auto' to observe TTY-derived value
|
||||
# Do NOT set OCP_TUI_HOME for the recommended setup — leaving it unset is what enables
|
||||
# the credential-isolated home. Set it only to opt into the legacy symlinked-creds mode.
|
||||
```
|
||||
|
||||
Then restart OCP. At boot you will see (with the env token set, isolated home auto-selected):
|
||||
|
||||
```
|
||||
⚠️ TUI-mode ON — single-user only; do NOT enable on a multi-user OCP ...
|
||||
TUI-mode: ON home=/home/user/.ocp-tui/home cwd=/home/user/.ocp-tui/work auth=env-token (credential-isolated home — no credentials.json) wallclock=120000ms maxConcurrent=2
|
||||
```
|
||||
|
||||
## What changes / what doesn't
|
||||
|
||||
- **Callers see no API change.** The response is a normal OpenAI completion object or chunked SSE — identical wire format.
|
||||
- **Real streaming is opt-in (`OCP_TUI_STREAM=1`), and off by default.** By default TUI-mode buffers the full response and replays it as chunked SSE — you see a delay, then the complete response. Set `OCP_TUI_STREAM=1` and `stream:true` turns emit real SSE `delta.content` chunks as `claude` renders them, sourced from `claude`'s own `MessageDisplay` hook (byte-faithful raw markdown, on the subscription pool, no `-p`). Two honest caveats: granularity is **block-level** — the hook fires once per rendered block, so a handful of chunks per answer, scaling with length, not token-by-token; and it moves the **first** byte, not the last, so a consumer that must parse a complete reply gains nothing. The transcript stays authoritative: every streamed turn is asserted against it at the end, and a turn whose stream disagrees is **failed rather than served** (watch `tui.streamDivergences` on `/health`). Evidence: [`plans/2026-07-13-tui-latency/streaming-spike.md`](plans/2026-07-13-tui-latency/streaming-spike.md).
|
||||
- **Cache and singleflight work normally.** TUI-mode writes the buffered response to the cache on success; cache-hits skip the interactive turn entirely.
|
||||
- **The host's `CLAUDE.md` / auto-memory is never injected.** OCP is a proxy — the proxied client (OpenClaw / your IDE) owns its own context and memory. TUI-mode always runs `claude` with `CLAUDE_CODE_DISABLE_CLAUDE_MDS` + `CLAUDE_CODE_DISABLE_AUTO_MEMORY`, so a `CLAUDE.md` on the OCP host can never leak into proxied turns (verified live; see #4). Built-in tool schemas + the interactive system prompt remain (the inherent ~20–35K context floor of interactive mode); MCP is hard-disabled.
|
||||
- **Authenticate via `CLAUDE_CODE_OAUTH_TOKEN` in a credential-isolated home (recommended).** tmux does not forward the parent process's env to the pane, so OCP sets the token explicitly on the spawned `claude` when `CLAUDE_CODE_OAUTH_TOKEN` is present. With the env token set and `OCP_TUI_HOME` unset, OCP runs claude in a **credential-isolated home** (`$HOME/.ocp-tui/home`) that has **no `credentials.json`** — so the env token is the only credential and is authoritative, and claude never runs the token-refresh path. This both stops a stale `credentials.json` from shadowing the token and ends the refresh-token corruption behind the permanent `Please run /login · API Error: 401` (full two-layer root cause, live proof, and fix in [Troubleshooting § the permanent TUI-mode 401](troubleshooting.md#tui-401)). Transcripts land under the same isolated home, so the answer-reader is unaffected. Without the env token, claude falls back to the real home's `credentials.json` (byte-for-byte the previous behaviour). (The token is visible in `ps` on the pane command — acceptable for the single-user A-path; the multi-user B-path is refused at boot.) See ADR 0007 PR-C / PR-D amendments.
|
||||
- **Stale tmux sessions are reaped.** The pane's `claude` is a child of the tmux server (not OCP), so OCP cannot reap it directly; `claude` zombies can otherwise accumulate as `<defunct>` over a long-running host. OCP reaps them at boot and on a 15-min idle sweep by issuing `tmux kill-server` — but **only when no foreign tmux session remains** (it never disrupts a co-hosted `olp-tui-*` instance). See ADR 0007 PR-C amendment.
|
||||
- **Default path unchanged.** Unset `CLAUDE_TUI_MODE` and restart → `callClaude` / `callClaudeStreaming` are used again, byte-for-byte identical to today.
|
||||
- **Concurrency is bounded separately.** TUI turns are heavy (per-request cold-boot + long wallclock), so the TUI path has its own limiter — `OCP_TUI_MAX_CONCURRENT` (default `2`), independent of `CLAUDE_MAX_CONCURRENT`. Excess turns queue; a full queue returns a 503. Tune it up only on a host that can run more interactive `claude` sessions at once.
|
||||
- **Optional warm pane pool (`OCP_TUI_POOL_SIZE`, default off).** Pre-boots panes so a request skips the cold boot — measured p50 `10.17s` → `6.00s` (−41%). Pooled panes are **single-use** (one turn, then killed and replaced in the background), each carrying its own fresh `--session-id`, so one session still means one exchange and no earlier-turn text can leak into a later answer. They are named `ocp-tui-<port>-p<hex>` and coexist with the reaper by design: the sweep **drains the pool first**, then reaps (so `kill-server` still flushes `<defunct>` zombies), then the pool refills in the background. Drain→reap→resume is synchronous, so no request can land mid-sweep; a request arriving while the pool is still re-booting simply misses it and cold-boots. A live pooled pane is never reaped — **including one that is still booting**, whose tmux session already exists — while an *orphaned* one (left by a previous process generation) still is.
|
||||
|
||||
## ⚠️ Latency: TUI mode has a ~6-second floor, and it is immovable
|
||||
|
||||
**TUI mode cannot serve real-time or interactive-latency consumers.** This is a hard property of the
|
||||
path, stated plainly so you can rule it out before building on it:
|
||||
|
||||
| | measured |
|
||||
|---|---|
|
||||
| **TTFT floor (first token)** | **≈ 6 s** — immovable |
|
||||
| cold boot → input bar ready | ~1 s (per request; not the bottleneck) |
|
||||
| OCP's own overhead above the CLI | ~4 s (n=1 same-turn decomposition) |
|
||||
| direct Anthropic API, same prompt (for scale) | 0.84–1.64 s |
|
||||
|
||||
The ~6 s floor is the `claude` CLI itself: it always injects the full Claude Code system prompt plus
|
||||
its tool definitions before your prompt, on every turn, no matter what you ask. No flag removes it
|
||||
(`--exclude-dynamic-system-prompt-sections` was measured: **no effect** on the floor). Extended
|
||||
thinking is *not* the cause — `OCP_TUI_EFFORT` already defaults to `low`, which is what cuts a
|
||||
formerly-inherited `xhigh` down to this floor and collapses its variance.
|
||||
|
||||
On top of the floor you pay the model's generation time (a function of output length). Progressive
|
||||
output is not wired up **yet** (see "No real token streaming" above — it is achievable and planned),
|
||||
so today a turn returns as one blob once generation completes. Note that streaming, when it lands,
|
||||
will move the *first* byte earlier — it does **not** shorten the turn, and a consumer that needs the
|
||||
complete answer gains nothing from it.
|
||||
|
||||
**Use TUI mode for**: batch, background, and latency-insensitive work where the subscription pool is
|
||||
the point. **Do not use it for**: anything a person is waiting on interactively, or any consumer with
|
||||
a sub-5-second budget. Full measurements and methodology:
|
||||
[`plans/2026-07-13-tui-latency/`](plans/2026-07-13-tui-latency/).
|
||||
|
||||
## Monitoring drift via `/health`
|
||||
|
||||
`GET /health` includes a `tui` block so you can poll for a silent billing-pool drift (the top risk under the announced split, if it re-lands — a lost TTY flipping `cc_entrypoint` from `cli` to `sdk-cli` would still return answers but land in the metered pool). The block is **always present** (with `enabled:false` when TUI-mode is off):
|
||||
|
||||
```jsonc
|
||||
"tui": {
|
||||
"enabled": true, // CLAUDE_TUI_MODE === "true"
|
||||
"entrypointMode": "cli", // OCP_TUI_ENTRYPOINT (cli | auto | off)
|
||||
"lastEntrypoint": "cli", // last cc_entrypoint observed in a transcript, or null
|
||||
"entrypointMismatches": 0, // count of cli-expected-but-got-other turns — ALERT if this climbs
|
||||
"inflight": 1, // TUI turns running right now
|
||||
"queued": 0, // TUI turns waiting for a concurrency slot
|
||||
"maxConcurrent": 2, // OCP_TUI_MAX_CONCURRENT
|
||||
"pool": { // warm pane pool — null when OCP_TUI_POOL_SIZE=0 (the default)
|
||||
"size": 2, // target warm panes (OCP_TUI_POOL_SIZE)
|
||||
"warm": 2, // panes ready right now — each is a LIVE idle claude process
|
||||
"booting": 0, // replacement panes currently pre-booting
|
||||
"model": "claude-sonnet-4-6", // the model being warmed (the most recently requested one)
|
||||
"hits": 12, // requests served by a warm pane
|
||||
"misses": 1, // requests that fell back to the cold boot (the 1st is always one)
|
||||
"boots": 14, // panes successfully pre-booted
|
||||
"bootFailures": 0, // pre-boots that genuinely never reached the input bar — WATCH this
|
||||
"cancelled": 4, // in-flight boots OCP killed on purpose (drain / model switch) — not faults
|
||||
"dropped": 8 // panes discarded unused (drain sweep / expired / unhealthy)
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Alert on `entrypointMismatches > 0` (or `lastEntrypoint !== "cli"`): it means a turn drew from the metered Agent SDK pool instead of the subscription. `inflight` / `queued` show how close the TUI path is to its concurrency cap.
|
||||
|
||||
With the pool on, `hits` / `misses` is the hit rate (a steady single-model consumer should sit near 100% after the first request), and `warm` is your standing idle-process cost. A climbing `bootFailures` means panes are not reaching their input bar — the pool then degrades safely to the cold path, but latency reverts to the un-pooled numbers. `cancelled` counts boots OCP killed *on purpose* (a drain, a model switch) and is **not** a fault signal — do not alert on it. A steadily climbing `dropped` is likewise normal: the 15-min reap sweep drains and re-boots the pool on every tick so `kill-server` can still flush `<defunct>` zombies.
|
||||
|
||||
## Kill-switch
|
||||
|
||||
```bash
|
||||
unset CLAUDE_TUI_MODE
|
||||
# restart OCP
|
||||
```
|
||||
|
||||
The stream-json path is restored immediately. No other change is needed.
|
||||
|
||||
## Operator checklist for the (paused) billing split
|
||||
|
||||
> **Status:** the 2026-06-15 split never took effect — Anthropic paused it on the effective date (see the status note at the top of this section). **Nothing needs flipping while the pause holds.** The checklist is retained verbatim as the runbook for if/when a reworked change lands (Anthropic has promised advance notice).
|
||||
|
||||
Under the announced regime, every host serving traffic must be flipped to TUI-mode **and** canary-verified before the effective date, or it will bill the metered Agent SDK credit pool instead of the subscription.
|
||||
|
||||
- **[Flip/rollback runbook](runbooks/tui-flip-rollback.md)** — how to set `CLAUDE_TUI_MODE=true` on systemd (Linux) and launchd (macOS) hosts. Covers the `daemon-reload` requirement (systemd) and the `bootout`+`bootstrap` cycle requirement (launchd — `launchctl kickstart -k` does not reload plist env).
|
||||
- **[615-canary runbook](runbooks/615-canary.md)** — after each flip, run one quiesced request and compare the Agent SDK credit balance before and after. `entrypoint:cli` in the transcript (the `cc_entrypoint` billing classifier) is necessary but not sufficient — only a stable credit balance confirms the subscription pool is being used. Balance check is a manual step (no known programmatic API for the Agent SDK credit pool balance).
|
||||
|
||||
## Architecture and design decisions
|
||||
|
||||
See [`adr/0007-tui-interactive-mode.md`](adr/0007-tui-interactive-mode.md) for the full rationale, home-strategy options, MCP-disable mechanism, coexistence rules, and the B-path (multi-tenant isolation) roadmap.
|
||||
|
||||
## TUI-mode environment variables
|
||||
|
||||
The README [Environment Variables](../README.md#environment-variables) table lists these as one-line pointers; the full behaviour of each lives here.
|
||||
|
||||
<a id="ocp-tui-stream"></a>
|
||||
### `OCP_TUI_STREAM` — real SSE streaming (opt-in)
|
||||
|
||||
`OCP_TUI_STREAM` default `0` (off). When `=1`, `stream:true` requests emit **real SSE `delta.content` chunks as `claude` generates them**, instead of buffering the turn and replaying it. Deltas come from `claude`'s own `MessageDisplay` hook (registered with `--settings` on the ordinary interactive spawn — banner-verified to stay on the subscription pool, `· Claude Max`). Granularity is **block-level**, not token-level. The transcript remains authoritative: the streamed text is asserted equal to it at end-of-turn, the auth-banner and truncation gates still run before anything is committed, and only the transcript text is cached. A turn whose stream cannot be reconciled with the transcript is **refused** (SSE error frame, not cached) and counted as `tui.streamDivergences` on `/health`. A total hook failure (e.g. `--settings` stops registering it after a `claude` version bump) is a *different, silent* failure mode — every streamed turn still succeeds, fully buffered, with no divergence and no error — so it is counted separately as `tui.streamZeroDeltaTurns` (streamed turns where the hook fired **zero** times) and logged as `tui_stream_zero_deltas`; watch it alongside `streamDivergences`. Default off — the buffered path is unchanged and remains the stable default. ⚠️ **Tool-using turns:** the transcript keeps only the model's **last** assistant message, so if the model narrates before calling a tool ("I'll check that file…") and that narration exceeds `OCP_TUI_STREAM_HOLDBACK`, it has already been streamed and cannot be retracted — the turn is then **refused** rather than served (measured live: Opus narrated 475 chars before a `Bash` call). If your deployment lets the model use tools (the TUI default, and anything with `OCP_TUI_FULL_TOOLS=1`), either raise `OCP_TUI_STREAM_HOLDBACK` above the typical narration length — the narration then stays held back and is correctly discarded, at the cost of a later first chunk — or leave streaming off. Streaming is best suited to tool-light chat proxying. See ADR 0007 (2026-07-13 amendment).
|
||||
|
||||
Two related streaming knobs:
|
||||
|
||||
- **`OCP_TUI_STREAM_DIR`** (default `$HOME/.ocp-tui/stream`) — directory holding the static `MessageDisplay` hook script + settings file, and the per-session delta sink (`<session-id>.jsonl`, removed at turn teardown). One sink **per session-id** — this is what keeps concurrent TUI turns (`OCP_TUI_MAX_CONCURRENT` ≥ 2) from interleaving one client's deltas into another's stream.
|
||||
- **`OCP_TUI_STREAM_POLL_MS`** (default `100`) — interval at which OCP drains the delta sink. The hook fires at block granularity (seconds apart), so a finer poll buys nothing.
|
||||
|
||||
<a id="ocp-tui-stream-holdback"></a>
|
||||
### `OCP_TUI_STREAM_HOLDBACK`
|
||||
|
||||
`OCP_TUI_STREAM_HOLDBACK` default `100`. (TUI-mode, streaming) Characters withheld before the first chunk reaches the client. Two jobs. (1) It keeps the **auth-banner gate** alive under streaming, via a guarantee with two required halves: (i) nothing is emitted for a message until its trimmed accumulation exceeds 100 chars — past the default banner detector's reach, since real banners are ≤100 chars — and (ii) once a message boundary follows an emit, nothing further is ever emitted for the rest of the turn, and the turn is refused outright. Half (i) alone only covers a turn's first message; half (ii) is what covers an error banner rendered as a *later* message (e.g. after tool-using prose). Raise the holdback if you replace the detector via `CLAUDE_TUI_ERROR_PATTERNS` with patterns that can match longer messages — that only affects half (i); OCP warns at boot if you do. (2) It is the knob for **tool-using turns** — see the `OCP_TUI_STREAM` caveat above. Answers shorter than the holdback are simply delivered whole at end-of-turn, exactly as the buffered path does.
|
||||
|
||||
<a id="ocp-tui-pool-size"></a>
|
||||
### `OCP_TUI_POOL_SIZE` — warm pane pool
|
||||
|
||||
`OCP_TUI_POOL_SIZE` default `0` (off). Number of **pre-booted warm `claude` panes** kept ready, so a request does not pay the cold boot. `0` disables the pool entirely — the request path is then exactly the cold-boot path. Max `4`; an unparseable value disables it rather than guessing. **Measured on a Mac mini (Sonnet 4.6, `--effort low`): end-to-end p50 `10.17s` (n=6, pool off) → `6.00s` (n=12 warm hits) — −4.2 s / −41%** — the pool recovers both the ~1.2 s boot *and* ~2.9 s of post-input-bar init that a pane which has been idle a moment has already finished. **Cost:** each warm pane is a *live idle `claude` process* held whether or not a request ever arrives (peak processes ≈ pool size + `OCP_TUI_MAX_CONCURRENT` + 1 booting replacement) — which is why it is opt-in. Panes are **single-use**: one turn, then killed and replaced in the background. The **first request after start (and after any model switch) is always a cold miss** — the pool warms the most recently requested model, since OCP cannot know which model the next caller wants. See [`plans/2026-07-13-tui-latency/`](plans/2026-07-13-tui-latency/).
|
||||
|
||||
<a id="ocp-tui-full-tools"></a>
|
||||
### `OCP_TUI_FULL_TOOLS` — full tool surface (single-user only)
|
||||
|
||||
`OCP_TUI_FULL_TOOLS` default *(unset)*. (TUI-mode, **single-user only**) When `=1`, grant the interactive session the **same tool surface as the `-p` path** — `--allowedTools` (+ optional `--mcp-config`, read from `CLAUDE_ALLOWED_TOOLS` / `CLAUDE_MCP_CONFIG`) — instead of the default MCP-walled, built-in-tools-only set. Lets a trusted single-operator TUI deployment run a **tool-using / MCP agent** (e.g. an OpenClaw assistant) on the subscription pool. Safe because TUI **refuses to boot under `AUTH_MODE=multi`** (hard exit) — no guest key can ever reach the TUI path, so this gate cannot expose tools to an untrusted caller. (Under `AUTH_MODE=shared` + `OCP_TUI_ALLOW_LAN=1`, anyone holding the single shared key reaches it — that is the existing TUI trust model, unchanged.) Note: `--dangerously-skip-permissions` / `CLAUDE_SKIP_PERMISSIONS` is **not** supported for TUI — claude v2.1.x shows an interactive bypass-acceptance screen in headless tmux that cannot be answered, bricking the pane. Use scratch-home `settings.json` `additionalDirectories` instead. See ADR 0007.
|
||||
|
||||
<a id="tui-other-vars"></a>
|
||||
### Other TUI-mode variables
|
||||
|
||||
- **`OCP_TUI_MAX_CONCURRENT`** (default `2`) — Max concurrent interactive TUI turns. **Independent** of `CLAUDE_MAX_CONCURRENT` (which bounds the `-p`/stream-json path; TUI never uses it). A TUI turn is heavy (per-request cold-boot of tmux+claude + up to `CLAUDE_TUI_WALLCLOCK_MS` wallclock), so the default is low to keep small hosts (e.g. a Pi 4) alive under a burst. Excess turns **queue** (bounded); a full queue yields a 503. See ADR 0007 PR-B amendment.
|
||||
- **`OCP_TUI_ENTRYPOINT`** (default `cli`) — Billing-classifier labeling: `cli` (default) pins `cc_entrypoint=cli` deterministically; `auto` lets claude self-classify via TTY detection; `off` leaves the inherited env untouched. Honest only when the spawn is a genuine interactive PTY — see the "Billing-classifier labeling" section above and ADR 0007.
|
||||
- **`OCP_TUI_EFFORT`** (default `low`) — Effort level passed to the interactive `claude` as an explicit `--effort` flag: `low` (default), `medium`, `high`, `xhigh`, `max`, or `inherit` to omit the flag (the pre-flag behaviour: the pane inherits a HOME-dependent effort — the operator's `~/.claude/settings.json` `effortLevel` in real-home mode, claude's built-in default in env-token scratch mode). Explicit `low` cuts measured TTFT p50 by ~40% and collapses run-to-run variance ~15× versus an inherited `xhigh` (see [`plans/2026-07-13-tui-latency/`](plans/2026-07-13-tui-latency/)); proxied requests rarely benefit from extended thinking. Banner-verified to stay on the subscription pool (`· Claude Max`). An invalid value logs a warning and falls back to `low`.
|
||||
- **`OCP_TUI_HOME`** (default *(auto)*) — `HOME` claude runs under. **When unset, OCP picks it for you:** if `CLAUDE_CODE_OAUTH_TOKEN` is set → a **credential-isolated** scratch home `$HOME/.ocp-tui/home` (no `credentials.json`, env-token auth — **recommended**); if no env token → the operator's real home (legacy shared `credentials.json`). Setting this to an **explicit** path overrides the auto-default. The credential handling at that path still follows the env token: **with** the env token it is credential-free (env-token auth, no `credentials.json` written); **without** the env token (and the path ≠ real home) it uses the legacy symlinked-credentials scratch mode, which carries the credential-fork caveat — see ADR 0007. If you previously set this to the real home (or any home containing a `credentials.json`) and hit a permanent 401, unset it — see [Troubleshooting § the permanent TUI-mode 401](troubleshooting.md#tui-401).
|
||||
- **`CLAUDE_TUI_WALLCLOCK_MS`** (default `120000`) — Maximum time in ms to wait for the native transcript to signal turn completion. Increase for long Opus thinking turns.
|
||||
- **`OCP_TUI_CWD`** (default `$HOME/.ocp-tui/work`) — Scratch working directory where interactive claude sessions run. Transcripts land under `<HOME>/.claude/projects/<encoded-cwd>/`. Created automatically.
|
||||
- **`CLAUDE_CODE_OAUTH_TOKEN`** — the recommended TUI credential; when set (and `OCP_TUI_HOME` unset) it selects the credential-isolated home. Full precedence and the 401 root cause it prevents are in [Troubleshooting § the permanent TUI-mode 401](troubleshooting.md#tui-401).
|
||||
@@ -0,0 +1,79 @@
|
||||
Part of [OCP](../README.md) — the full upgrade manual (`ocp update` paths, manual flags, rollback, and OpenClaw auto-sync). The README keeps a short stub with the one-liner.
|
||||
|
||||
# Upgrading
|
||||
|
||||
The simplest path: ask your AI.
|
||||
|
||||
Paste this prompt:
|
||||
|
||||
```
|
||||
Upgrade my OCP. Run `ocp update` and follow whatever it says.
|
||||
If it tells me to run `claude auth login`, I'll do that.
|
||||
```
|
||||
|
||||
What `ocp update` does:
|
||||
|
||||
- **Patch bump** (e.g. `v3.21.0 → v3.21.1`):
|
||||
light path (git pull + npm install + restart).
|
||||
- **Cross-minor** (e.g. `v3.18 → v3.22`):
|
||||
full path: pre-flight check, snapshot, `setup.mjs` (with plist env-merge),
|
||||
service restart, post-flight `/health` and `/v1/models` verification.
|
||||
- **Old version** (< v3.4.0):
|
||||
fresh-install. Pre-v3.4 lacked admin-key/usage-db, so there is nothing to
|
||||
migrate. Your OAuth token (managed by the Claude Code CLI, not OCP) is
|
||||
preserved; you do not need to re-OAuth unless your token expired
|
||||
separately.
|
||||
|
||||
Snapshots are saved to `~/.ocp/upgrade-snapshot-<ISO-ts>/` and never
|
||||
auto-deleted. Clean old ones with `rm -rf ~/.ocp/upgrade-snapshot-*` once
|
||||
you're confident the upgrade is stable.
|
||||
|
||||
## Manual upgrade — same command, no AI
|
||||
|
||||
```bash
|
||||
ocp update # smart-pick path
|
||||
ocp update --check # show available updates, don't apply
|
||||
ocp update --dry-run # preview plan
|
||||
ocp update --target v3.13.0 # pin a specific version
|
||||
ocp update --rollback --yes # restore most recent snapshot (--yes confirms)
|
||||
ocp update --rollback --list # list snapshots, no mutation
|
||||
ocp update --rollback --dry-run # preview rollback plan
|
||||
```
|
||||
|
||||
## When upgrade fails
|
||||
|
||||
`ocp update` prints a recovery line on failure. To restore from the snapshot:
|
||||
|
||||
```bash
|
||||
ocp update --rollback --yes # --yes confirms the destructive restore
|
||||
ocp doctor
|
||||
```
|
||||
|
||||
If `ocp doctor` still reports problems after rollback, open a GitHub issue
|
||||
with the snapshot path and the doctor JSON output (`ocp doctor --json`).
|
||||
|
||||
## OpenClaw Auto-Sync (v3.11.0+)
|
||||
|
||||
Whenever the model list in [`models.json`](../models.json) changes, `ocp update` automatically reconciles your OpenClaw config so the model dropdown stays in sync — no more "I upgraded OCP but my Telegram bot still shows the old models" surprises.
|
||||
|
||||
**What gets synced** (and only this — all other config keys are preserved):
|
||||
- `models.providers."claude-local".models` in `~/.openclaw/openclaw.json`
|
||||
- `agents.defaults.models["claude-local/*"]` aliases
|
||||
|
||||
**Safety**:
|
||||
- Timestamped backup written before every change: `~/.openclaw/openclaw.json.bak.<ms>`
|
||||
- Idempotent — already-in-sync runs are a no-op (no backup, no rewrite)
|
||||
- Non-fatal — sync failure does NOT abort `ocp update`; `/v1/models` still works
|
||||
- Skips silently if OpenClaw is not installed (`~/.openclaw/openclaw.json` missing)
|
||||
|
||||
**Manual trigger** (e.g. after fixing a hand-edited config, or for the one-time v3.10.0→v3.11.0 bootstrap quirk):
|
||||
```bash
|
||||
node ~/ocp/scripts/sync-openclaw.mjs
|
||||
node ~/ocp/scripts/sync-openclaw.mjs --quiet # silent unless changes
|
||||
```
|
||||
|
||||
**Opt-out**: `ocp update` only invokes the sync if `node` and `scripts/sync-openclaw.mjs` are both present. Removing the script disables auto-sync; the rest of `ocp update` still works.
|
||||
|
||||
**One-time bootstrap caveat (v3.10.0 → v3.11.0 only)**: the first `ocp update` to v3.11.0 runs the *old* `cmd_update` already loaded into your shell, so the new sync hook does NOT fire on this single jump. Run `node ~/ocp/scripts/sync-openclaw.mjs` once manually. Every future update from v3.11.0+ syncs automatically. (Also captured in the README Troubleshooting section as a bootstrap quirk.)
|
||||
|
||||
**Other IDEs** (Cline / Aider / Cursor / opencode) query `/v1/models` live, so they pick up new models on the next request — no sync needed. Continue.dev users edit their own `config.json` model id manually.
|
||||
@@ -0,0 +1,488 @@
|
||||
// keys.mjs — API key management and usage tracking for OCP LAN mode
|
||||
// Uses Node.js built-in SQLite (node:sqlite) — zero external dependencies.
|
||||
import { DatabaseSync } from "node:sqlite";
|
||||
import { randomBytes, createHash } from "node:crypto";
|
||||
import { join } from "node:path";
|
||||
import { mkdirSync, chmodSync } from "node:fs";
|
||||
import { homedir } from "node:os";
|
||||
|
||||
// Resolved LAZILY, on first getDb() — not at module top-level. Two reasons, and the second is
|
||||
// the bug this fixes:
|
||||
//
|
||||
// 1. Merely IMPORTING keys.mjs should not, as a side effect, create directories in the
|
||||
// operator's home.
|
||||
// 2. OCP_DIR_OVERRIDE exists so the test suite can point the key store at a scratch dir — and
|
||||
// because ESM hoists imports, a top-level `const OCP_DIR = ...` here would be evaluated
|
||||
// BEFORE an importing module's body could set the env var. Eager resolution made the
|
||||
// override unsettable in the one place that needs it. (test-features.mjs carried a comment
|
||||
// claiming it could "set env before the first getDb() call" — it could not, because nothing
|
||||
// here ever read an env var. So `npm test` wrote real, UNREVOKED api_keys rows into the
|
||||
// operator's live ~/.ocp/ocp.db: two per run, unbounded — 737 junk keys against 12 real ones
|
||||
// on the maintainer's host — and two concurrent runs raced one file, which is the ~1-in-6
|
||||
// flake in `listKeys includes quota fields`.)
|
||||
//
|
||||
// The override is gated on NODE_ENV === "test", and that gate is the ACTUAL guard. An earlier
|
||||
// cut of this fix relied on the variable merely having an awkward name — i.e. a naming convention
|
||||
// plus a comment — which is precisely the failure mode this whole change exists to indict (a
|
||||
// comment describing an intention that nothing enforces). The two-key gate means NEITHER var
|
||||
// alone does anything: a stray OCP_DIR_OVERRIDE with no NODE_ENV is inert, and NODE_ENV=test with
|
||||
// no override just resolves the default dir.
|
||||
//
|
||||
// This gate does NOT, by itself, prove a production daemon can't be redirected — an earlier
|
||||
// version of this comment overclaimed that ("a production server runs without NODE_ENV, so it
|
||||
// CANNOT honor the override no matter how the variable got in"). That is only true while the
|
||||
// daemon's env actually lacks NODE_ENV=test, which is an assumption, not something this file can
|
||||
// enforce. What makes it hold in the shipped configuration is defense-in-depth in OCP's launchers:
|
||||
// the plist/systemd units strip both vars on every (re)install (scripts/lib/plist-merge.mjs
|
||||
// NEVER_PRESERVE), and `ocp` restart's manual nohup fallback strips them (`env -u`). So a server
|
||||
// OCP itself started cannot carry the test-only redirection. The one residual path is an operator
|
||||
// who hand-launches `node server.mjs` with BOTH vars explicitly exported, bypassing every
|
||||
// launcher — a case no library-level gate can catch. The loud getDb() log below ("NOT the default
|
||||
// ~/.ocp/ocp.db") is the backstop there: a wrong key store is at least never silent (in
|
||||
// AUTH_MODE=multi that would otherwise be a total auth outage with nothing on /health to show it).
|
||||
function resolveOcpDir() {
|
||||
const override = process.env.NODE_ENV === "test" ? process.env.OCP_DIR_OVERRIDE : null;
|
||||
const dir = override || join(homedir(), ".ocp");
|
||||
mkdirSync(dir, { recursive: true, mode: 0o700 });
|
||||
// Tighten the directory mode in case it already existed with broader permissions.
|
||||
try { chmodSync(dir, 0o700); } catch { /* ignore EPERM on pre-existing dirs */ }
|
||||
return dir;
|
||||
}
|
||||
|
||||
let db;
|
||||
let dbPath; // resolved on first open, alongside the db handle
|
||||
|
||||
export function getDb() {
|
||||
if (!db) {
|
||||
dbPath = join(resolveOcpDir(), "ocp.db");
|
||||
// Say which store we opened. Silence was the other half of the bug: a server on the wrong
|
||||
// key store looks exactly like a server on the right one until every request 401s.
|
||||
if (dbPath !== join(homedir(), ".ocp", "ocp.db")) {
|
||||
console.error(`[keys] key store: ${dbPath} (NOT the default ~/.ocp/ocp.db)`);
|
||||
}
|
||||
db = new DatabaseSync(dbPath);
|
||||
db.exec("PRAGMA journal_mode = WAL");
|
||||
db.exec("PRAGMA foreign_keys = ON");
|
||||
initSchema();
|
||||
// Tighten mode on the DB file (0600) after creation / first open.
|
||||
try { chmodSync(dbPath, 0o600); } catch { /* ignore — same-user access still works */ }
|
||||
}
|
||||
return db;
|
||||
}
|
||||
|
||||
// Which file the key store actually opened. Exported so a test can ASSERT it is not the
|
||||
// operator's real db — the bug this replaced was invisible precisely because nothing checked.
|
||||
export function getDbPath() { return dbPath; }
|
||||
|
||||
function initSchema() {
|
||||
db.exec(`
|
||||
CREATE TABLE IF NOT EXISTS api_keys (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
key TEXT UNIQUE NOT NULL,
|
||||
name TEXT NOT NULL,
|
||||
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||
revoked INTEGER NOT NULL DEFAULT 0
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS usage_log (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
key_id INTEGER,
|
||||
key_name TEXT NOT NULL DEFAULT 'anonymous',
|
||||
model TEXT NOT NULL,
|
||||
prompt_chars INTEGER NOT NULL DEFAULT 0,
|
||||
response_chars INTEGER NOT NULL DEFAULT 0,
|
||||
elapsed_ms INTEGER NOT NULL DEFAULT 0,
|
||||
success INTEGER NOT NULL DEFAULT 1,
|
||||
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||
FOREIGN KEY (key_id) REFERENCES api_keys(id)
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_created ON usage_log(created_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_key ON usage_log(key_id);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS response_cache (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
hash TEXT UNIQUE NOT NULL,
|
||||
model TEXT NOT NULL,
|
||||
response TEXT NOT NULL,
|
||||
created_at TEXT NOT NULL DEFAULT (datetime('now')),
|
||||
last_hit_at TEXT,
|
||||
hits INTEGER NOT NULL DEFAULT 0
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS idx_cache_hash ON response_cache(hash);
|
||||
CREATE INDEX IF NOT EXISTS idx_cache_created ON response_cache(created_at);
|
||||
`);
|
||||
|
||||
// Idempotent migrations: add quota columns if they don't exist yet.
|
||||
for (const col of [
|
||||
"ALTER TABLE api_keys ADD COLUMN quota_daily INTEGER DEFAULT NULL",
|
||||
"ALTER TABLE api_keys ADD COLUMN quota_weekly INTEGER DEFAULT NULL",
|
||||
"ALTER TABLE api_keys ADD COLUMN quota_monthly INTEGER DEFAULT NULL",
|
||||
]) {
|
||||
try { db.exec(col); } catch (e) {
|
||||
// SQLite throws "duplicate column name" if already present — safe to ignore.
|
||||
if (!e.message?.includes("duplicate column")) throw e;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── Key CRUD ──
|
||||
|
||||
export function createKey(name) {
|
||||
const key = "ocp_" + randomBytes(24).toString("base64url");
|
||||
const d = getDb();
|
||||
const stmt = d.prepare("INSERT INTO api_keys (key, name) VALUES (?, ?)");
|
||||
const result = stmt.run(key, name);
|
||||
return { id: result.lastInsertRowid, key, name };
|
||||
}
|
||||
|
||||
export function listKeys() {
|
||||
const d = getDb();
|
||||
return d.prepare(
|
||||
"SELECT id, key, name, created_at, revoked, quota_daily, quota_weekly, quota_monthly FROM api_keys ORDER BY created_at DESC"
|
||||
).all().map(({ key, ...rest }) => ({
|
||||
...rest,
|
||||
keyPreview: key.slice(0, 8) + "..." + key.slice(-4),
|
||||
}));
|
||||
}
|
||||
|
||||
export function revokeKey(idOrName) {
|
||||
const d = getDb();
|
||||
const stmt = d.prepare(
|
||||
"UPDATE api_keys SET revoked = 1 WHERE (id = ? OR name = ?) AND revoked = 0"
|
||||
);
|
||||
return stmt.run(idOrName, idOrName).changes > 0;
|
||||
}
|
||||
|
||||
export function validateKey(key) {
|
||||
const d = getDb();
|
||||
const row = d.prepare(
|
||||
"SELECT id, name FROM api_keys WHERE key = ? AND revoked = 0"
|
||||
).get(key);
|
||||
return row || null;
|
||||
}
|
||||
|
||||
// ── Usage recording ──
|
||||
|
||||
export function recordUsage({ keyId, keyName, model, promptChars, responseChars, elapsedMs, success }) {
|
||||
const d = getDb();
|
||||
d.prepare(`
|
||||
INSERT INTO usage_log (key_id, key_name, model, prompt_chars, response_chars, elapsed_ms, success)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
`).run(keyId ?? null, keyName || "anonymous", model, promptChars, responseChars, elapsedMs, success ? 1 : 0);
|
||||
}
|
||||
|
||||
// ── Usage queries ──
|
||||
|
||||
export function getUsageByKey({ since, until } = {}) {
|
||||
const d = getDb();
|
||||
let where = "WHERE 1=1";
|
||||
const params = [];
|
||||
if (since) { where += " AND created_at >= ?"; params.push(since); }
|
||||
if (until) { where += " AND created_at <= ?"; params.push(until); }
|
||||
|
||||
return d.prepare(`
|
||||
SELECT
|
||||
key_name,
|
||||
COUNT(*) as requests,
|
||||
SUM(CASE WHEN success = 1 THEN 1 ELSE 0 END) as successes,
|
||||
SUM(CASE WHEN success = 0 THEN 1 ELSE 0 END) as errors,
|
||||
SUM(prompt_chars) as total_prompt_chars,
|
||||
SUM(response_chars) as total_response_chars,
|
||||
SUM(elapsed_ms) as total_elapsed_ms,
|
||||
AVG(elapsed_ms) as avg_elapsed_ms,
|
||||
MIN(created_at) as first_request,
|
||||
MAX(created_at) as last_request
|
||||
FROM usage_log
|
||||
${where}
|
||||
GROUP BY key_name
|
||||
ORDER BY requests DESC
|
||||
`).all(...params);
|
||||
}
|
||||
|
||||
export function getUsageTimeline({ keyName, hours = 24 } = {}) {
|
||||
const d = getDb();
|
||||
const since = new Date(Date.now() - hours * 3600000).toISOString();
|
||||
let where = "WHERE created_at >= ?";
|
||||
const params = [since];
|
||||
if (keyName) { where += " AND key_name = ?"; params.push(keyName); }
|
||||
|
||||
return d.prepare(`
|
||||
SELECT
|
||||
strftime('%Y-%m-%dT%H:00:00', created_at) as hour,
|
||||
COUNT(*) as requests,
|
||||
SUM(prompt_chars) as prompt_chars,
|
||||
SUM(response_chars) as response_chars,
|
||||
AVG(elapsed_ms) as avg_elapsed_ms
|
||||
FROM usage_log
|
||||
${where}
|
||||
GROUP BY hour
|
||||
ORDER BY hour
|
||||
`).all(...params);
|
||||
}
|
||||
|
||||
export function getRecentUsage(limit = 50) {
|
||||
const d = getDb();
|
||||
return d.prepare(`
|
||||
SELECT key_name, model, prompt_chars, response_chars, elapsed_ms, success, created_at
|
||||
FROM usage_log
|
||||
ORDER BY created_at DESC
|
||||
LIMIT ?
|
||||
`).all(limit);
|
||||
}
|
||||
|
||||
// ── SQLite datetime helper ──
|
||||
// SQLite datetime('now') stores as 'YYYY-MM-DD HH:MM:SS' (no T, no Z).
|
||||
// JavaScript .toISOString() produces 'YYYY-MM-DDTHH:MM:SS.sssZ'.
|
||||
// String comparison between the two breaks for same-day ranges (T > space).
|
||||
// This helper formats Date to match SQLite's format for correct comparisons.
|
||||
function sqliteDatetime(date) {
|
||||
return date.toISOString().replace("T", " ").replace(/\.\d{3}Z$/, "");
|
||||
}
|
||||
|
||||
// ── Quota management ──
|
||||
|
||||
// Returns { period, limit, used, resetsIn } if a quota is exceeded, null otherwise.
|
||||
// Anonymous/admin callers (keyId === null) are never subject to quotas.
|
||||
export function checkQuota(keyId, _keyName) {
|
||||
if (keyId === null || keyId === undefined) return null;
|
||||
|
||||
const d = getDb();
|
||||
const keyRow = d.prepare(
|
||||
"SELECT quota_daily, quota_weekly, quota_monthly FROM api_keys WHERE id = ? AND revoked = 0"
|
||||
).get(keyId);
|
||||
if (!keyRow) return null;
|
||||
|
||||
const now = new Date();
|
||||
|
||||
// UTC period boundaries (SQLite-compatible format)
|
||||
const startOfToday = sqliteDatetime(new Date(Date.UTC(now.getUTCFullYear(), now.getUTCMonth(), now.getUTCDate())));
|
||||
const sevenDaysAgo = sqliteDatetime(new Date(Date.now() - 7 * 86400000));
|
||||
const thirtyDaysAgo = sqliteDatetime(new Date(Date.now() - 30 * 86400000));
|
||||
|
||||
// Next reset times for human display
|
||||
const tomorrowUTC = new Date(Date.UTC(now.getUTCFullYear(), now.getUTCMonth(), now.getUTCDate() + 1));
|
||||
function msToHuman(ms) {
|
||||
if (ms <= 0) return "now";
|
||||
const h = Math.floor(ms / 3600000);
|
||||
const m = Math.floor((ms % 3600000) / 60000);
|
||||
if (h >= 24) { const d = Math.floor(h / 24); return `${d}d ${h % 24}h`; }
|
||||
return h > 0 ? `${h}h ${m}m` : `${m}m`;
|
||||
}
|
||||
|
||||
// Single query for all periods (widest window = monthly)
|
||||
const row = d.prepare(`
|
||||
SELECT
|
||||
SUM(CASE WHEN created_at >= ? THEN 1 ELSE 0 END) as daily_cnt,
|
||||
SUM(CASE WHEN created_at >= ? THEN 1 ELSE 0 END) as weekly_cnt,
|
||||
COUNT(*) as monthly_cnt
|
||||
FROM usage_log
|
||||
WHERE key_id = ? AND success = 1 AND created_at >= ?
|
||||
`).get(startOfToday, sevenDaysAgo, keyId, thirtyDaysAgo);
|
||||
|
||||
const checks = [
|
||||
{ period: "daily", limit: keyRow.quota_daily, used: row?.daily_cnt ?? 0, resetsIn: msToHuman(tomorrowUTC - now) },
|
||||
{ period: "weekly", limit: keyRow.quota_weekly, used: row?.weekly_cnt ?? 0, resetsIn: "rolling 7-day window" },
|
||||
{ period: "monthly", limit: keyRow.quota_monthly, used: row?.monthly_cnt ?? 0, resetsIn: "rolling 30-day window" },
|
||||
];
|
||||
|
||||
for (const { period, limit, used, resetsIn } of checks) {
|
||||
if (limit === null || limit === undefined) continue;
|
||||
if (used >= limit) {
|
||||
return { period, limit, used, resetsIn };
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// Set quota for a key. Only updates fields explicitly present in the input object.
|
||||
// Pass null to clear a specific limit. Omit a field to leave it unchanged.
|
||||
export function updateKeyQuota(idOrName, updates = {}) {
|
||||
const d = getDb();
|
||||
const setClauses = [];
|
||||
const params = [];
|
||||
if ("daily" in updates) { setClauses.push("quota_daily = ?"); params.push(updates.daily ?? null); }
|
||||
if ("weekly" in updates) { setClauses.push("quota_weekly = ?"); params.push(updates.weekly ?? null); }
|
||||
if ("monthly" in updates){ setClauses.push("quota_monthly = ?");params.push(updates.monthly ?? null); }
|
||||
if (setClauses.length === 0) return false;
|
||||
params.push(idOrName, idOrName);
|
||||
const result = d.prepare(
|
||||
`UPDATE api_keys SET ${setClauses.join(", ")} WHERE id = ? OR name = ?`
|
||||
).run(...params);
|
||||
return result.changes > 0;
|
||||
}
|
||||
|
||||
// Returns { daily: { limit, used }, weekly: { limit, used }, monthly: { limit, used } }
|
||||
export function getKeyQuota(keyId) {
|
||||
const d = getDb();
|
||||
const keyRow = d.prepare(
|
||||
"SELECT quota_daily, quota_weekly, quota_monthly FROM api_keys WHERE id = ?"
|
||||
).get(keyId);
|
||||
if (!keyRow) return null;
|
||||
|
||||
const now = new Date();
|
||||
const startOfToday = sqliteDatetime(new Date(Date.UTC(now.getUTCFullYear(), now.getUTCMonth(), now.getUTCDate())));
|
||||
const sevenDaysAgo = sqliteDatetime(new Date(Date.now() - 7 * 86400000));
|
||||
const thirtyDaysAgo = sqliteDatetime(new Date(Date.now() - 30 * 86400000));
|
||||
|
||||
const row = d.prepare(`
|
||||
SELECT
|
||||
SUM(CASE WHEN created_at >= ? THEN 1 ELSE 0 END) as daily_cnt,
|
||||
SUM(CASE WHEN created_at >= ? THEN 1 ELSE 0 END) as weekly_cnt,
|
||||
COUNT(*) as monthly_cnt
|
||||
FROM usage_log
|
||||
WHERE key_id = ? AND success = 1 AND created_at >= ?
|
||||
`).get(startOfToday, sevenDaysAgo, keyId, thirtyDaysAgo);
|
||||
|
||||
return {
|
||||
daily: { limit: keyRow.quota_daily ?? null, used: row?.daily_cnt ?? 0 },
|
||||
weekly: { limit: keyRow.quota_weekly ?? null, used: row?.weekly_cnt ?? 0 },
|
||||
monthly: { limit: keyRow.quota_monthly ?? null, used: row?.monthly_cnt ?? 0 },
|
||||
};
|
||||
}
|
||||
|
||||
// ── Response cache ──
|
||||
|
||||
// Generate a cache key from model + messages + request params that affect output.
|
||||
// opts.keyId isolates per-API-key cache pools (v2 hash format).
|
||||
// When keyId is absent/null/empty, falls back to "anon" (shared anonymous pool).
|
||||
export function cacheHash(model, messages, opts = {}) {
|
||||
const keyId = opts.keyId || "anon";
|
||||
const h = createHash("sha256");
|
||||
h.update(`v2|k:${keyId}|`);
|
||||
h.update(model);
|
||||
if (opts.temperature != null) h.update(`t:${opts.temperature}`);
|
||||
if (opts.max_tokens != null) h.update(`mt:${opts.max_tokens}`);
|
||||
if (opts.top_p != null) h.update(`tp:${opts.top_p}`);
|
||||
// #176: fold the server's boot-config epoch into the key, so a config change that shapes
|
||||
// answers (operator system prompt, wrapper text, allowed tools, NO_CONTEXT) invalidates
|
||||
// the persistent cache instead of serving answers composed under the old config. Callers
|
||||
// that omit it (older paths, tests) hash byte-identically to before.
|
||||
if (opts.configEpoch != null) h.update(`ce:${opts.configEpoch}|`);
|
||||
// Structured-output (OpenAI response_format / json_mode) requests must never share a cache slot
|
||||
// with the conversational answer to the same prompt, nor with a different schema — the steering
|
||||
// instruction and validated JSON payload differ. Keying on the detected descriptor isolates them.
|
||||
// Absent for normal requests → hashes are byte-identical to pre-change.
|
||||
if (opts.structured != null) h.update(`s:${JSON.stringify(opts.structured)}`);
|
||||
for (const m of messages) {
|
||||
h.update(m.role || "");
|
||||
h.update(typeof m.content === "string" ? m.content : JSON.stringify(m.content));
|
||||
}
|
||||
return h.digest("hex");
|
||||
}
|
||||
|
||||
// Check whether any message (or content part) carries an Anthropic cache_control field.
|
||||
// If true, OCP should skip its own cache to avoid interfering with prompt-caching intent.
|
||||
export function hasCacheControl(messages) {
|
||||
for (const m of messages || []) {
|
||||
if (m && typeof m === "object") {
|
||||
if (m.cache_control) return true;
|
||||
if (Array.isArray(m.content)) {
|
||||
for (const part of m.content) {
|
||||
if (part && typeof part === "object" && part.cache_control) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Look up a cached response. Returns { response, hits } or null.
|
||||
// Also updates last_hit_at and increments hits counter on hit.
|
||||
export function getCachedResponse(hash, ttlMs) {
|
||||
const d = getDb();
|
||||
const cutoff = sqliteDatetime(new Date(Date.now() - ttlMs));
|
||||
const row = d.prepare(
|
||||
"SELECT id, response, hits FROM response_cache WHERE hash = ? AND created_at >= ?"
|
||||
).get(hash, cutoff);
|
||||
if (!row) return null;
|
||||
// Update hit stats
|
||||
d.prepare("UPDATE response_cache SET hits = hits + 1, last_hit_at = datetime('now') WHERE id = ?").run(row.id);
|
||||
return { response: row.response, hits: row.hits + 1 };
|
||||
}
|
||||
|
||||
// Store a response in the cache
|
||||
export function setCachedResponse(hash, model, response) {
|
||||
const d = getDb();
|
||||
// Upsert: if hash already exists (race condition), just update
|
||||
d.prepare(`
|
||||
INSERT INTO response_cache (hash, model, response) VALUES (?, ?, ?)
|
||||
ON CONFLICT(hash) DO UPDATE SET response = excluded.response, created_at = datetime('now'), hits = 0
|
||||
`).run(hash, model, response);
|
||||
}
|
||||
|
||||
// Clear all cached responses, or expired ones only
|
||||
export function clearCache(ttlMs = null) {
|
||||
const d = getDb();
|
||||
if (ttlMs === null) {
|
||||
const result = d.prepare("DELETE FROM response_cache").run();
|
||||
return result.changes;
|
||||
}
|
||||
const cutoff = sqliteDatetime(new Date(Date.now() - ttlMs));
|
||||
const result = d.prepare("DELETE FROM response_cache WHERE created_at < ?").run(cutoff);
|
||||
return result.changes;
|
||||
}
|
||||
|
||||
// Get cache statistics
|
||||
export function getCacheStats() {
|
||||
const d = getDb();
|
||||
const total = d.prepare("SELECT COUNT(*) as cnt FROM response_cache").get()?.cnt ?? 0;
|
||||
const totalHits = d.prepare("SELECT SUM(hits) as total FROM response_cache").get()?.total ?? 0;
|
||||
const sizeBytes = d.prepare("SELECT SUM(LENGTH(response)) as size FROM response_cache").get()?.size ?? 0;
|
||||
return { entries: total, totalHits, sizeBytes };
|
||||
}
|
||||
|
||||
// ── Singleflight stampede protection ──
|
||||
|
||||
// In-memory singleflight Map: hash → { promise, requesters }
|
||||
// Deduplicates concurrent identical cache-miss flows so only one upstream call runs.
|
||||
// Per ADR 0005 / spec D4: in-process scope only (single Node process per host).
|
||||
const inflightMap = new Map();
|
||||
|
||||
// `retryIf` (optional, audit finding M1): a predicate applied on the FOLLOWER path only.
|
||||
// When a follower joins an existing flight and the shared promise rejects with an error for
|
||||
// which retryIf(err) is true (in practice: the LEADER's client disconnected while queued —
|
||||
// an error that is personal to the leader, not a verdict about the upstream), the follower
|
||||
// does NOT inherit that rejection. Instead it re-enters singleflight with its OWN fn: it
|
||||
// either becomes the new leader (the map entry is already deleted — see the finally below,
|
||||
// which runs before any follower's catch because it is attached upstream of the promise the
|
||||
// followers await) or joins a flight another retrying follower just created. The leader's
|
||||
// own rejection is never retried here — its error belongs to it (leader path returns the
|
||||
// bare promise). Callers that pass no retryIf get the exact pre-M1 share-everything behavior.
|
||||
export function singleflight(hash, fn, retryIf) {
|
||||
const existing = inflightMap.get(hash);
|
||||
if (existing) {
|
||||
existing.requesters++;
|
||||
if (!retryIf) return existing.promise;
|
||||
return existing.promise.catch((err) => {
|
||||
if (!retryIf(err)) throw err;
|
||||
return singleflight(hash, fn, retryIf);
|
||||
});
|
||||
}
|
||||
// Wrap fn() in Promise.resolve().then() so synchronous throws don't escape.
|
||||
const promise = Promise.resolve().then(fn).finally(() => {
|
||||
inflightMap.delete(hash);
|
||||
});
|
||||
inflightMap.set(hash, { promise, requesters: 1 });
|
||||
return promise;
|
||||
}
|
||||
|
||||
export function getInflightStats() {
|
||||
let totalRequesters = 0;
|
||||
for (const entry of inflightMap.values()) totalRequesters += entry.requesters;
|
||||
return {
|
||||
inflight: inflightMap.size,
|
||||
requesters: totalRequesters,
|
||||
};
|
||||
}
|
||||
|
||||
// Find a key by id or name (returns { id, name } or null)
|
||||
export function findKey(idOrName) {
|
||||
const d = getDb();
|
||||
return d.prepare("SELECT id, name FROM api_keys WHERE id = ? OR name = ?").get(idOrName, idOrName) || null;
|
||||
}
|
||||
|
||||
export function closeDb() {
|
||||
if (db) { db.close(); db = null; dbPath = undefined; } // clear both — a path to a closed db is a footgun
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
/**
|
||||
* OCP shared constants — single source of truth.
|
||||
*
|
||||
* Any literal that appears in more than one place across server.mjs, setup.mjs,
|
||||
* scripts/* belongs here so port-drift / URL-drift cascades cannot recur.
|
||||
*
|
||||
* Background: from 2026-05-08 (PR #71 dogfood accident) through 2026-05-13
|
||||
* (v3.16.3) a single hardcoded "3478" in scripts/upgrade.mjs + scripts/doctor.mjs
|
||||
* cascaded into every downstream config write, ultimately taking out the
|
||||
* OpenClaw "大内总管" Telegram agent. See CHANGELOG v3.16.2 and v3.16.3.
|
||||
*
|
||||
* Adding a new constant: prefer ALL_CAPS_SNAKE_CASE. Document the consumers.
|
||||
* If a literal is referenced from a shell script (ocp, ocp-connect, setup.sh)
|
||||
* that can't import .mjs, add a `// keep in sync with lib/constants.mjs` note
|
||||
* at the shell-script reference; CI grep prevents drift.
|
||||
*/
|
||||
|
||||
// Default TCP port the OCP HTTP proxy listens on. Set by env CLAUDE_PROXY_PORT
|
||||
// at runtime; this is the fallback when env is unset.
|
||||
// Consumers: server.mjs, setup.mjs, scripts/upgrade.mjs, scripts/doctor.mjs,
|
||||
// scripts/sync-openclaw.mjs. Shell scripts ocp / ocp-connect keep the literal
|
||||
// "3456" in sync with this value (see CI gate in .github/workflows/alignment.yml).
|
||||
export const DEFAULT_PORT = 3456;
|
||||
|
||||
// Localhost bind for client-side fetches (curl, health checks).
|
||||
export const LOCAL_HOST = "127.0.0.1";
|
||||
|
||||
// OpenAI-compatible API base path appended to the proxy URL.
|
||||
export const OPENAI_API_BASE = "/v1";
|
||||
|
||||
// Convenience: full local URL the OCP proxy listens on by default.
|
||||
// scripts that want to probe locally can use this directly.
|
||||
export const LOCAL_PROXY_URL = `http://${LOCAL_HOST}:${DEFAULT_PORT}`;
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
// OCP env-var parsing helpers.
|
||||
//
|
||||
// Fail-closed positive-integer parsing for numeric caps (body size, image
|
||||
// byte/count limits). A misconfigured cap must NEVER silently disable a guard:
|
||||
// `parseInt("unlimited", 10)` is NaN and `x > NaN` is always false, so a naive
|
||||
// parse of CLAUDE_MAX_BODY_SIZE=unlimited would remove the body-size limit
|
||||
// entirely (unbounded body → OOM). Likewise CLAUDE_MAX_BODY_SIZE=5MB naively
|
||||
// parses to 5 (bytes) and bricks the proxy. So a present-but-invalid value is
|
||||
// REJECTED (default kept, caller warns), not accepted. (PR #154 review F3.)
|
||||
//
|
||||
// Pure (no env access, no IO) so it is unit-testable without a live server.
|
||||
|
||||
// Parse `raw` as a strictly-positive base-10 integer of bytes/count (no unit
|
||||
// suffix). Returns { value, ok, reason }:
|
||||
// - missing/empty → { value: def, ok: true } (use default)
|
||||
// - valid positive int → { value: n, ok: true }
|
||||
// - anything else → { value: def, ok: false, reason } (fail closed)
|
||||
// Rejects: NaN ("unlimited"), non-positive ("0", "-1"), unit-suffixed ("5MB"),
|
||||
// and fractional/ambiguous ("20.5", "0x10") values — String(n) !== trimmed catches
|
||||
// any input parseInt only partially consumed.
|
||||
export function parsePositiveInt(raw, def) {
|
||||
if (raw === undefined || raw === null || raw === "") return { value: def, ok: true };
|
||||
const trimmed = String(raw).trim();
|
||||
const n = parseInt(trimmed, 10);
|
||||
if (!Number.isFinite(n) || n <= 0 || String(n) !== trimmed) {
|
||||
return { value: def, ok: false, reason: "not a strictly-positive integer (bytes/count, no unit suffix)" };
|
||||
}
|
||||
return { value: n, ok: true };
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
// OCP multimodal helpers — OpenAI `image_url` content parts → Anthropic image
|
||||
// blocks fed to `claude -p --input-format stream-json`. (issue #110)
|
||||
//
|
||||
// Class B.1 (OpenAI-compatibility surface). Protocol authority is OpenAI's
|
||||
// chat/completions spec — the multimodal `content` parts shape
|
||||
// (https://platform.openai.com/docs/guides/vision and
|
||||
// https://platform.openai.com/docs/api-reference/chat/create#chat-create-messages,
|
||||
// `image_url` part with `image_url.url` = data URI or http(s) URL). Authorized
|
||||
// by ADR 0006. This module introduces NO field beyond OpenAI's published shape:
|
||||
// the OpenAI-side vocabulary read here is `type:"image_url"` +
|
||||
// `image_url:{url, detail?}`; the Anthropic-side vocabulary written here
|
||||
// (`type:"image", source:{type:"base64"|"url", ...}`) is the CLI's native
|
||||
// stream-json input contract, not an OCP invention.
|
||||
//
|
||||
// Kept as a pure module (no I/O, no network, no process state) mirroring the
|
||||
// lib/*.mjs pattern so it is unit-testable without a live server. server.mjs is
|
||||
// the only consumer; it owns spawning, caps configuration, and HTTP status.
|
||||
|
||||
// Anthropic vision-supported image media types. A data URI whose media type is
|
||||
// outside this set is rejected with a clear 4xx rather than forwarded (the API
|
||||
// would reject it anyway; failing early gives a better error).
|
||||
export const SUPPORTED_IMAGE_TYPES = new Set([
|
||||
"image/jpeg",
|
||||
"image/png",
|
||||
"image/gif",
|
||||
"image/webp",
|
||||
]);
|
||||
|
||||
// Default caps. server.mjs overrides these from env; they live here so the
|
||||
// pure transform is self-contained and testable.
|
||||
export const DEFAULT_MULTIMODAL_OPTS = {
|
||||
allowRemoteUrl: false, // http(s) image URLs are OFF by default (v1: data URIs only)
|
||||
maxImageBytes: 5 * 1024 * 1024, // per-image decoded-byte cap
|
||||
maxImages: 20, // max image parts across the whole request
|
||||
maxTotalImageBytes: 20 * 1024 * 1024, // aggregate decoded-byte cap
|
||||
maxTextChars: Infinity, // text-char budget (server passes MAX_PROMPT_CHARS); Infinity = no truncation
|
||||
};
|
||||
|
||||
// Typed error so server.mjs can map to the right HTTP status + OpenAI-shaped
|
||||
// error body. `status` is the HTTP code; `type` is the OpenAI error `type`.
|
||||
export class MultimodalError extends Error {
|
||||
constructor(code, status, message) {
|
||||
super(message);
|
||||
this.name = "MultimodalError";
|
||||
this.code = code;
|
||||
this.status = status;
|
||||
this.type = "invalid_request_error"; // OpenAI error `type` for 4xx client errors
|
||||
}
|
||||
}
|
||||
|
||||
// True if any message carries an OpenAI `image_url` content part. Cheap guard so
|
||||
// the byte-for-byte text path is only left when an image is genuinely present.
|
||||
export function hasImageContent(messages) {
|
||||
if (!Array.isArray(messages)) return false;
|
||||
for (const m of messages) {
|
||||
if (m && Array.isArray(m.content)) {
|
||||
for (const part of m.content) {
|
||||
if (part && part.type === "image_url") return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Extract the URL string from an OpenAI image_url part. Spec form is
|
||||
// `{type:"image_url", image_url:{url, detail?}}`; many OpenAI-compatible clients
|
||||
// also send `image_url` as a bare string. Accept both (input leniency — no new
|
||||
// output field). `detail` (auto|low|high) is OpenAI-only and has no Anthropic
|
||||
// analogue, so it is read-and-ignored.
|
||||
function imageUrlOf(part) {
|
||||
const iu = part.image_url;
|
||||
if (typeof iu === "string") return iu;
|
||||
if (iu && typeof iu.url === "string") return iu.url;
|
||||
return null;
|
||||
}
|
||||
|
||||
// Parse a base64 data URI: `data:[<media_type>][;base64],<data>`.
|
||||
// Returns { mediaType, data (base64), bytes (decoded size) } or throws MultimodalError.
|
||||
function parseDataUri(uri) {
|
||||
const comma = uri.indexOf(",");
|
||||
if (comma === -1) {
|
||||
throw new MultimodalError("invalid_data_uri", 400, "Malformed image data URI (no comma).");
|
||||
}
|
||||
const meta = uri.slice(5, comma); // strip leading "data:"
|
||||
const segs = meta.split(";");
|
||||
const mediaType = (segs[0] || "").trim().toLowerCase();
|
||||
const isBase64 = segs.slice(1).some((s) => s.trim().toLowerCase() === "base64");
|
||||
if (!isBase64) {
|
||||
throw new MultimodalError("invalid_data_uri", 400, "Only base64-encoded image data URIs are supported.");
|
||||
}
|
||||
if (!SUPPORTED_IMAGE_TYPES.has(mediaType)) {
|
||||
throw new MultimodalError(
|
||||
"unsupported_image_type",
|
||||
400,
|
||||
`Unsupported image media type '${mediaType || "(none)"}'. Supported: ${[...SUPPORTED_IMAGE_TYPES].join(", ")}.`
|
||||
);
|
||||
}
|
||||
// Strip incidental whitespace/newlines some encoders insert into data URIs.
|
||||
const data = uri.slice(comma + 1).replace(/\s/g, "");
|
||||
if (data.length === 0 || !/^[A-Za-z0-9+/]+={0,2}$/.test(data)) {
|
||||
throw new MultimodalError("invalid_data_uri", 400, "Image data URI payload is not valid base64.");
|
||||
}
|
||||
// Decoded size from base64 length (minus padding); avoids decoding the buffer
|
||||
// just to measure it.
|
||||
const padding = data.endsWith("==") ? 2 : data.endsWith("=") ? 1 : 0;
|
||||
const bytes = Math.floor((data.length * 3) / 4) - padding;
|
||||
return { mediaType, data, bytes };
|
||||
}
|
||||
|
||||
// Convert a single OpenAI image_url part to an Anthropic image block, enforcing
|
||||
// caps via the mutable `acc` accumulator ({ images, bytes }). Throws MultimodalError.
|
||||
function imagePartToBlock(part, opts, acc) {
|
||||
const url = imageUrlOf(part);
|
||||
if (!url) {
|
||||
throw new MultimodalError("invalid_image_url", 400, "image_url part is missing a URL.");
|
||||
}
|
||||
|
||||
acc.images += 1;
|
||||
if (acc.images > opts.maxImages) {
|
||||
throw new MultimodalError("too_many_images", 413, `Too many images in request (max ${opts.maxImages}).`);
|
||||
}
|
||||
|
||||
if (url.startsWith("data:")) {
|
||||
const { mediaType, data, bytes } = parseDataUri(url);
|
||||
if (bytes > opts.maxImageBytes) {
|
||||
throw new MultimodalError("image_too_large", 413, `Image exceeds per-image size limit (${opts.maxImageBytes} bytes).`);
|
||||
}
|
||||
acc.bytes += bytes;
|
||||
if (acc.bytes > opts.maxTotalImageBytes) {
|
||||
throw new MultimodalError("images_too_large", 413, `Total image payload exceeds limit (${opts.maxTotalImageBytes} bytes).`);
|
||||
}
|
||||
return { type: "image", source: { type: "base64", media_type: mediaType, data } };
|
||||
}
|
||||
|
||||
if (/^https?:\/\//i.test(url)) {
|
||||
if (!opts.allowRemoteUrl) {
|
||||
throw new MultimodalError(
|
||||
"remote_url_disabled",
|
||||
400,
|
||||
"Remote image URLs are disabled. Enable CLAUDE_IMAGE_ALLOW_URL=1 to allow http(s) image URLs, or pass the image as a base64 data URI."
|
||||
);
|
||||
}
|
||||
// Passthrough as an Anthropic url-source block. OCP does NOT fetch the URL
|
||||
// itself (no OCP-side SSRF surface); the fetch is performed upstream by the
|
||||
// Anthropic API. Best-effort: unreachable/blocked URLs surface as an API error.
|
||||
return { type: "image", source: { type: "url", url } };
|
||||
}
|
||||
|
||||
throw new MultimodalError("unsupported_url_scheme", 400, "image_url must be a base64 data URI or an http(s) URL.");
|
||||
}
|
||||
|
||||
// Role prefix mirrors messagesToPrompt()'s text-path labeling so a multi-turn
|
||||
// conversation reads the same whether or not it carries images. System messages
|
||||
// are handled by the caller via --system-prompt and never reach here.
|
||||
function rolePrefix(role) {
|
||||
if (role === "assistant") return "[Assistant] ";
|
||||
return ""; // user / tool / anything else: verbatim, as in the text path
|
||||
}
|
||||
|
||||
// Build the Anthropic content-block array for a single stream-json user
|
||||
// envelope. Mirrors the text path's "collapse the whole conversation into one
|
||||
// turn passed via stdin" model (OCP runs stateless, full context per spawn), but
|
||||
// preserves image position relative to text and keeps images out of the text
|
||||
// char budget entirely. Returns { blocks, stats } or throws MultimodalError.
|
||||
export function buildImageBlocks(messages, opts = {}) {
|
||||
const o = { ...DEFAULT_MULTIMODAL_OPTS, ...opts };
|
||||
const blocks = [];
|
||||
const acc = { images: 0, bytes: 0 };
|
||||
let textChars = 0;
|
||||
let firstMessage = true;
|
||||
|
||||
const pushText = (text) => {
|
||||
if (!text) return;
|
||||
blocks.push({ type: "text", text });
|
||||
textChars += text.length;
|
||||
};
|
||||
|
||||
for (const m of messages) {
|
||||
const prefix = rolePrefix(m.role);
|
||||
// Separate messages with a blank line, matching messagesToPrompt's "\n\n" join.
|
||||
const sep = firstMessage ? "" : "\n\n";
|
||||
firstMessage = false;
|
||||
let prefixEmitted = false;
|
||||
const emitPrefixWith = (t) => {
|
||||
if (prefixEmitted) return t;
|
||||
prefixEmitted = true;
|
||||
return sep + prefix + t;
|
||||
};
|
||||
|
||||
if (typeof m.content === "string") {
|
||||
pushText(emitPrefixWith(m.content));
|
||||
continue;
|
||||
}
|
||||
if (!Array.isArray(m.content)) {
|
||||
// null / object content: mirror contentToText's fallback.
|
||||
const t = m.content == null ? "" : JSON.stringify(m.content);
|
||||
pushText(emitPrefixWith(t));
|
||||
continue;
|
||||
}
|
||||
|
||||
for (const part of m.content) {
|
||||
if (part && part.type === "text" && typeof part.text === "string") {
|
||||
pushText(emitPrefixWith(part.text));
|
||||
} else if (part && part.type === "image_url") {
|
||||
// Ensure the role prefix isn't lost when a message leads with an image.
|
||||
if (!prefixEmitted && prefix) pushText(emitPrefixWith(""));
|
||||
blocks.push(imagePartToBlock(part, o, acc));
|
||||
} else {
|
||||
// audio / file / unknown parts: preserve the existing placeholder
|
||||
// behavior (issue #110) — deferred to a future version.
|
||||
pushText(emitPrefixWith("[non-text content omitted]"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Defensive: a stream-json user turn must have at least one content block.
|
||||
if (blocks.length === 0) blocks.push({ type: "text", text: "" });
|
||||
|
||||
// Enforce the text-char budget (PR #154 review F2). Without this, attaching a
|
||||
// single tiny image would let unbounded text bypass the gateway's runaway-context
|
||||
// guard entirely — messagesToPrompt truncates the text path, so the multimodal
|
||||
// path must too. Image blocks are preserved and are NOT counted (they are bounded
|
||||
// by the byte/count caps above).
|
||||
const budgeted = enforceTextBudget(blocks, o.maxTextChars);
|
||||
|
||||
return {
|
||||
blocks: budgeted.blocks,
|
||||
stats: {
|
||||
imageCount: acc.images,
|
||||
totalImageBytes: acc.bytes,
|
||||
textChars: budgeted.textChars,
|
||||
originalTextChars: budgeted.originalTextChars,
|
||||
truncated: budgeted.truncated,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// Enforce a text-char budget over already-built content blocks, mirroring
|
||||
// messagesToPrompt's "keep the tail, drop the oldest" truncation. Image blocks are
|
||||
// preserved in place; only text blocks count against the budget. A truncation note
|
||||
// is prepended when anything is dropped. Returns
|
||||
// { blocks, truncated, originalTextChars, textChars }.
|
||||
function enforceTextBudget(blocks, maxTextChars) {
|
||||
let originalTextChars = 0;
|
||||
for (const b of blocks) if (b.type === "text") originalTextChars += b.text.length;
|
||||
if (!(maxTextChars > 0) || originalTextChars <= maxTextChars) {
|
||||
return { blocks, truncated: false, originalTextChars, textChars: originalTextChars };
|
||||
}
|
||||
// Keep the most recent text up to the budget; trim within the boundary block and
|
||||
// drop older text blocks. Non-text (image) blocks always survive, in order.
|
||||
let budget = maxTextChars;
|
||||
const out = [];
|
||||
for (let i = blocks.length - 1; i >= 0; i--) {
|
||||
const b = blocks[i];
|
||||
if (b.type !== "text") { out.unshift(b); continue; }
|
||||
if (budget <= 0) continue; // older text fully dropped
|
||||
if (b.text.length <= budget) {
|
||||
out.unshift(b);
|
||||
budget -= b.text.length;
|
||||
} else {
|
||||
out.unshift({ type: "text", text: b.text.slice(b.text.length - budget) });
|
||||
budget = 0;
|
||||
}
|
||||
}
|
||||
out.unshift({ type: "text", text: "[System] Note: older text content was truncated to fit the context limit." });
|
||||
let textChars = 0;
|
||||
for (const b of out) if (b.type === "text") textChars += b.text.length;
|
||||
return { blocks: out, truncated: true, originalTextChars, textChars };
|
||||
}
|
||||
|
||||
// Serialize the non-system conversation to a single newline-terminated
|
||||
// stream-json user message for `claude -p --input-format stream-json` stdin.
|
||||
// Returns { payload, stats } or throws MultimodalError.
|
||||
export function buildStreamJsonInput(messages, opts = {}) {
|
||||
const { blocks, stats } = buildImageBlocks(messages, opts);
|
||||
const envelope = { type: "user", message: { role: "user", content: blocks } };
|
||||
return { payload: JSON.stringify(envelope) + "\n", stats };
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
// OCP network helpers — shared so server.mjs and tests use one definition. (issue #125)
|
||||
|
||||
// A bind address is "loopback" only if it cannot be reached from another host.
|
||||
// Any other address (0.0.0.0, ::, a concrete LAN/Tailscale IP, etc.) is
|
||||
// network-exposed and must trigger the TUI LAN gate.
|
||||
export function isLoopbackBind(addr) {
|
||||
return addr === "127.0.0.1" || addr === "::1" || addr === "localhost" ||
|
||||
addr === "::ffff:127.0.0.1" || /^127\./.test(addr);
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
// lib/prompt.mjs — pure operator-append step for the system prompt.
|
||||
//
|
||||
// Extracted so the rule is unit-testable (the suite never imports server.mjs — it
|
||||
// boots a listener). server.mjs composes wrapper + client system messages exactly as
|
||||
// before, then passes the result through this. With CLAUDE_SYSTEM_PROMPT unset the
|
||||
// return is the INPUT STRING UNCHANGED — the default path stays byte-for-byte
|
||||
// identical, which is the repo's bar for touching a request-shaping function.
|
||||
//
|
||||
// The operator prompt goes LAST deliberately: a server-wide directive ("answer in
|
||||
// Chinese") should read as the final instruction, not something a client system
|
||||
// message overrides by coming later. Whitespace-only values are treated as unset —
|
||||
// a stray space in a service unit's Environment= line must not inject "\n\n " into
|
||||
// every request.
|
||||
export function appendOperatorPrompt(base, operatorAppend) {
|
||||
const op = typeof operatorAppend === "string" ? operatorAppend.trim() : "";
|
||||
return op ? `${base}\n\n${op}` : base;
|
||||
}
|
||||
|
||||
// Derive the default prompt-char budget from the models.json SPOT (ADR 0009).
|
||||
//
|
||||
// The old default was a hand-set constant (150000 chars ≈ 37.5k English tokens) from the
|
||||
// 200k-window era — silently far below what the advertised contextWindow promises. Instead
|
||||
// of picking a new constant that will also rot, the default now FOLLOWS the SPOT:
|
||||
//
|
||||
// budget = max(models[].contextWindow) × charsPerToken
|
||||
//
|
||||
// charsPerToken = 3 is deliberately conservative: English runs ~4 chars/token, CJK ~1–1.5.
|
||||
// At ×3, a 200k-token window yields 600,000 chars — full window for English, and CJK text
|
||||
// hits the model's real window at roughly the same point the cap fires, so we truncate
|
||||
// (graceful, tail-first) rather than let the upstream reject the request outright.
|
||||
//
|
||||
// The floor guards the degenerate cases (empty/missing models[], absent contextWindow):
|
||||
// fall back to the historical constant rather than 0 — a zero budget would truncate every
|
||||
// request to nothing, which is fail-OPEN in the "serve garbage" sense. CLAUDE_MAX_PROMPT_CHARS
|
||||
// remains an absolute operator override at the call site (server.mjs); this function is only
|
||||
// the unset-env default.
|
||||
export function derivePromptCharBudget(models, { charsPerToken = 3, floor = 150000 } = {}) {
|
||||
const windows = (Array.isArray(models) ? models : [])
|
||||
.map(m => m?.contextWindow)
|
||||
.filter(w => Number.isFinite(w) && w > 0);
|
||||
if (windows.length === 0) return floor;
|
||||
return Math.max(floor, Math.max(...windows) * charsPerToken);
|
||||
}
|
||||
|
||||
// Resolve the effective budget from the env var + SPOT. TRUTHINESS (not != null) on the env
|
||||
// value deliberately: an EMPTY value ("CLAUDE_MAX_PROMPT_CHARS=" in a systemd EnvironmentFile
|
||||
// or .env) must mean "use the default" — exactly the old `parseInt(env || "150000")` contract.
|
||||
// Treating "" as explicit gives parseInt("") = NaN, and a NaN cap silently DISABLES the
|
||||
// runaway-context guard while injecting a false "[System] Note: 0 older messages were
|
||||
// truncated" line into every prompt (caught in PR #179 review). Non-empty garbage still
|
||||
// parses to NaN — the pre-existing class, slated for parseIntEnv routing in PR #154.
|
||||
export function resolvePromptCharBudget(rawEnv, models, opts) {
|
||||
return rawEnv ? parseInt(rawEnv, 10) : derivePromptCharBudget(models, opts);
|
||||
}
|
||||
|
||||
// OCP_LOCAL_TOOLS system-prompt wrapper selection (pure).
|
||||
//
|
||||
// OCP's `-p` path prepends a fixed wrapper to every request's system prompt. The DEFAULT wrapper
|
||||
// tells the model it has NO local filesystem/shell/env access — the right posture for a shared or
|
||||
// multi-tenant gateway. But a single-user, loopback-bound instance (e.g. an OpenClaw agent talking
|
||||
// to its own local OCP) DOES legitimately have tools — the `-p` path already passes `--allowedTools`
|
||||
// and the CLI's built-in tools are available — so the default wrapper actively gaslights the model
|
||||
// into refusing to use tools it holds. `OCP_LOCAL_TOOLS=1` swaps in a positive wrapper for that case.
|
||||
//
|
||||
// This does NOT expand the tool surface: tools are governed solely by `--allowedTools` /
|
||||
// `--disallowedTools` (multi-tenant mode `--disallowedTools` the whole FS surface regardless of the
|
||||
// wrapper). It only changes the PROMPT the operator's own model reads. Pure so it is unit-testable.
|
||||
export function selectPromptWrapper(localToolsEnabled, negativeWrapper, positiveWrapper) {
|
||||
return localToolsEnabled ? positiveWrapper : negativeWrapper;
|
||||
}
|
||||
|
||||
// Boot-time safety gate for OCP_LOCAL_TOOLS, mirroring the OCP_TUI_FULL_TOOLS model (ADR 0007): a
|
||||
// positive "you may use local tools" wrapper must never reach an untrusted caller. Returns a fatal
|
||||
// message string when the flag is enabled in an unsafe deployment, or null when it is safe/disabled.
|
||||
// Fail-closed: any of multi-tenant auth, a non-loopback bind, or an anonymous key is refused. Pure —
|
||||
// the caller does the process.exit so this stays testable.
|
||||
export function localToolsSafetyError({ enabled, authMode, loopbackBind, anonymousKey }) {
|
||||
if (!enabled) return null;
|
||||
if (authMode === "multi") {
|
||||
return "OCP_LOCAL_TOOLS=1 is incompatible with CLAUDE_AUTH_MODE=multi — a guest/anonymous prompt would be told it may drive the operator's filesystem/shell. Single-user only.";
|
||||
}
|
||||
if (!loopbackBind) {
|
||||
return "OCP_LOCAL_TOOLS=1 requires a loopback bind (127.0.0.1/::1) — a network-exposed positive-tools wrapper could reach an untrusted peer. Bind to loopback, or leave OCP_LOCAL_TOOLS off.";
|
||||
}
|
||||
if (anonymousKey) {
|
||||
return "OCP_LOCAL_TOOLS=1 is unsafe with PROXY_ANONYMOUS_KEY set — anonymous callers could reach the local-tools-enabled model without a named key. Remove PROXY_ANONYMOUS_KEY, or leave OCP_LOCAL_TOOLS off.";
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
// Pure, dependency-injected primitives for the `-p` spawn-token resolution + HOME-isolation
|
||||
// layer. Extracted from server.mjs (findings F3 / F5 / F6, 2026-07-07) so the concurrency,
|
||||
// caching and expiry logic is unit-testable WITHOUT booting the server or mocking execFileSync /
|
||||
// child_process.spawn / fs. server.mjs owns all I/O (macOS keychain exec, process spawn, fs);
|
||||
// this module owns only pure decision logic.
|
||||
//
|
||||
// ALIGNMENT NOTE: none of this touches the OAuth wire machinery (no endpoint / header / body).
|
||||
// OCP still NEVER performs a refresh_token grant itself — these helpers only READ + GATE a token
|
||||
// that some other process (the operator's real claude, or a spawned claude under the real HOME)
|
||||
// refreshes. That property is load-bearing (issue #112) and preserved.
|
||||
|
||||
// Promise-chain mutex. `acquire()` resolves to a `release()` fn; the NEXT `acquire()` does not
|
||||
// resolve until the current holder calls its `release()`. Serializes async critical sections
|
||||
// without busy-waiting. release() is idempotent.
|
||||
export function createSerialMutex() {
|
||||
let tail = Promise.resolve();
|
||||
return {
|
||||
acquire() {
|
||||
let release;
|
||||
const gate = new Promise((r) => { release = r; });
|
||||
const prev = tail;
|
||||
tail = tail.then(() => gate);
|
||||
// Hand the caller its release fn only after the previous holder has released.
|
||||
return prev.then(() => {
|
||||
let released = false;
|
||||
return function releaseMutex() { if (!released) { released = true; release(); } };
|
||||
});
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// Short-TTL memo. `get(produce, now)` returns the cached value while `now - storedAt < ttlMs`,
|
||||
// otherwise calls `produce()` and re-stores. A miss that produces null/undefined is STILL stored
|
||||
// (so a genuinely-absent source is not re-probed on every call within the TTL window). `now` is
|
||||
// injectable for testing.
|
||||
export function createTtlCache({ ttlMs }) {
|
||||
let value;
|
||||
let at = -Infinity;
|
||||
let has = false;
|
||||
return {
|
||||
get(produce, now = Date.now()) {
|
||||
if (has && now - at < ttlMs) return value;
|
||||
value = produce();
|
||||
at = now;
|
||||
has = true;
|
||||
return value;
|
||||
},
|
||||
clear() { has = false; value = undefined; at = -Infinity; },
|
||||
};
|
||||
}
|
||||
|
||||
// Pure expiry gate. Returns true when `creds` carries a known expiry that is at/within `bufferMs`
|
||||
// of `now`. Creds WITHOUT `expiresAt` (e.g. long-lived env tokens) are never treated as expiring.
|
||||
// This gate is applied to the CACHED creds on EVERY use — which is precisely why a short-TTL
|
||||
// keychain cache (createTtlCache) cannot reintroduce the #146 forever-stale-token regression: the
|
||||
// cache bounds how often we re-READ the keychain, but the expiry decision is recomputed per use.
|
||||
export function isTokenExpiring(creds, now = Date.now(), bufferMs = 300000) {
|
||||
return !!(creds && creds.expiresAt && now + bufferMs >= creds.expiresAt);
|
||||
}
|
||||
|
||||
// Order candidate keychain labels so the last-known-good label is tried first (avoids the
|
||||
// wrong-label miss that doubles the `security` exec count on the hot path). Pure: performs no
|
||||
// read. Returns a fresh array; input is not mutated.
|
||||
export function orderLabelsLastGoodFirst(labels, lastGood) {
|
||||
if (!lastGood || !labels.includes(lastGood)) return labels.slice();
|
||||
return [lastGood, ...labels.filter((l) => l !== lastGood)];
|
||||
}
|
||||
@@ -0,0 +1,318 @@
|
||||
// ── OpenAI Structured Outputs (response_format) — pure helpers ───────────────
|
||||
//
|
||||
// OCP's `/v1/chat/completions` (Class B.1, ADR 0006) advertises OpenAI compatibility but forwards
|
||||
// to `claude -p`, which has no native `response_format`. Asked for JSON, the coding-assistant CLI
|
||||
// typically replies with prose, a Markdown table, or a ```json fenced block — none of which is
|
||||
// `JSON.parse`-able. These helpers implement the OpenAI `response_format` contract on top of that:
|
||||
// detect the request, build a strict JSON-only steering instruction, then extract and validate the
|
||||
// JSON the model returns. All functions here are pure (no I/O) so they are unit-tested directly;
|
||||
// the retry loop that calls the model lives in server.mjs (runStructuredCompletion).
|
||||
//
|
||||
// Spec authority (B.1, ADR 0006): OpenAI chat/completions `response_format`
|
||||
// https://platform.openai.com/docs/api-reference/chat/create#chat-create-response_format
|
||||
// No field or behaviour beyond that published shape is introduced.
|
||||
|
||||
export class StructuredOutputError extends Error {
|
||||
constructor(reason, raw) {
|
||||
super(`structured output could not be produced: ${reason}`);
|
||||
this.name = "StructuredOutputError";
|
||||
this.reason = reason;
|
||||
this.raw = raw;
|
||||
}
|
||||
}
|
||||
|
||||
// Fail-closed parse of the OCP_STRUCTURED_MAX_ATTEMPTS retry cap. `Math.max(1, parseInt("abc",10))`
|
||||
// === `Math.max(1, NaN)` === NaN, and a retry loop bounded by `attempt < NaN` never runs → 0 spawns,
|
||||
// every structured request silently refuses. So any non-integer / non-finite / <1 value keeps the
|
||||
// documented default instead (and warns), rather than bricking the feature. (PR #153 review round 2.)
|
||||
export function resolveMaxAttempts(raw, { fallback = 3, warn } = {}) {
|
||||
if (raw === undefined || raw === null || raw === "") return fallback;
|
||||
const n = parseInt(raw, 10);
|
||||
if (!Number.isFinite(n) || n < 1) {
|
||||
if (typeof warn === "function") {
|
||||
warn(`Ignoring invalid OCP_STRUCTURED_MAX_ATTEMPTS="${raw}" (want integer >= 1); using default ${fallback}.`);
|
||||
}
|
||||
return fallback;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// Returns { mode: "schema", schema, name?, strict } | { mode: "json_object" } | null.
|
||||
// Supports the OpenAI shapes: response_format:{type:"json_schema",json_schema:{schema,strict,name}}
|
||||
// and response_format:{type:"json_object"}, a lenient response_format:{schema} fallback, and the
|
||||
// widely-used (non-standard) top-level `json_mode: true` flag as a json_object alias.
|
||||
export function detectStructuredOutput(parsed) {
|
||||
const rf = parsed?.response_format;
|
||||
if (rf && typeof rf === "object") {
|
||||
if (rf.type === "json_schema") {
|
||||
const js = (rf.json_schema && typeof rf.json_schema === "object") ? rf.json_schema : {};
|
||||
const schema = js.schema || rf.schema || null;
|
||||
return { mode: "schema", schema, name: js.name, strict: js.strict === true };
|
||||
}
|
||||
if (rf.type === "json_object") return { mode: "json_object" };
|
||||
if (rf.schema && typeof rf.schema === "object") {
|
||||
return { mode: "schema", schema: rf.schema, strict: rf.strict === true };
|
||||
}
|
||||
}
|
||||
// Non-standard convenience alias honored by several OpenAI-compatible clients.
|
||||
if (parsed?.json_mode === true) return { mode: "json_object" };
|
||||
return null;
|
||||
}
|
||||
|
||||
export function jsonTypeOf(v) {
|
||||
if (v === null) return "null";
|
||||
if (Array.isArray(v)) return "array";
|
||||
return typeof v; // object | string | number | boolean
|
||||
}
|
||||
|
||||
export function jsonTypeMatches(t, value) {
|
||||
switch (t) {
|
||||
case "string": return typeof value === "string";
|
||||
case "number": return typeof value === "number" && Number.isFinite(value);
|
||||
case "integer": return typeof value === "number" && Number.isInteger(value);
|
||||
case "boolean": return typeof value === "boolean";
|
||||
case "object": return value !== null && typeof value === "object" && !Array.isArray(value);
|
||||
case "array": return Array.isArray(value);
|
||||
case "null": return value === null;
|
||||
default: return true; // unknown type keyword → do not fail on it
|
||||
}
|
||||
}
|
||||
|
||||
const jsonDeepEqual = (a, b) => JSON.stringify(a) === JSON.stringify(b);
|
||||
|
||||
// Resolve a local JSON-Pointer `$ref` (e.g. "#/$defs/Step" or "#/definitions/Step") against the
|
||||
// document root. Only same-document refs are supported (that is all the OpenAI SDK emits); a remote
|
||||
// or unresolvable ref returns null and the caller skips validation for it rather than failing.
|
||||
function resolveRef(ref, root) {
|
||||
if (typeof ref !== "string" || !ref.startsWith("#/") || !root) return null;
|
||||
const parts = ref.slice(2).split("/").map(p => p.replace(/~1/g, "/").replace(/~0/g, "~"));
|
||||
let cur = root;
|
||||
for (const p of parts) {
|
||||
if (cur && typeof cur === "object" && Object.prototype.hasOwnProperty.call(cur, p)) cur = cur[p];
|
||||
else return null;
|
||||
}
|
||||
return (cur && typeof cur === "object") ? cur : null;
|
||||
}
|
||||
|
||||
// Minimal JSON-Schema validator: covers the subset OpenAI structured outputs use — type (incl.
|
||||
// arrays of types / integer), required, properties, additionalProperties (no invented keys), items
|
||||
// (list + tuple), enum, const, nullability (type:["x","null"] or nullable:true), min/maxItems, and
|
||||
// $ref/$defs + allOf/anyOf/oneOf composition (which the official OpenAI SDK emits heavily via
|
||||
// zodResponseFormat / client.beta.chat.completions.parse). `root` carries the top-level schema so
|
||||
// same-document $refs resolve. Returns error strings ([] = valid).
|
||||
// `refChain` tracks the $ref pointers resolved on the CURRENT path WITHOUT consuming data (a $ref
|
||||
// hop, or an allOf/anyOf/oneOf branch, all re-validate the same value). A pointer reappearing on
|
||||
// that chain is a pure ref→ref (or ref→composition→ref) cycle that recurses forever independent of
|
||||
// the data — we fail closed on it. Data-consuming recursion (properties/items/additionalProperties)
|
||||
// deliberately resets the chain (default []): it always terminates because a JSON value is a finite
|
||||
// tree, so a legitimately recursive schema (Node→child:Node) must NOT be flagged as a cycle. A depth
|
||||
// cap backstops any threading mistake. (PR #153 review round 2, cyclic-$ref blocker.)
|
||||
const REF_DEPTH_CAP = 512;
|
||||
export function validateJsonSchema(value, schema, path = "$", strict = false, root = schema, refChain = []) {
|
||||
const errors = [];
|
||||
if (!schema || typeof schema !== "object") return errors;
|
||||
if (refChain.length > REF_DEPTH_CAP) { // defensive backstop; refChain cycle-check below is primary
|
||||
errors.push(`${path}: $ref resolution too deep (possible cycle)`);
|
||||
return errors;
|
||||
}
|
||||
|
||||
// $ref: resolve against the document root ($defs / definitions) and validate the target. Without
|
||||
// this a nested {$ref:"#/$defs/Step"} presents as {no type, no properties} — and under strict that
|
||||
// used to wrongly reject every real key as "additional property not allowed" (the flagship OpenAI
|
||||
// SDK shape). Sibling keywords alongside $ref (rare) are merged over the resolved target.
|
||||
if (typeof schema.$ref === "string") {
|
||||
if (refChain.includes(schema.$ref)) { // cyclic $ref (a→b→a, or self a→a) — fail closed.
|
||||
errors.push(`${path}: cyclic $ref detected (${schema.$ref})`);
|
||||
return errors;
|
||||
}
|
||||
const resolved = resolveRef(schema.$ref, root);
|
||||
if (!resolved) return errors; // unresolvable ref → cannot validate; do not fail
|
||||
const { $ref, ...siblings } = schema;
|
||||
return validateJsonSchema(value, { ...resolved, ...siblings }, path, strict, root, [...refChain, schema.$ref]);
|
||||
}
|
||||
|
||||
// Composition. allOf: every branch must pass. anyOf: at least one. oneOf: exactly one.
|
||||
// These re-validate the SAME value → thread refChain so a ref cycle routed through a branch is caught.
|
||||
if (Array.isArray(schema.allOf)) {
|
||||
for (const sub of schema.allOf) errors.push(...validateJsonSchema(value, sub, path, strict, root, refChain));
|
||||
}
|
||||
if (Array.isArray(schema.anyOf)) {
|
||||
if (!schema.anyOf.some(sub => validateJsonSchema(value, sub, path, strict, root, refChain).length === 0)) {
|
||||
errors.push(`${path}: does not match any of the allowed schemas (anyOf)`);
|
||||
}
|
||||
}
|
||||
if (Array.isArray(schema.oneOf)) {
|
||||
const matches = schema.oneOf.filter(sub => validateJsonSchema(value, sub, path, strict, root, refChain).length === 0).length;
|
||||
if (matches !== 1) errors.push(`${path}: must match exactly one allowed schema (oneOf), matched ${matches}`);
|
||||
}
|
||||
|
||||
// Nullability takes precedence: a null value is valid whenever the schema permits null (its type
|
||||
// union includes "null", or nullable:true), regardless of enum/const. This mirrors OpenAI
|
||||
// structured-output behaviour — nullable fields accept null even when a bare enum (as generated by
|
||||
// Home Assistant's extended_openai_conversation) omits null from its value list.
|
||||
const allowsNull = schema.nullable === true
|
||||
|| (Array.isArray(schema.type) ? schema.type.includes("null") : schema.type === "null");
|
||||
if (value === null && allowsNull) return errors;
|
||||
|
||||
if (Array.isArray(schema.enum) && !schema.enum.some(e => jsonDeepEqual(e, value))) {
|
||||
errors.push(`${path}: not one of the allowed enum values`);
|
||||
}
|
||||
if ("const" in schema && !jsonDeepEqual(schema.const, value)) {
|
||||
errors.push(`${path}: does not equal the required const value`);
|
||||
}
|
||||
|
||||
if (schema.type !== undefined) {
|
||||
const types = Array.isArray(schema.type) ? schema.type : [schema.type];
|
||||
const nullable = schema.nullable === true || types.includes("null");
|
||||
const ok = types.some(t => jsonTypeMatches(t, value)) || (value === null && nullable);
|
||||
if (!ok) {
|
||||
errors.push(`${path}: expected ${types.join("|")}${schema.nullable ? "|null" : ""}, got ${jsonTypeOf(value)}`);
|
||||
return errors; // type mismatch — deeper checks are meaningless
|
||||
}
|
||||
}
|
||||
if (value === null) return errors;
|
||||
|
||||
const vt = jsonTypeOf(value);
|
||||
if (vt === "object") {
|
||||
const props = schema.properties || {};
|
||||
for (const r of (schema.required || [])) {
|
||||
if (!Object.prototype.hasOwnProperty.call(value, r)) errors.push(`${path}.${r}: required property missing`);
|
||||
}
|
||||
const addl = schema.additionalProperties;
|
||||
// Only treat strict as implying "no additional properties" when this object actually declares
|
||||
// its own `properties` and is NOT a composite (allOf/anyOf/oneOf put the real keys in sub-schemas,
|
||||
// which are validated separately above). Inferring closure from an EMPTY properties map — the
|
||||
// shape an unresolved $ref or a pure-composition node presents — would reject every real key.
|
||||
// An explicit additionalProperties:false is always honoured. (PR #153 review, finding 1.)
|
||||
const isComposite = Array.isArray(schema.allOf) || Array.isArray(schema.anyOf) || Array.isArray(schema.oneOf);
|
||||
const noExtra = addl === false || (strict && addl === undefined && Object.keys(props).length > 0 && !isComposite);
|
||||
for (const k of Object.keys(value)) {
|
||||
if (props[k]) {
|
||||
errors.push(...validateJsonSchema(value[k], props[k], `${path}.${k}`, strict, root));
|
||||
} else if (isComposite) {
|
||||
// key may be defined in an allOf/anyOf/oneOf branch — already validated there; don't reject.
|
||||
} else if (noExtra) {
|
||||
errors.push(`${path}.${k}: additional property not allowed`);
|
||||
} else if (addl && typeof addl === "object") {
|
||||
errors.push(...validateJsonSchema(value[k], addl, `${path}.${k}`, strict, root));
|
||||
}
|
||||
}
|
||||
} else if (vt === "array" && schema.items) {
|
||||
if (Array.isArray(schema.items)) {
|
||||
schema.items.forEach((s, i) => { if (i < value.length) errors.push(...validateJsonSchema(value[i], s, `${path}[${i}]`, strict, root)); });
|
||||
} else {
|
||||
value.forEach((item, i) => errors.push(...validateJsonSchema(item, schema.items, `${path}[${i}]`, strict, root)));
|
||||
}
|
||||
if (typeof schema.minItems === "number" && value.length < schema.minItems) errors.push(`${path}: fewer items than minItems ${schema.minItems}`);
|
||||
if (typeof schema.maxItems === "number" && value.length > schema.maxItems) errors.push(`${path}: more items than maxItems ${schema.maxItems}`);
|
||||
}
|
||||
return errors;
|
||||
}
|
||||
|
||||
// Crash-safe façade over validateJsonSchema (issue #181). The validator recurses on the DATA's
|
||||
// nesting depth (properties/items/additionalProperties), which the REF_DEPTH_CAP does NOT bound —
|
||||
// only the ref-chain is. A model reply nested ~2000 levels deep therefore overflowed the stack with
|
||||
// a RangeError, which the request handler caught as a generic HTTP 500 instead of the spec-correct
|
||||
// `refusal`. This wrapper converts ANY throw (the deep-data RangeError, or any future recursion
|
||||
// hazard) into a single validation error, so the structured-output retry loop treats a pathological
|
||||
// reply as "did not validate" → refusal — never a 500, never a crash. A well-formed reply is
|
||||
// unaffected: the inner validator returns and this just passes its errors through.
|
||||
export function validateJsonSchemaSafe(value, schema, path = "$", strict = false, root = schema) {
|
||||
try {
|
||||
return validateJsonSchema(value, schema, path, strict, root);
|
||||
} catch (e) {
|
||||
// Catch ONLY the deep-nesting stack overflow (the #181 vector) and turn it into a validation
|
||||
// miss → retry → refusal, never a 500. Any OTHER throw is a genuine bug: re-throw it so it
|
||||
// surfaces at error level instead of being silently masked as "did not validate" (reviewer
|
||||
// finding — a catch-all would hide a future TypeError behind a warn-level structured_retry).
|
||||
if (e instanceof RangeError) return [`${path}: schema validation aborted (value nesting too deep)`];
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
function tryJsonParse(s) {
|
||||
try { return { ok: true, value: JSON.parse(s) }; } catch { return { ok: false }; }
|
||||
}
|
||||
|
||||
// Find the next brace-balanced JSON span at or after `from`. String-aware: brackets inside quoted
|
||||
// strings are ignored. Returns { text, start, end } for the first complete top-level span, or null.
|
||||
function balancedSlice(s, from) {
|
||||
let start = -1;
|
||||
for (let i = from; i < s.length; i++) { if (s[i] === "{" || s[i] === "[") { start = i; break; } }
|
||||
if (start === -1) return null;
|
||||
let depth = 0, inStr = false, esc = false;
|
||||
for (let i = start; i < s.length; i++) {
|
||||
const c = s[i];
|
||||
if (inStr) {
|
||||
if (esc) esc = false;
|
||||
else if (c === "\\") esc = true;
|
||||
else if (c === '"') inStr = false;
|
||||
continue;
|
||||
}
|
||||
if (c === '"') { inStr = true; continue; }
|
||||
if (c === "{" || c === "[") depth++;
|
||||
else if (c === "}" || c === "]") { depth--; if (depth === 0) return { text: s.slice(start, i + 1), start, end: i }; }
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Extract a single JSON value from model text. Two modes (PR #153 review, finding 2):
|
||||
//
|
||||
// opts.whole === true (json_object mode): the ENTIRE reply, after trimming and stripping one code
|
||||
// fence, must parse as a single JSON value. json_object has no schema to validate against, so
|
||||
// this whole-reply parse is its ONLY guard — a reply like `I can't. The schema is {"type":"object"}`
|
||||
// must NOT parse-and-serve the embedded object as if it were the answer.
|
||||
//
|
||||
// default (schema mode): prose-wrapped JSON is tolerated (models often add a sentence), BUT a reply
|
||||
// containing MORE THAN ONE top-level JSON value is rejected rather than silently serving the first
|
||||
// — "Schema: {...}\n\nAnswer: {...}" or "Option A: {...} Option B: {...}" is ambiguous, not an
|
||||
// answer. The extracted value is still schema-validated by the caller.
|
||||
//
|
||||
// Returns { ok:true, value } | { ok:false, reason? }.
|
||||
export function extractJsonPayload(text, opts = {}) {
|
||||
if (typeof text !== "string") return { ok: false };
|
||||
let s = text.trim();
|
||||
const fence = s.match(/^```(?:json)?\s*([\s\S]*?)\s*```$/i);
|
||||
if (fence) s = fence[1].trim();
|
||||
|
||||
if (opts.whole) {
|
||||
const whole = tryJsonParse(s);
|
||||
return whole.ok ? whole : { ok: false, reason: "reply was not a single JSON value" };
|
||||
}
|
||||
|
||||
const direct = tryJsonParse(s);
|
||||
if (direct.ok) return direct;
|
||||
|
||||
const first = balancedSlice(s, 0);
|
||||
if (!first) return { ok: false };
|
||||
const parsedFirst = tryJsonParse(first.text);
|
||||
if (!parsedFirst.ok) return { ok: false };
|
||||
|
||||
// Reject ambiguity: a second parseable top-level JSON value means we cannot know which is the answer.
|
||||
const second = balancedSlice(s, first.end + 1);
|
||||
if (second && tryJsonParse(second.text).ok) {
|
||||
return { ok: false, reason: "reply contained more than one JSON value" };
|
||||
}
|
||||
return parsedFirst;
|
||||
}
|
||||
|
||||
// The strict JSON-only system instruction appended to the request (attempt 0), escalated with the
|
||||
// prior failure reason on retries.
|
||||
export function structuredSystemInstruction(structured, attempt, lastErr) {
|
||||
const schemaBlock = (structured.mode === "schema" && structured.schema)
|
||||
? `Your JSON MUST validate EXACTLY against this JSON Schema:\n${JSON.stringify(structured.schema)}\n`
|
||||
+ `- Include every required property.\n`
|
||||
+ `- Do NOT add any property that is not defined in the schema.\n`
|
||||
+ `- Respect all declared types, enums, and nullability.`
|
||||
: `Respond with a single valid JSON value.`;
|
||||
let text =
|
||||
`You are a strict JSON generator. Output a SINGLE JSON value and NOTHING else.
|
||||
- The response MUST begin with { or [ and end with the matching } or ].
|
||||
- Do NOT wrap the JSON in Markdown or code fences (no \`\`\`).
|
||||
- Do NOT include any prose, explanation, heading, comment, reasoning, or XML — only the raw JSON.
|
||||
${schemaBlock}`;
|
||||
if (attempt > 0) {
|
||||
text = `YOUR PREVIOUS RESPONSE WAS REJECTED (${lastErr}). Output ONLY the corrected raw JSON now, with no other text.\n\n` + text;
|
||||
}
|
||||
return text;
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,2 @@
|
||||
{"type":"user","entrypoint":"cli","cwd":"/tmp/tui-test","sessionId":"bbbb2222-3333-4444-5555-666677778888","version":"2.1.104","message":{"role":"user","content":[{"type":"text","text":"What is 2 + 2?"}]}}
|
||||
{"type":"assistant","entrypoint":"cli","cwd":"/tmp/tui-test","sessionId":"bbbb2222-3333-4444-5555-666677778888","version":"2.1.104","message":{"role":"assistant","model":"claude-haiku-4-5-20251001","stop_reason":"end_turn","content":[{"type":"text","text":"Failed to authenticate. API Error: 401 Invalid authentication credentials"}]}}
|
||||
@@ -0,0 +1,2 @@
|
||||
{"type":"user","entrypoint":"cli","cwd":"/tmp/tui-test","sessionId":"aaaa1111-2222-3333-4444-555566667777","version":"2.1.104","message":{"role":"user","content":[{"type":"text","text":"What is 2 + 2?"}]}}
|
||||
{"type":"assistant","entrypoint":"cli","cwd":"/tmp/tui-test","sessionId":"aaaa1111-2222-3333-4444-555566667777","version":"2.1.104","message":{"role":"assistant","model":"claude-haiku-4-5-20251001","stop_reason":"end_turn","content":[{"type":"text","text":"Please run /login · API Error: 401 Invalid authentication credentials"}]}}
|
||||
@@ -0,0 +1,2 @@
|
||||
{"type":"user","entrypoint":"cli","cwd":"/tmp/tui-test","sessionId":"bbbb1111-2222-3333-4444-555566667777","version":"2.1.104","message":{"role":"user","content":[{"type":"text","text":"Say PONG and nothing else."}]}}
|
||||
{"type":"assistant","entrypoint":"cli","cwd":"/tmp/tui-test","sessionId":"bbbb1111-2222-3333-4444-555566667777","version":"2.1.104","message":{"role":"assistant","model":"claude-haiku-4-5-20251001","stop_reason":"end_turn","content":[{"type":"text","text":"PONG"}]}}
|
||||
@@ -0,0 +1,321 @@
|
||||
import { rmSync } from "node:fs";
|
||||
|
||||
// TUI warm pane pool (docs/plans/2026-07-13-tui-latency backlog #3).
|
||||
//
|
||||
// WHAT IT IS: a small set of PRE-BOOTED `claude` panes, each already sitting at its
|
||||
// input bar, so a request does not pay the cold boot. Opt-in: OCP_TUI_POOL_SIZE=0
|
||||
// (default) disables it entirely and the request path is byte-for-byte today's.
|
||||
//
|
||||
// ── SINGLE-USE IS THE LOAD-BEARING RULE ─────────────────────────────────────
|
||||
// A pooled pane serves EXACTLY ONE turn and is then killed and replaced in the
|
||||
// background. Each pane carries its OWN fresh `--session-id`, fixed at boot, and the
|
||||
// turn locates its transcript by that id. So OCP's one-session-per-request model is
|
||||
// preserved: a session's transcript still holds exactly one logical exchange.
|
||||
// That is what keeps lib/tui/transcript.mjs's extractLatestAssistantText (which returns
|
||||
// the LAST text-bearing assistant entry in the whole file, not "text since the matching
|
||||
// user line") correct — see the scoping note there. A pane MUST NEVER serve a second
|
||||
// turn, and a session MUST NEVER be reset with /clear and reused: either would put two
|
||||
// exchanges in one transcript and leak the earlier turn's text into the later turn's
|
||||
// answer. Nothing here reuses a pane; keep it that way.
|
||||
//
|
||||
// ── WHY IT'S WORTH MORE THAN THE BOOT TIME ──────────────────────────────────
|
||||
// Measured on this host (n=6 through OCP, Sonnet 4.6, --effort low): the cold path
|
||||
// spends ~1.23 s reaching the input bar, but ALSO ~2.9 s inside the first turn beyond
|
||||
// what claude itself reports as the turn duration — post-input-bar init that a pane
|
||||
// which has been idle for a few seconds has already finished. A warm pane recovers both.
|
||||
//
|
||||
// ── COST (bounded, and paid whether or not a request arrives) ───────────────
|
||||
// Each warm pane is a LIVE `claude` process (plus its tmux pane) sitting idle. Peak
|
||||
// process count is (pool size) + (OCP_TUI_MAX_CONCURRENT in-flight turns) + (panes
|
||||
// currently booting as replacements). Pool size is clamped to POOL_MAX_SIZE.
|
||||
//
|
||||
// Pure + injectable (bootPane / killPane / paneHealthy / now) so test-features.mjs can
|
||||
// assert acquire / miss / refill / TTL / reaper-exemption with no tmux and no claude.
|
||||
|
||||
// Hard cap on OCP_TUI_POOL_SIZE. Each pane is an idle claude process; 4 is already a
|
||||
// lot of resident memory on a small host (a Pi serving a family) for zero in-flight work.
|
||||
export const POOL_MAX_SIZE = 4;
|
||||
|
||||
// A warm pane older than this is dropped on acquire rather than handed out. The periodic
|
||||
// reap tick (server.mjs) drains the pool every 15 min anyway, so this only bites when
|
||||
// that tick kept getting skipped because the TUI path was never idle. Guards against
|
||||
// handing out a pane whose `claude` has been sitting so long it may have drifted
|
||||
// (auto-compaction prompts, an idle-disconnect banner, an expired in-pane token).
|
||||
export const POOL_MAX_AGE_MS = 10 * 60 * 1000;
|
||||
|
||||
// Clamp the operator-supplied size into [0, POOL_MAX_SIZE]. A garbage value disables the
|
||||
// pool rather than guessing — an unparseable size must never silently boot 4 processes.
|
||||
export function resolvePoolSize(raw) {
|
||||
const n = parseInt(raw, 10);
|
||||
if (!Number.isFinite(n) || n <= 0) return 0;
|
||||
return Math.min(n, POOL_MAX_SIZE);
|
||||
}
|
||||
|
||||
export class TuiPanePool {
|
||||
// size: target number of warm panes (0 = disabled).
|
||||
// maxAgeMs: per-pane TTL (see POOL_MAX_AGE_MS).
|
||||
// mintPane: () => ({ sessionId, name }) — mints the identity of the NEXT pane. The POOL,
|
||||
// not the boot function, owns this: the tmux session springs into existence the
|
||||
// instant bootPane starts, so the pool must already know its NAME (see
|
||||
// _bootingPane below). Deriving the name from the sessionId also makes `tmux ls`
|
||||
// correlate to the transcript file.
|
||||
// bootPane: async (model, {sessionId, name}) => { name, sessionId, model, bootedAt } —
|
||||
// boots ONE pane under exactly that identity and resolves only once it is
|
||||
// input-ready; throws if it never becomes ready.
|
||||
// killPane: (name) => void — tmux kill-session. MUST be synchronous (see drain).
|
||||
// paneHealthy:(name) => bool — pane still exists AND is still at its input bar.
|
||||
constructor({ size, maxAgeMs = POOL_MAX_AGE_MS, mintPane, bootPane, killPane, paneHealthy, now = Date.now, log = () => {} }) {
|
||||
this.size = Math.max(0, Math.min(parseInt(size, 10) || 0, POOL_MAX_SIZE));
|
||||
// Fail fast at CONSTRUCTION, not at request time. refill() is called synchronously from
|
||||
// the request path (runTuiTurn), so a missing collaborator would otherwise surface as a
|
||||
// 500 on a live request instead of a loud error at boot.
|
||||
if (this.size > 0) {
|
||||
for (const [k, fn] of [["mintPane", mintPane], ["bootPane", bootPane], ["killPane", killPane], ["paneHealthy", paneHealthy]]) {
|
||||
if (typeof fn !== "function") throw new TypeError(`TuiPanePool: ${k} must be a function`);
|
||||
}
|
||||
}
|
||||
this.maxAgeMs = maxAgeMs;
|
||||
this._mintPane = mintPane;
|
||||
this._bootPane = bootPane;
|
||||
this._killPane = killPane;
|
||||
this._paneHealthy = paneHealthy;
|
||||
this._now = now;
|
||||
this._log = log;
|
||||
|
||||
this._panes = []; // warm, available panes: { name, sessionId, model, bootedAt }
|
||||
// The pane currently BOOTING, BY NAME ({sessionId, name, model}) — or null.
|
||||
//
|
||||
// WHY A NAME AND NOT A COUNT (this is a fixed bug, don't regress it): bootTuiPane creates
|
||||
// the tmux session SYNCHRONOUSLY and only THEN waits up to POOL_BOOT_MS (20 s) for the
|
||||
// input bar. So for up to 20 s there is a LIVE pooled tmux session. When the pool tracked
|
||||
// only a count, it could not NAME that session, so:
|
||||
// - liveNames() could not spare it and the periodic reap sweep KILLED it (and
|
||||
// kill-server'd on top), leaving the pool empty with nothing scheduled and firing the
|
||||
// very tui_pool_boot_failed WARN operators are told to alert on; and
|
||||
// - drain() could not kill it, so on shutdown it ORPHANED a live authenticated `claude`
|
||||
// (the boot's .then that was supposed to clean up never runs — gracefulShutdown calls
|
||||
// process.exit in the same tick).
|
||||
// Both are fixed by holding the identity here, before the session exists.
|
||||
this._bootingPane = null;
|
||||
// Generation counter. Bumped whenever an in-flight boot is CANCELLED (drain / model
|
||||
// switch). A boot compares the generation it started under against the current one:
|
||||
// if they differ, its pane was already killed by us and its settle is inert — in
|
||||
// particular a rejection is a CANCELLATION, not an operator-visible boot failure.
|
||||
this._gen = 0;
|
||||
this._paused = false; // true while drained; refill() is a no-op until resume()
|
||||
this.warmModel = null; // the model the pool currently warms — learned from traffic (see acquire)
|
||||
|
||||
this.hits = 0; // requests served by a warm pane
|
||||
this.misses = 0; // requests that fell back to the cold path
|
||||
this.boots = 0; // panes successfully pre-booted
|
||||
this.bootFailures = 0; // pre-boots that genuinely never reached the input bar
|
||||
this.cancelled = 0; // in-flight boots WE killed (drain / model switch) — not failures
|
||||
this.dropped = 0; // panes discarded unused (unhealthy / expired / wrong model / drained /
|
||||
// cancelled — a cancelled in-flight boot also lands here via _drop)
|
||||
}
|
||||
|
||||
get enabled() { return this.size > 0; }
|
||||
get warm() { return this._panes.length; }
|
||||
get booting() { return this._bootingPane ? 1 : 0; }
|
||||
|
||||
// The reaper's spare set: the EXACT names of every pane the pool currently owns and has NOT
|
||||
// handed out — the warm ones AND the one currently booting (whose tmux session is already
|
||||
// live; see _bootingPane). See the POOL/REAPER INVARIANT in lib/tui/session.mjs.
|
||||
// Fail-safe by construction: a pane leaves this set the instant it is acquired, dropped, or
|
||||
// cancelled, and if the pool is empty (or the process restarted) the set is empty — so an
|
||||
// orphaned pooled pane looks exactly like any other stale session and IS reaped.
|
||||
liveNames() {
|
||||
const names = new Set(this._panes.map((p) => p.name));
|
||||
if (this._bootingPane) names.add(this._bootingPane.name);
|
||||
return names;
|
||||
}
|
||||
|
||||
// Take a warm pane for `model`, or null (caller must fall back to the cold path — a MISS
|
||||
// is always safe, never an error). Synchronous: paneHealthy is a cheap tmux capture.
|
||||
//
|
||||
// The pool warms the MOST RECENTLY REQUESTED model (`warmModel`). There is no boot-time
|
||||
// pre-warm and no configured model: OCP cannot know which model the next caller wants, and
|
||||
// pre-booting a process for a model nobody asks for is pure waste. Consequence, stated
|
||||
// plainly: the FIRST request after start (and the first after a model switch) is always a
|
||||
// MISS. The pool pays off for the steady repeat traffic it exists to serve.
|
||||
acquire(model) {
|
||||
if (!this.enabled) return null;
|
||||
|
||||
// Retarget on a model switch: --model is fixed at spawn, so panes for another model are
|
||||
// useless. Drop them now (they are replaced by the next refill) rather than holding
|
||||
// processes for a model that is no longer being asked for. This includes any pane
|
||||
// currently BOOTING for the old model — its tmux session already exists, so leaving it to
|
||||
// die on resolve would both hold a useless process and block the next refill (one boot at
|
||||
// a time) for up to POOL_BOOT_MS.
|
||||
if (model !== this.warmModel) {
|
||||
for (const p of this._panes) { this._drop(p, "model_switch"); }
|
||||
this._panes = [];
|
||||
this._cancelBooting("model_switch");
|
||||
this.warmModel = model;
|
||||
}
|
||||
|
||||
while (this._panes.length) {
|
||||
const p = this._panes.shift();
|
||||
if (this._now() - p.bootedAt > this.maxAgeMs) { this._drop(p, "expired"); continue; }
|
||||
if (!this._paneHealthy(p.name)) { this._drop(p, "unhealthy"); continue; }
|
||||
this.hits++;
|
||||
return p; // caller OWNS it now: it is out of the registry (so out of the spare set),
|
||||
// and the caller's finally MUST kill it. Single-use — never returned here.
|
||||
}
|
||||
this.misses++;
|
||||
return null;
|
||||
}
|
||||
|
||||
// Bring the pool back up to `size` warm panes for `warmModel`. Fire-and-forget: never
|
||||
// awaited on the request path and never throws into it.
|
||||
//
|
||||
// SLOT ACCOUNTING: a refill boot deliberately does NOT take a TuiSemaphore slot. Those
|
||||
// slots bound concurrent *turns* (each up to the 120 s wallclock) and belong to real
|
||||
// requests; charging a background pre-boot against them would let the pool starve the
|
||||
// traffic it exists to speed up. It cannot leak a slot either, because it never holds one.
|
||||
//
|
||||
// SERIALIZED, ONE BOOT AT A TIME (and re-kicked on success until the pool is at target).
|
||||
// An earlier version launched all `want` boots at once; live at size=2 that put two cold
|
||||
// `claude` boots plus an in-flight turn on the CPU together, and a refill overran even the
|
||||
// generous pool readiness cap (tui_pool_boot_failed). Booting sequentially keeps each boot
|
||||
// near its uncontended ~1.2 s, bounds the CPU burst the pool can cause, and still has the
|
||||
// replacement pane warm long before the next request arrives.
|
||||
//
|
||||
// A genuinely FAILED boot deliberately does NOT re-kick the chain — that is the backoff. A
|
||||
// persistently failing boot (bad claude binary, no auth) would otherwise spin, respawning
|
||||
// forever. The next natural trigger (the following request's refill, or the reap tick's
|
||||
// resume) retries it. A CANCELLED boot is different: we killed it on purpose, nothing is
|
||||
// wrong, and resume() is expected to start a fresh one immediately.
|
||||
refill() {
|
||||
if (!this.enabled || this._paused || !this.warmModel) return;
|
||||
if (this._bootingPane) return; // one boot in flight at a time
|
||||
if (this._panes.length >= this.size) return; // already at target
|
||||
|
||||
const model = this.warmModel;
|
||||
const gen = this._gen;
|
||||
// Mint the identity BEFORE booting: bootPane creates the tmux session synchronously, so
|
||||
// the pool must be able to name (and therefore spare, and kill) it from this moment on.
|
||||
const ident = this._mintPane();
|
||||
this._bootingPane = { ...ident, model };
|
||||
let enlisted = false;
|
||||
Promise.resolve()
|
||||
.then(() => this._bootPane(model, ident))
|
||||
.then((pane) => {
|
||||
// The world may have moved while we booted. If our generation was cancelled, kill the
|
||||
// pane here rather than ASSUMING _cancelBooting already did.
|
||||
//
|
||||
// Why not just `return`: _cancelBooting kills by name, but the tmux session only EXISTS
|
||||
// once _bootPane has actually run — and _bootPane is queued on a microtask (above). A
|
||||
// caller that does refill() and then drain() in the SAME synchronous block would have
|
||||
// _cancelBooting find nothing to kill (a no-op), bump the generation, and then this
|
||||
// microtask would create the session, boot it fine, and — under a bare `return` — walk
|
||||
// away from a LIVE authenticated `claude` that nothing owns. That is M1b in a new costume.
|
||||
// No current call site does that, so this is defense-in-depth, not a live bug — but ADR
|
||||
// 0008 and the reap-tick comment in server.mjs both explicitly contemplate a boot-time
|
||||
// pre-warm, which is exactly the shape that would reach it.
|
||||
//
|
||||
// Killing an already-dead session is a harmless no-op (_drop swallows it), so this is
|
||||
// idempotent whether or not _cancelBooting got there first.
|
||||
if (gen !== this._gen) { this._drop(pane, "cancelled_late"); return; }
|
||||
// Otherwise: still possible the pool filled or retargeted without a cancellation.
|
||||
if (this._paused || model !== this.warmModel || this._panes.length >= this.size) {
|
||||
this._drop(pane, "stale_boot");
|
||||
return;
|
||||
}
|
||||
this._panes.push(pane);
|
||||
this.boots++;
|
||||
enlisted = true;
|
||||
})
|
||||
.catch((e) => {
|
||||
// A rejection from a CANCELLED generation is not a fault: it is almost always
|
||||
// "tui_pane_not_ready", thrown because WE killed the pane out from under the boot.
|
||||
// Counting it as a bootFailure would fire the exact WARN operators are told to alert
|
||||
// on, for a completely healthy drain. Stay silent — _cancelBooting already counted
|
||||
// this as a cancellation, so do NOT count it again here.
|
||||
if (gen !== this._gen) return;
|
||||
this.bootFailures++;
|
||||
this._log("warn", "tui_pool_boot_failed", { model, error: e && e.message });
|
||||
})
|
||||
.finally(() => {
|
||||
// ONLY the current generation's boot owns the booting slot. A stale settle must not
|
||||
// clear a slot that a newer boot (started by resume()) already holds.
|
||||
if (gen === this._gen) this._bootingPane = null;
|
||||
if (enlisted) this.refill(); // continue toward target, still one at a time
|
||||
});
|
||||
}
|
||||
|
||||
// Kill the in-flight boot's pane, SYNCHRONOUSLY, and invalidate its generation. Returns 1
|
||||
// if there was one, else 0. The tmux session already exists (bootPane created it before it
|
||||
// started waiting for readiness), so this is a real kill, not a cancellation flag.
|
||||
_cancelBooting(reason) {
|
||||
if (!this._bootingPane) return 0;
|
||||
this._gen++; // the in-flight boot's settle is now inert
|
||||
this._drop(this._bootingPane, reason); // synchronous kill-session
|
||||
this._bootingPane = null;
|
||||
this.cancelled++;
|
||||
return 1;
|
||||
}
|
||||
|
||||
// Kill every pane the pool owns — warm AND currently booting — and stop refilling. Returns
|
||||
// how many were killed.
|
||||
//
|
||||
// Called (a) before the periodic reap sweep — reapStaleTuiSessions can only reap defunct
|
||||
// `claude` zombies via kill-server, and kill-server is suppressed while any live pooled pane
|
||||
// exists (including a booting one), so without this drain the pool would permanently disable
|
||||
// zombie reaping; and (b) on graceful shutdown, so no pane outlives the process as an orphan.
|
||||
//
|
||||
// EVERY KILL HERE IS SYNCHRONOUS, and that is load-bearing. It is NOT safe to leave the
|
||||
// booting pane to clean itself up on resolve: gracefulShutdown calls process.exit() in the
|
||||
// same tick as this drain (TUI panes are children of the tmux SERVER, not of node, so
|
||||
// node's activeProcesses set is empty on a TUI host and the "wait for children" path exits
|
||||
// immediately). A .then()/.catch() scheduled here would never run, and the pane would
|
||||
// survive as an orphaned, authenticated, idle `claude`.
|
||||
drain() {
|
||||
this._paused = true;
|
||||
let n = this._panes.length;
|
||||
for (const p of this._panes) this._drop(p, "drain");
|
||||
this._panes = [];
|
||||
n += this._cancelBooting("drain_booting");
|
||||
return n;
|
||||
}
|
||||
|
||||
// Undo drain() and start refilling again. Because drain() CANCELLED the in-flight boot
|
||||
// (rather than leaving it pending), the booting slot is free and this really does start a
|
||||
// fresh boot — the pool is never left empty with nothing scheduled.
|
||||
resume() {
|
||||
this._paused = false;
|
||||
this.refill();
|
||||
}
|
||||
|
||||
// /health surface (additive).
|
||||
stats() {
|
||||
return {
|
||||
size: this.size,
|
||||
warm: this._panes.length,
|
||||
booting: this.booting,
|
||||
model: this.warmModel,
|
||||
hits: this.hits,
|
||||
misses: this.misses,
|
||||
boots: this.boots,
|
||||
bootFailures: this.bootFailures,
|
||||
cancelled: this.cancelled,
|
||||
dropped: this.dropped,
|
||||
};
|
||||
}
|
||||
|
||||
_drop(pane, reason) {
|
||||
this.dropped++;
|
||||
try { this._killPane(pane.name); } catch { /* already gone */ }
|
||||
// F5: every drop path (expired / unhealthy / model_switch / drain / cancelled_late /
|
||||
// stale_boot) ends up here, and the reap tick drains the WHOLE pool on every tick — so
|
||||
// without this, every warm pane's sink orphans in streamDir with no GC path (killPane only
|
||||
// reaches the tmux session, never the pane's OWN files). Best-effort: pane.streamFile is
|
||||
// undefined for a still-booting identity (the sink path is only known once bootPane
|
||||
// resolves) and rmSync(force:true) is already a no-op on a missing file, so this never
|
||||
// throws into the reaper regardless of which drop path got here.
|
||||
if (pane.streamFile) {
|
||||
try { rmSync(pane.streamFile, { force: true }); } catch { /* best-effort GC */ }
|
||||
}
|
||||
this._log("info", "tui_pool_pane_dropped", { name: pane.name, reason });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,192 @@
|
||||
// TUI-path concurrency limiter (audit finding C-4).
|
||||
//
|
||||
// WHY THIS EXISTS, SEPARATE FROM server.mjs's MAX_CONCURRENT:
|
||||
// The global MAX_CONCURRENT gate lives in spawnClaudeProcess() (the -p / stream-json
|
||||
// path). callClaudeTui() NEVER calls spawnClaudeProcess — it calls runTuiTurn(), which
|
||||
// boots a full interactive `claude` inside a fresh tmux session. So nothing bounded the
|
||||
// TUI path: N concurrent TUI requests spawned N simultaneous cold-boot tmux+claude
|
||||
// processes. On a small host (a Pi 4 serving a family) a burst of ~5 is an OOM risk, and
|
||||
// it also multiplies subscription rate-limit pressure. This is an INDEPENDENT limiter for
|
||||
// the TUI path that mirrors MAX_CONCURRENT's intent without coupling to it (the two pools
|
||||
// are different shapes: a stream-json spawn is cheap and fast; a TUI turn is a heavy
|
||||
// cold-boot + up to 120s wallclock).
|
||||
//
|
||||
// QUEUE vs REJECT: we QUEUE (await a slot), mirroring the spirit of MAX_CONCURRENT's
|
||||
// intent not to drop requests, rather than rejecting immediately. To avoid unbounded
|
||||
// memory growth from a runaway client, the wait queue itself is bounded by maxQueue
|
||||
// (default: a generous multiple of the concurrency limit). When the queue is full, run()
|
||||
// rejects with a tui_queue_full error (the caller surfaces it as a 503) — a deterministic
|
||||
// backpressure signal rather than silent OOM.
|
||||
//
|
||||
// Pure + importable so test-features.mjs can assert the bound directly (no server boot).
|
||||
|
||||
// Thrown by acquire() when the caller-supplied AbortSignal fires before a slot was granted
|
||||
// (audit finding F2 — a client that disconnects while queued must never receive a slot; the
|
||||
// queue entry is spliced out, not just flagged, so `queued` accounting stays exact). Distinct
|
||||
// `name` lets callers (server.mjs acquireClaudeSlot) tell "client went away" apart from
|
||||
// "queue is full" without string-matching the message.
|
||||
export class SemaphoreAbortError extends Error {
|
||||
constructor(message) { super(message); this.name = "SemaphoreAbortError"; }
|
||||
}
|
||||
|
||||
export class TuiSemaphore {
|
||||
// limit: max concurrent slots. maxQueue: max waiters before run() rejects with backpressure.
|
||||
constructor(limit, { maxQueue } = {}) {
|
||||
this.limit = Math.max(1, parseInt(limit, 10) || 1);
|
||||
// Default queue cap: 32× the limit. Large enough that real family-burst traffic never
|
||||
// hits it, small enough that a pathological flood can't grow the queue without bound.
|
||||
this.maxQueue = Number.isFinite(maxQueue) ? maxQueue : this.limit * 32;
|
||||
this._inflight = 0;
|
||||
this._waiters = []; // FIFO queue of resolve callbacks waiting for a slot
|
||||
}
|
||||
|
||||
get inflight() { return this._inflight; }
|
||||
get queued() { return this._waiters.length; }
|
||||
|
||||
// Runtime-adjust the concurrency limit (audit finding F1 — a PATCH /settings maxConcurrent
|
||||
// change must actually take effect, not just be ignored until every currently-inflight task
|
||||
// happens to finish). Lowering the limit is handled lazily by release() (see below) — it
|
||||
// simply stops re-granting until inflight drains under the new, lower limit. Raising the
|
||||
// limit has immediate headroom, so we wake as many queued waiters as now fit.
|
||||
setLimit(limit) {
|
||||
this.limit = Math.max(1, parseInt(limit, 10) || 1);
|
||||
while (this._inflight < this.limit && this._waiters.length > 0) {
|
||||
const next = this._waiters.shift();
|
||||
this._inflight++;
|
||||
next();
|
||||
}
|
||||
}
|
||||
|
||||
// Acquire a slot. Resolves once a slot is free (immediately if under the limit, otherwise
|
||||
// when an in-flight task releases). Rejects synchronously-ish if the wait queue is full.
|
||||
// `signal` (optional AbortSignal, F2) lets the caller cancel a QUEUED wait — e.g. wired to
|
||||
// a client's socket "close" event so a request that disconnects before a slot is granted
|
||||
// is removed from the queue instead of eventually being handed a slot for a dead socket.
|
||||
// If `signal` is already aborted, reject immediately without ever touching the queue.
|
||||
acquire(signal) {
|
||||
if (signal?.aborted) {
|
||||
return Promise.reject(new SemaphoreAbortError("acquire aborted before requesting a slot"));
|
||||
}
|
||||
if (this._inflight < this.limit) {
|
||||
this._inflight++;
|
||||
return Promise.resolve();
|
||||
}
|
||||
if (this._waiters.length >= this.maxQueue) {
|
||||
return Promise.reject(new Error(
|
||||
`tui_queue_full: TUI concurrency limit (${this.limit}) reached and wait queue ` +
|
||||
`(${this.maxQueue}) is full`));
|
||||
}
|
||||
return new Promise((resolve, reject) => {
|
||||
let waiter; // the FIFO entry — captured so onAbort can find + splice exactly this one
|
||||
const onAbort = () => {
|
||||
const idx = this._waiters.indexOf(waiter);
|
||||
if (idx === -1) return; // already granted a slot (shifted out by release()/setLimit) — too late to cancel
|
||||
this._waiters.splice(idx, 1); // remove, not just flag — keeps `queued` accounting exact
|
||||
reject(new SemaphoreAbortError("acquire aborted while queued"));
|
||||
};
|
||||
waiter = () => {
|
||||
signal?.removeEventListener("abort", onAbort);
|
||||
resolve();
|
||||
};
|
||||
signal?.addEventListener("abort", onAbort, { once: true });
|
||||
this._waiters.push(waiter);
|
||||
});
|
||||
}
|
||||
|
||||
// Release a slot. Always frees the caller's own slot first, then re-grants it to the next
|
||||
// waiter ONLY if the (post-decrement) inflight count is still under the current limit (F1
|
||||
// fix). This is what makes a runtime-lowered limit actually bite: if the limit was lowered
|
||||
// while over-subscribed, releases stop re-granting and inflight drains toward the new limit
|
||||
// instead of a freed slot being handed straight back out at the old, higher occupancy.
|
||||
release() {
|
||||
if (this._inflight > 0) this._inflight--;
|
||||
if (this._inflight < this.limit) {
|
||||
const next = this._waiters.shift();
|
||||
if (next) {
|
||||
this._inflight++;
|
||||
next();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Run fn() under one slot. Releases in a finally so a throw (PR-A's honesty gates,
|
||||
// wallclock truncation, paste-not-landed, tmux spawn failure) NEVER leaks a slot.
|
||||
// `signal` (optional, F2) is forwarded to acquire() so a queued run() can be cancelled.
|
||||
async run(fn, signal) {
|
||||
await this.acquire(signal);
|
||||
try {
|
||||
return await fn();
|
||||
} finally {
|
||||
this.release();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── TUI drift observability (audit C-5) — pure helpers, importable for testing ──
|
||||
|
||||
// Record an observed cc_entrypoint into the (mutable) tuiStats counter. Sets lastEntrypoint
|
||||
// unconditionally and increments entrypointMismatches when the spawn was supposed to be
|
||||
// subscription-pool ("cli") but the transcript reported something else (a silent drift to
|
||||
// the metered Agent SDK pool — the audit's top risk after the 6/15 billing flip).
|
||||
// Returns true iff this observation was a mismatch (so the caller can also emit a log).
|
||||
export function recordTuiEntrypoint(tuiStats, observed, expectedMode = "cli") {
|
||||
tuiStats.lastEntrypoint = observed ?? null;
|
||||
const mismatch = expectedMode === "cli" && observed !== "cli";
|
||||
if (mismatch) tuiStats.entrypointMismatches++;
|
||||
return mismatch;
|
||||
}
|
||||
|
||||
// Build the additive /health `tui` block (ADR 0007 PR-B amendment). Pure: given the
|
||||
// config + live counters, returns the exact object embedded in /health. New fields only —
|
||||
// behaviour-preserving for existing /health consumers (grandfathered B.2 under ADR 0006).
|
||||
//
|
||||
// `pool` (optional, warm pane pool — lib/tui/pool.mjs): a TuiPanePool, or null/undefined
|
||||
// when the pool is off (the default). Reported as `pool: null` when off so the block's
|
||||
// shape stays stable, and as the pool's stats (size / warm / hits / misses / …) when on —
|
||||
// the operator's window onto both the hit rate and the standing idle-process cost.
|
||||
//
|
||||
// Streaming fields (backlog #2, OCP_TUI_STREAM) are ADDITIVE too:
|
||||
// streamEnabled — is real (MessageDisplay-hook) SSE streaming on for TUI turns?
|
||||
// streamTurns — streamed turns ATTEMPTED, counted before the truncation/auth-banner
|
||||
// gates run (F6) — so a turn REFUSED by those gates still shows up
|
||||
// here, which is exactly the turn an operator most wants visible.
|
||||
// Counting only turns that survived the gates would silently exclude
|
||||
// a turn's worst-case outcome from its own denominator.
|
||||
// streamDeltas — MessageDisplay hook fires OBSERVED, including held-back ones (F6) —
|
||||
// NOT only the ones forwarded to a client. This is what makes
|
||||
// streamZeroDeltaTurns meaningful: a turn can have streamDeltas
|
||||
// incrementing while still emitting nothing to the client (fully held
|
||||
// back, e.g. a short answer), which is healthy, vs. a hook that fired
|
||||
// zero times at all, which is not (see streamZeroDeltaTurns).
|
||||
// streamTopUps — turns where the delta stream was a safe PREFIX of the transcript but
|
||||
// not equal to it; OCP topped up from the transcript and served T.
|
||||
// Benign but worth watching — a persistent rate means the hook is
|
||||
// losing fires.
|
||||
// streamDivergences — turns REFUSED because emitted bytes were not a prefix of the
|
||||
// transcript. THE field to alert on for CORRECTNESS: it means the hook
|
||||
// and the transcript disagreed and OCP chose to fail rather than serve
|
||||
// unverifiable text.
|
||||
// streamZeroDeltaTurns — streamed turns where the hook fired ZERO times (F7). THE field to
|
||||
// alert on for AVAILABILITY: streamTopUps climbing is one fire dropped
|
||||
// here and there (benign); this climbing means the hook is not firing
|
||||
// AT ALL — e.g. `--settings` silently stopped registering it (a claude
|
||||
// version bump), or F3's truncated-script failure mode — and every
|
||||
// streamed turn is quietly degrading to fully-buffered with no error.
|
||||
export function buildTuiHealthBlock({ enabled, entrypointMode, maxConcurrent, streamEnabled = false }, tuiStats, semaphore, pool = null) {
|
||||
return {
|
||||
enabled,
|
||||
entrypointMode, // cli | auto | off
|
||||
lastEntrypoint: tuiStats.lastEntrypoint, // last observed cc_entrypoint, or null
|
||||
entrypointMismatches: tuiStats.entrypointMismatches,
|
||||
inflight: semaphore.inflight, // current concurrent TUI turns
|
||||
queued: semaphore.queued, // turns waiting for a slot
|
||||
maxConcurrent,
|
||||
pool: pool ? pool.stats() : null, // warm pane pool, or null when disabled
|
||||
streamEnabled,
|
||||
streamTurns: tuiStats.streamTurns ?? 0,
|
||||
streamDeltas: tuiStats.streamDeltas ?? 0,
|
||||
streamTopUps: tuiStats.streamTopUps ?? 0,
|
||||
streamDivergences: tuiStats.streamDivergences ?? 0,
|
||||
streamZeroDeltaTurns: tuiStats.streamZeroDeltaTurns ?? 0,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,782 @@
|
||||
// TUI-mode session driver: hosts an interactive `claude` in tmux, submits one
|
||||
// serialized prompt, awaits the transcript reader, tears down. OCP-specific.
|
||||
//
|
||||
// Authority: claude CLI v2.1.158 interactive mode (no -p / no --output-format
|
||||
// => cc_entrypoint=cli). Submission recipe validated by spikes T3/T6 on PI231.
|
||||
// See docs/superpowers/specs/2026-05-30-tui-mode-production-design.md.
|
||||
//
|
||||
// Trust handling: rather than answer the trust-folder dialog interactively (which
|
||||
// only appears on a cwd's FIRST encounter — sending a defensive "1" to an already
|
||||
// trusted cwd would inject a stray prompt turn), we PRE-TRUST the scratch cwd by
|
||||
// seeding <home>/.claude.json. Every turn then boots dialog-free and identical.
|
||||
import { spawnSync } from "node:child_process";
|
||||
import { mkdtempSync, writeFileSync, readFileSync, mkdirSync, existsSync, rmSync, statSync, renameSync, symlinkSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { readTuiTranscript } from "./transcript.mjs";
|
||||
import { prepareStreamHook, streamFilePath, parseDeltaChunk } from "./stream.mjs";
|
||||
|
||||
// F7 fix (audit finding, LOW): the prefix used to be a bare, host-wide constant
|
||||
// ("ocp-tui-"), so a SECOND OCP instance on the same host (e.g. a temporary
|
||||
// verification instance stood up alongside production — a real pattern used during
|
||||
// PR #144/#146 verification) would boot-reap and potentially kill-server the OTHER
|
||||
// instance's LIVE sessions: the coexistence guard below only ever spared foreign
|
||||
// PRODUCT prefixes (olp-tui-*), never a second ocp-tui-* instance on a different port.
|
||||
//
|
||||
// Fix: scope the prefix to the instance's own listen port. The port is the natural
|
||||
// stable per-instance discriminator on one host (two OCP instances cannot share a
|
||||
// port), so `ocp-tui-<port>-` uniquely namespaces this instance's sessions and makes
|
||||
// a same-host sibling OCP instance look exactly like a foreign product (olp-tui-*) to
|
||||
// the coexistence guard — its `ocp-tui-<otherPort>-*` sessions never match our own
|
||||
// prefix and are therefore never reaped/kill-server'd by us.
|
||||
//
|
||||
// LEGACY_SESSION_PREFIX / LEGACY_SESSION_NAME_RE describe the OLD bare-prefix shape
|
||||
// (pre-this-fix), retained ONLY for the boot-time legacy-zombie migration handled in
|
||||
// reapStaleTuiSessions (see comment there). No code path in this version ever CREATES
|
||||
// a legacy-shaped session name again — sessionPrefixForPort() is the only session-name
|
||||
// prefix constructor used going forward.
|
||||
export const LEGACY_SESSION_PREFIX = "ocp-tui-";
|
||||
// Exact legacy shape: LEGACY_SESSION_PREFIX + sessionId.slice(0, 8), where sessionId is
|
||||
// a randomUUID() — so the suffix is always exactly 8 lowercase hex characters with NO
|
||||
// further separator. The new port-scoped shape always inserts a "-" between the port
|
||||
// digits and the 8-hex suffix (see sessionPrefixForPort), so this regex can never match
|
||||
// a new-shape name: a new-shape suffix is `<port digits>-<8 hex>` (contains a literal
|
||||
// "-"), which `[0-9a-f]{8}$` anchored immediately after the prefix cannot satisfy.
|
||||
export const LEGACY_SESSION_NAME_RE = /^ocp-tui-[0-9a-f]{8}$/;
|
||||
|
||||
// Build this instance's own session-name prefix, scoped by its listen port so a
|
||||
// second OCP instance on the same host (different port) is never mistaken for "ours".
|
||||
export function sessionPrefixForPort(port) {
|
||||
return `ocp-tui-${port}-`;
|
||||
}
|
||||
|
||||
const TMUX = process.env.OCP_TUI_TMUX_BIN || "tmux";
|
||||
|
||||
const defaultTmux = (args, opts = {}) =>
|
||||
spawnSync(TMUX, args, { encoding: "utf8", ...opts });
|
||||
|
||||
// Kill ONLY our own stale sessions. Scoped to sessionPrefixForPort(port) so a co-hosted
|
||||
// OLP test instance's `olp-tui-*` sessions — AND a co-hosted second OCP instance's
|
||||
// `ocp-tui-<otherPort>-*` sessions — are never touched (F7 fix).
|
||||
//
|
||||
// Defunct-reaping (PI231 incident): the pane's `claude` process is a child of the
|
||||
// long-lived tmux SERVER daemon, NOT of the OCP node process — `tmux new-session -d`
|
||||
// returns the instant the server forks the pane, so node never becomes its parent and
|
||||
// therefore can NEVER waitpid()/reap it (a SIGKILL still needs the *parent* to reap, and
|
||||
// here that parent is the tmux server). `kill-session` destroys the session but the server
|
||||
// can leave the pane's `claude` (and any grandchildren claude spawned) as `<defunct>`
|
||||
// zombies that only the server can reap. Over many per-request spawn+teardown cycles these
|
||||
// accumulate (live evidence on PI231: 25 defunct `<claude>` over 30 days; `tmux kill-server`
|
||||
// dropped it 25→3). The only node-reachable action that ACTUALLY reaps them — rather than
|
||||
// merely re-signalling — is to stop the tmux server: when the server exits, the kernel
|
||||
// reparents its surviving children to init (PID 1), which reaps them immediately.
|
||||
//
|
||||
// `port` (required) is this instance's own listen port (server.mjs's PORT / lib/constants.mjs
|
||||
// DEFAULT_PORT resolution) — the SPOT for "which sessions are ours."
|
||||
//
|
||||
// ── POOL/REAPER INVARIANT (warm pane pool — lib/tui/pool.mjs) ───────────────────────────
|
||||
// A warm pooled pane is one of OUR OWN `ocp-tui-<port>-*` sessions that is ALIVE AND IDLE
|
||||
// BY DESIGN — and the periodic sweep runs precisely when the instance is idle, i.e. exactly
|
||||
// when the pool is full. Without an exemption the sweep would kill every warm pane on every
|
||||
// tick (and kill-server on top). The exemption is `spare`: a set of EXACT session names the
|
||||
// caller declares live. Three properties, all load-bearing:
|
||||
//
|
||||
// 1. A LIVE POOLED PANE IS NEVER REAPED — INCLUDING ONE THAT IS STILL BOOTING. It is in
|
||||
// `spare` (the pool's live registry), so it is skipped by name. The booting case is not
|
||||
// a footnote, it is the one that bit us: bootTuiPane creates the tmux session
|
||||
// SYNCHRONOUSLY and only then waits up to POOL_BOOT_MS for the input bar, so a pooled
|
||||
// session can be live for ~20 s before its boot resolves. The pool therefore mints the
|
||||
// pane's NAME up front and holds it in `_bootingPane`, so liveNames() can name — and
|
||||
// spare — a session whose boot has not finished. (An earlier version tracked only a
|
||||
// COUNT of in-flight boots; the sweep could not name that session and killed it.)
|
||||
// 2. A LEAKED/ORPHANED POOLED PANE IS STILL REAPED. Membership is by EXACT NAME from a
|
||||
// live in-memory registry — NOT by "looks pooled" (name shape). A pane the pool no
|
||||
// longer owns (handed out, dropped, cancelled, or left behind by a previous process
|
||||
// generation — whose registry died with it) is absent from `spare` and is killed like
|
||||
// any other stale session. Fail-safe: forgetting to pass `spare` reaps MORE, never less.
|
||||
// 3. KILL-SERVER NEVER KILLS A LIVE POOL PANE. A spared session suppresses kill-server
|
||||
// exactly as a foreign session does (it is a live child of the tmux server). The
|
||||
// consequence — that a permanently-full pool would permanently disable the defunct-
|
||||
// zombie reaping that ONLY kill-server can do — is resolved in server.mjs by DRAINING
|
||||
// the pool immediately before the sweep, so `spare` is empty on the normal tick and
|
||||
// kill-server still fires. `spare` is the belt-and-braces: a reap call site that
|
||||
// forgets to drain still cannot kill a live pane.
|
||||
//
|
||||
// `spare` (default: none) — iterable of session names, or a Set. Ignored when the pool is off.
|
||||
//
|
||||
// `includeLegacy` (default false): when true, sessions matching the exact OLD bare-prefix
|
||||
// shape (LEGACY_SESSION_NAME_RE) are ALSO treated as ours for kill-session purposes. This is
|
||||
// the boot-time legacy migration: an operator upgrading past this fix could otherwise be left
|
||||
// with orphaned bare-prefix zombie sessions from the PREVIOUS (pre-fix) process generation of
|
||||
// this SAME instance, since no live instance of the new version ever creates that shape again
|
||||
// — a legacy-shaped session found at boot is therefore presumed to be this instance's own
|
||||
// leftover, not a stranger's. Passed true ONLY from the one-time boot-reap call site in
|
||||
// server.mjs; the periodic idle-reap sweep does NOT set it, so a lingering legacy session
|
||||
// during steady-state is conservatively treated as foreign (correctly blocking kill-server)
|
||||
// rather than assumed to be ours on every 15-minute tick. Residual (accepted, documented):
|
||||
// if a genuinely-still-running PRE-FIX OCP instance is coexisting on the same host at the
|
||||
// exact moment a new instance boots, its live legacy-shaped session could be reaped — the
|
||||
// same class of residual risk the audit finding itself accepts ("no live instance of the new
|
||||
// version creates them"); this PR does not regress that scenario, it only removes the far
|
||||
// more common same-version collision (the actual F7 finding).
|
||||
export function reapStaleTuiSessions({ tmux = defaultTmux, port, includeLegacy = false, spare = null } = {}) {
|
||||
const r = tmux(["list-sessions", "-F", "#{session_name}"]);
|
||||
if (!r || r.status !== 0) return 0; // no tmux server / no sessions
|
||||
const names = String(r.stdout || "").split("\n").map((s) => s.trim()).filter(Boolean);
|
||||
const ownPrefix = sessionPrefixForPort(port);
|
||||
const spared = spare instanceof Set ? spare : new Set(spare || []);
|
||||
let killed = 0;
|
||||
let othersRemain = false;
|
||||
let sparedLive = 0;
|
||||
for (const name of names) {
|
||||
// Property 1+2: exemption is by EXACT NAME from the pool's live registry. A pooled-
|
||||
// LOOKING name that is not in the registry is an orphan and falls through to the
|
||||
// normal kill path below.
|
||||
if (spared.has(name)) { sparedLive++; continue; }
|
||||
const isOwn = name.startsWith(ownPrefix);
|
||||
const isLegacyOwn = includeLegacy && LEGACY_SESSION_NAME_RE.test(name);
|
||||
if (isOwn || isLegacyOwn) {
|
||||
tmux(["kill-session", "-t", name]);
|
||||
killed++;
|
||||
} else {
|
||||
othersRemain = true; // a session we do NOT own (olp-tui-*, a sibling ocp-tui-<otherPort>-*,
|
||||
// or — outside includeLegacy — a legacy-shaped name) — never kill-server
|
||||
}
|
||||
}
|
||||
// Reap defunct `claude` zombies: safe ONLY when the server is now ours-only/empty.
|
||||
// kill-server is what actually reaps (server exit reparents survivors to init); a
|
||||
// per-session kill cannot, since node is not the zombies' parent.
|
||||
//
|
||||
// Property 3: a SPARED session is a live child of this tmux server, so kill-server would
|
||||
// kill it — it therefore suppresses kill-server exactly as a foreign session does. On the
|
||||
// normal sweep the pool is drained first, so sparedLive is 0 and kill-server still fires.
|
||||
if (!othersRemain && sparedLive === 0) {
|
||||
tmux(["kill-server"]);
|
||||
}
|
||||
return killed;
|
||||
}
|
||||
|
||||
// ── Task 5: runTuiTurn ───────────────────────────────────────────────────
|
||||
|
||||
// Boot + paste-settle timing. Conservative defaults validated on PI231; env-tunable.
|
||||
const BOOT_MS = parseInt(process.env.OCP_TUI_BOOT_MS || "4000", 10); // max wait for input-ready
|
||||
// Readiness cap for a POOL pre-boot. Deliberately far more generous than BOOT_MS: BOOT_MS is
|
||||
// tight because a client is blocked on it, whereas a warm-pane boot happens in the background
|
||||
// with nobody waiting. Observed live at size=2: a refill booting alongside an in-flight turn
|
||||
// exceeded 4000 ms and was discarded (tui_pool_boot_failed), quietly costing hit rate for a
|
||||
// pane that was merely slow, not broken. Scales with OCP_TUI_BOOT_MS if an operator raises it.
|
||||
export const POOL_BOOT_MS = BOOT_MS * 5;
|
||||
const READY_POLL_MS = parseInt(process.env.OCP_TUI_READY_POLL_MS || "400", 10); // readiness / paste-verify poll interval
|
||||
const PASTE_VERIFY_MS = parseInt(process.env.OCP_TUI_PASTE_VERIFY_MS || "5000", 10); // max wait for pasted prompt to render
|
||||
// Hook-sink drain interval when streaming. 100ms: the hook fires at BLOCK granularity
|
||||
// (~5-7 fires per answer, seconds apart), so a finer poll buys nothing and a coarser one
|
||||
// would add visible lag to the first delta. Cheap — one readFileSync of a small file.
|
||||
const STREAM_POLL_MS = parseInt(process.env.OCP_TUI_STREAM_POLL_MS || "100", 10);
|
||||
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
// Capture the visible tmux pane as plain text (for readiness / paste verification).
|
||||
function tuiCapturePane(tmux, tmuxName) {
|
||||
const r = tmux(["capture-pane", "-p", "-t", tmuxName]);
|
||||
return (r && typeof r.stdout === "string") ? r.stdout : "";
|
||||
}
|
||||
|
||||
// True once claude's input bar is rendered and ready for keystrokes.
|
||||
function tuiInputReady(pane) {
|
||||
return /\? for shortcuts/.test(pane);
|
||||
}
|
||||
|
||||
// True once the pasted prompt has POSITIVELY landed in the input box. We only trust
|
||||
// affirmative signals — NOT "the placeholder is gone", which is unreliable (claude's
|
||||
// placeholder uses a curly quote `"`, randomized example text, and renders the big paste
|
||||
// a beat after paste-buffer returns; a "placeholder-gone" heuristic false-positived on the
|
||||
// still-empty box and made us submit Enter into nothing → issue #130 hang). Landed iff:
|
||||
// (a) the bracketed-paste indicator "[Pasted text" is present (large/multi-line paste), OR
|
||||
// (b) the prompt's own leading text appears in the pane (short/literal paste).
|
||||
function tuiPromptLanded(pane, prompt) {
|
||||
const flatPane = pane.replace(/\s+/g, " ");
|
||||
if (flatPane.includes("[Pasted text")) return true;
|
||||
const firstLine = String(prompt).split("\n").map(s => s.trim()).find(Boolean) || "";
|
||||
const needle = firstLine.replace(/\s+/g, " ").slice(0, 24);
|
||||
// C-4/#133: threshold lowered 3 → 2. A prompt whose first non-blank line is 1–2
|
||||
// chars ("hi", "ok") previously NEVER matched (needle.length >= 3) and never
|
||||
// surfaced "[Pasted text", so EVERY short prompt 5s-failed with tui_paste_not_landed
|
||||
// (live-reproduced: "hi"). The input box starts EMPTY (the curly-quote placeholder
|
||||
// is excluded by the affirmative-signal design above), so a >=2-char needle present
|
||||
// in the pane is the pasted prompt, not placeholder noise — false-positive risk is
|
||||
// low. We keep >=2 rather than >=1 because a single visible char is more likely to
|
||||
// collide with incidental glyphs in claude's chrome (borders, the "❯" prompt mark);
|
||||
// 2 chars is the floor that lands real prompts while staying conservative.
|
||||
return needle.length >= 2 && flatPane.includes(needle);
|
||||
}
|
||||
|
||||
async function pollUntil(fn, { timeoutMs, intervalMs }) {
|
||||
const deadline = Date.now() + timeoutMs;
|
||||
while (Date.now() < deadline) {
|
||||
try { if (fn()) return true; } catch { /* ignore, keep polling */ }
|
||||
await sleep(intervalMs);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Single-quote escaper for sh -c arguments.
|
||||
function shq(s) {
|
||||
return `'${String(s).replace(/'/g, "'\\''")}'`;
|
||||
}
|
||||
|
||||
// Pre-trust the scratch cwd by seeding the trust record in <home>/.claude.json so
|
||||
// the trust-folder dialog never appears. Verified-live trust shape:
|
||||
// projects["<cwd>"] = { hasTrustDialogAccepted: true, allowedTools: [], ... }
|
||||
// Idempotent + best-effort: a missing/unreadable .claude.json must not abort a
|
||||
// turn (a fresh cwd would then show the dialog once; the boot wait tolerates it).
|
||||
// Must run BEFORE the session boots so claude reads the trusted record at startup.
|
||||
export function ensureTuiCwdTrusted(home, cwd) {
|
||||
if (!home || !cwd) return;
|
||||
const path = `${home}/.claude.json`;
|
||||
let j, mode;
|
||||
try {
|
||||
j = JSON.parse(readFileSync(path, "utf8"));
|
||||
mode = statSync(path).mode & 0o777;
|
||||
} catch { return; }
|
||||
j.projects = j.projects || {};
|
||||
const entry = j.projects[cwd] || {};
|
||||
if (entry.hasTrustDialogAccepted === true) return; // already trusted, no rewrite
|
||||
entry.hasTrustDialogAccepted = true;
|
||||
if (!Array.isArray(entry.allowedTools)) entry.allowedTools = [];
|
||||
j.projects[cwd] = entry;
|
||||
// Atomic write (temp + rename on the same fs), preserving mode, so a crash
|
||||
// mid-write can never truncate the user's real ~/.claude.json. We seed ONLY the
|
||||
// per-project trust flag — NOT bypassPermissionsModeAccepted: the driver never
|
||||
// passes --dangerously-skip-permissions, so the bypass dialog cannot appear, and
|
||||
// onboarding completion is an A-path precondition (the host already runs claude).
|
||||
// NOTE: when the A-path moves to a dedicated scratch HOME (task #26), this writes
|
||||
// a file we fully own, removing the real-config-mutation concern entirely.
|
||||
try {
|
||||
const tmp = `${path}.ocp-tui.${process.pid}.tmp`;
|
||||
writeFileSync(tmp, JSON.stringify(j, null, 2), { mode });
|
||||
renameSync(tmp, path);
|
||||
} catch { /* best effort */ }
|
||||
}
|
||||
|
||||
// Resolve the HOME the TUI `claude` runs under. Three intents, decided by the env
|
||||
// token + an explicit OCP_TUI_HOME override:
|
||||
//
|
||||
// - ENV-TOKEN MODE (default when CLAUDE_CODE_OAUTH_TOKEN is set AND OCP_TUI_HOME is
|
||||
// unset): a CREDENTIAL-FREE scratch home at `<realHome>/.ocp-tui/home`. There is
|
||||
// deliberately NO .credentials.json (no symlink, no copy), so the only credential
|
||||
// claude can find is the long-lived env token (passed by buildTuiCmd). This is what
|
||||
// actually FORCES env-token auth — see the prepareTuiHome comment for why passing
|
||||
// the token alone is insufficient.
|
||||
// - EXPLICIT OVERRIDE: whatever OCP_TUI_HOME names (back-compat; an operator who set it
|
||||
// keeps exactly that home).
|
||||
// - REAL-HOME (default when the env token is unset): the operator's real home, shared
|
||||
// credentials.json — byte-for-byte the pre-fix behaviour for credentials.json hosts.
|
||||
//
|
||||
// Pure + deterministic so server.mjs and the tests share one decision. `configuredHome`
|
||||
// is the raw OCP_TUI_HOME value (undefined/empty => unset).
|
||||
export const DEFAULT_TUI_SCRATCH_HOME = (realHome) => `${realHome}/.ocp-tui/home`;
|
||||
export function resolveTuiHome({ realHome, configuredHome, envTokenSet }) {
|
||||
if (configuredHome) return configuredHome; // explicit override wins (back-compat)
|
||||
if (envTokenSet) return DEFAULT_TUI_SCRATCH_HOME(realHome); // credential-free scratch
|
||||
return realHome; // legacy real-home default
|
||||
}
|
||||
|
||||
// Prepare the HOME claude runs under. Three modes:
|
||||
// - real-home (tuiHome === realHome OR falsy): no isolation; just trust the cwd
|
||||
// in the real ~/.claude.json. The legacy default when no env token is set.
|
||||
// - ENV-TOKEN scratch-home (envTokenMode === true): a dedicated HOME with a seeded
|
||||
// .claude.json (onboarded + trusts only the scratch cwd) and its own projects/ dir,
|
||||
// and DELIBERATELY NO .credentials.json (no symlink, no copy). claude then has no
|
||||
// credentials file to read, so it authenticates via CLAUDE_CODE_OAUTH_TOKEN (passed
|
||||
// by buildTuiCmd) — which is authoritative precisely because nothing shadows it.
|
||||
// - legacy scratch-home (envTokenMode falsy, tuiHome !== realHome): the historical
|
||||
// mode that SYMLINKS the real .credentials.json. Retained only for an operator who
|
||||
// explicitly set OCP_TUI_HOME without an env token; see the caveat below.
|
||||
//
|
||||
// WHY ENV-TOKEN MODE IS THE FIX (proven live on PI231, claude 2.1.104):
|
||||
// env token passed + a broken ~/.claude/.credentials.json present → 401.
|
||||
// env token passed + credentials.json moved aside → real answer.
|
||||
// Interactive `claude` PREFERS .credentials.json over the env var (unlike `-p`, where the
|
||||
// env token wins), so a stale/corrupt credentials.json SHADOWS the env token. Passing the
|
||||
// token is necessary but insufficient; the TUI claude must run in a HOME with NO
|
||||
// credentials.json so the env token is the only credential. This ALSO ends the refresh-
|
||||
// corruption incident at the root: with no credentials file, claude never runs the token-
|
||||
// refresh path, so the single-use refresh token can never be rotated (and corrupted) by the
|
||||
// spawn+kill cycle. (This RESOLVES — not reintroduces — the ADR 0007 scratch-home concern:
|
||||
// the old caveat was about a SYMLINKED credentials.json being forked on refresh; here there
|
||||
// is no credentials file to fork and no refresh ever happens.)
|
||||
//
|
||||
// ⚠️ LEGACY SCRATCH-HOME CAVEAT (envTokenMode falsy, symlink path): claude rewrites
|
||||
// .credentials.json on token refresh, REPLACING the symlink with a regular-file copy → the
|
||||
// scratch home FORKS the OAuth credentials and a refresh can invalidate the real-home token.
|
||||
// That path is therefore safe only with a DEDICATED OAuth or for ephemeral use. The env-token
|
||||
// mode above avoids this entirely.
|
||||
//
|
||||
// Idempotent + best-effort: any failure degrades toward the dialog/cap, never corrupts.
|
||||
// Run BEFORE the session boots.
|
||||
export function prepareTuiHome(realHome, tuiHome, cwd, { envTokenMode = false } = {}) {
|
||||
if (!tuiHome || tuiHome === realHome) { ensureTuiCwdTrusted(realHome, cwd); return; }
|
||||
try {
|
||||
const claudeDir = `${tuiHome}/.claude`;
|
||||
mkdirSync(`${claudeDir}/projects`, { recursive: true });
|
||||
if (!envTokenMode) {
|
||||
// Legacy mode ONLY: symlink the real credentials (never copy the token); refresh if
|
||||
// missing. Env-token mode deliberately skips this — no credentials file at all.
|
||||
const link = `${claudeDir}/.credentials.json`;
|
||||
if (!existsSync(link)) {
|
||||
try { symlinkSync(`${realHome}/.claude/.credentials.json`, link); } catch { /* best effort */ }
|
||||
}
|
||||
}
|
||||
// Seed .claude.json ONCE (if absent): onboarded + trust ONLY the scratch cwd.
|
||||
// In env-token mode start from a MINIMAL config (do NOT copy the real ~/.claude.json —
|
||||
// a credential-isolated home should not inherit the operator's account/config state);
|
||||
// in legacy mode carry the onboarded real config minus the user's project history.
|
||||
const seedPath = `${tuiHome}/.claude.json`;
|
||||
if (!existsSync(seedPath)) {
|
||||
let base = {};
|
||||
if (!envTokenMode) {
|
||||
try { base = JSON.parse(readFileSync(`${realHome}/.claude.json`, "utf8")); } catch { /* fresh */ }
|
||||
}
|
||||
base.hasCompletedOnboarding = true;
|
||||
base.projects = { [cwd]: { hasTrustDialogAccepted: true, allowedTools: [] } };
|
||||
writeFileSync(seedPath, JSON.stringify(base, null, 2), { mode: 0o600 });
|
||||
}
|
||||
} catch { /* best effort */ }
|
||||
// Ensure the cwd is trusted in the scratch config (idempotent; atomic).
|
||||
ensureTuiCwdTrusted(tuiHome, cwd);
|
||||
}
|
||||
|
||||
// Build interactive claude argv: NO -p, NO --output-format (=> cc_entrypoint=cli).
|
||||
// MCP hard-disabled: --strict-mcp-config (no --mcp-config) is the only mechanism
|
||||
// that stops account-attached managed MCP from connecting (spec §5.2 / T6),
|
||||
// belt-and-braces with --disallowedTools "mcp__*".
|
||||
// A-PATH ONLY: built-in tools are left enabled (acceptable single-user). Deployment B
|
||||
// (guest keys) MUST additionally pass --tools "" per spec §5.2(2) as the credential
|
||||
// wall before this argv is reachable for owner_tier=guest — guard that in PR-3 wiring.
|
||||
//
|
||||
// `stream` (optional, OCP_TUI_STREAM): { file, settings } — when present, the pane gets
|
||||
// (a) OCP_TUI_STREAM_FILE in its env — read by the static MessageDisplay hook script to
|
||||
// decide WHERE to append this pane's deltas. Delivered as env (not baked into the
|
||||
// settings file) so the settings file stays STATIC and a pre-booted warm pane works.
|
||||
// Verified live: a claude hook inherits the pane's environment.
|
||||
// (b) --settings <file> — registers the MessageDisplay hook.
|
||||
// VERIFIED LIVE (claude 2.1.207, this host) before shipping, because both were spawn-level
|
||||
// risks:
|
||||
// - the startup banner is UNCHANGED with --settings: "Sonnet 4.6 with low effort ·
|
||||
// Claude Max" (subscription pool). --settings is NOT a --bare-class flag — it does not
|
||||
// silently drop the subscription pool. Transcript entrypoint stayed "cli".
|
||||
// - --settings MERGES into the settings hierarchy, it does NOT clobber <HOME>/.claude/
|
||||
// settings.json: with --settings passed, the user-level settings.json's `env` block was
|
||||
// still applied to the hook's environment. So the isolated-HOME settings story the TUI
|
||||
// already relies on (permissions / additionalDirectories — see prepareTuiHome and the
|
||||
// OCP_TUI_FULL_TOOLS note above) survives intact.
|
||||
// When absent, the argv is byte-for-byte the pre-streaming argv.
|
||||
export function buildTuiCmd(claudeBin, model, sessionId, ehome, entrypointMode, stream = null) {
|
||||
// Deliver claude's env via an `env` prefix on the PANE COMMAND — tmux does NOT forward the
|
||||
// spawning process's environment to the pane, and `new-session -e` needs tmux ≥3.2 (the cloud
|
||||
// host runs 2.7), so this is the only portable, reliable mechanism (verified live 2026-06-01:
|
||||
// passing {env} to spawnSync left the pane with only HOME). DISABLE_AUTOUPDATER pins the version
|
||||
// (no "What's new" splash that delayed input-readiness); CLAUDE_CODE_ENTRYPOINT labels the
|
||||
// billing pool (set below per entrypointMode).
|
||||
//
|
||||
// CLAUDE_CODE_DISABLE_CLAUDE_MDS + DISABLE_AUTO_MEMORY: OCP is a PROXY, not a Claude Code
|
||||
// session. The proxied client (OpenClaw / an IDE) owns its own context and memory; the HOST's
|
||||
// CLAUDE.md and auto-memory must NEVER leak into the agent OCP runs on the user's behalf.
|
||||
// Without these, claude loads the host's project/user CLAUDE.md + memory into every proxied
|
||||
// turn — verified live 2026-06-02: a cwd CLAUDE.md ("end every reply with QUACKMARKER_42") was
|
||||
// obeyed by the proxied turn until these flags were set, after which it was not. Unconditional
|
||||
// by design (not gated): proxy purity is not an opt-in. Harmless on hosts with no CLAUDE.md
|
||||
// (the common case — they suppress nothing). Mirrors the -p path's CLAUDE_NO_CONTEXT vars.
|
||||
const sets = [
|
||||
`HOME=${shq(ehome)}`,
|
||||
"DISABLE_AUTOUPDATER=1",
|
||||
"CLAUDE_CODE_DISABLE_OFFICIAL_MARKETPLACE_AUTOINSTALL=1",
|
||||
"CLAUDE_CODE_DISABLE_CLAUDE_MDS=1",
|
||||
"CLAUDE_CODE_DISABLE_AUTO_MEMORY=1",
|
||||
];
|
||||
// CLAUDE_CODE_OAUTH_TOKEN: tmux does NOT forward the parent process's env to the pane (the
|
||||
// same reason the whole env is delivered as an `env` prefix above — verified live 2026-06-01),
|
||||
// so the token MUST be set explicitly here or the spawned `claude` never sees it. Without it,
|
||||
// the TUI claude falls back to authenticating via <HOME>/.claude/.credentials.json, whose
|
||||
// single-use refresh token gets corrupted by the per-request spawn + `kill-session` teardown
|
||||
// racing claude's token-rotation write (the PI231 incident: refresh token ended up an empty
|
||||
// string → permanent 401 "Please run /login", re-login re-corrupted on the next spawn). With
|
||||
// the long-lived OAuth token in env, claude authenticates via the token and never touches the
|
||||
// credentials.json refresh path — matching how the stable oracle / Mac-mini hosts already run.
|
||||
//
|
||||
// SECURITY: the token appears in the pane command (ps-visible). This is acceptable for the
|
||||
// single-user A-path — it mirrors the existing plaintext-token practice (server.mjs reads the
|
||||
// same CLAUDE_CODE_OAUTH_TOKEN env at getOAuthCredentials()), and the multi-user B-path is
|
||||
// already refused at boot (TUI + AUTH_MODE=multi is a hard FATAL). Read from process.env here,
|
||||
// consistent with how buildTuiCmd already reads OCP_TUI_FULL_TOOLS / CLAUDE_ALLOWED_TOOLS below.
|
||||
//
|
||||
// When the env is unset (e.g. a host that intentionally relies on credentials.json), no token
|
||||
// is added — behaviour is byte-for-byte unchanged from before this fix.
|
||||
if (process.env.CLAUDE_CODE_OAUTH_TOKEN) {
|
||||
sets.push(`CLAUDE_CODE_OAUTH_TOKEN=${shq(process.env.CLAUDE_CODE_OAUTH_TOKEN)}`);
|
||||
}
|
||||
// Streaming sink: the pane's own per-session delta file (see the `stream` note above).
|
||||
if (stream && stream.file) sets.push(`OCP_TUI_STREAM_FILE=${shq(stream.file)}`);
|
||||
const unset = ["CLAUDECODE", "ANTHROPIC_API_KEY", "ANTHROPIC_BASE_URL", "ANTHROPIC_AUTH_TOKEN"];
|
||||
if (entrypointMode === "cli") sets.push("CLAUDE_CODE_ENTRYPOINT=cli");
|
||||
else if (entrypointMode === "auto") unset.push("CLAUDE_CODE_ENTRYPOINT"); // let claude self-classify via TTY
|
||||
const envPrefix = ["env", ...unset.map((u) => `-u ${u}`), ...sets].join(" ");
|
||||
|
||||
// Tool surface.
|
||||
// DEFAULT (safe): hard-disable MCP (--strict-mcp-config + --disallowedTools mcp__*);
|
||||
// built-in tools stay on, acceptable for single-user A-path.
|
||||
// OCP_TUI_FULL_TOOLS=1: grant the SAME tool surface as the -p A-path
|
||||
// (--allowedTools [+ --mcp-config]), so a SINGLE-USER / trusted TUI deployment can
|
||||
// run a tool-using agent (e.g. an OpenClaw assistant that needs Bash/Read/Write/MCP)
|
||||
// on the subscription pool. ALWAYS uses --allowedTools (CLAUDE_SKIP_PERMISSIONS /
|
||||
// --dangerously-skip-permissions is intentionally removed: claude v2.1.x shows an
|
||||
// interactive bypass-acceptance screen in headless tmux that nothing can answer →
|
||||
// the turn hangs until the wallclock cap, bricks the pane; not recoverable without a
|
||||
// human at a keyboard). Use scratch-home settings.json additionalDirectories instead.
|
||||
let toolArgs;
|
||||
if (process.env.OCP_TUI_FULL_TOOLS === "1") {
|
||||
toolArgs = [];
|
||||
const allowed = (process.env.CLAUDE_ALLOWED_TOOLS ||
|
||||
"Bash,Read,Write,Edit,Glob,Grep,WebSearch,WebFetch,Agent")
|
||||
.split(",").map((s) => s.trim()).filter(Boolean);
|
||||
// shq EACH token: buildTuiCmd returns a SHELL STRING (run by tmux via sh -c), unlike
|
||||
// buildCliArgs which returns an argv array to spawn(). claude accepts scoped specifiers
|
||||
// like "Bash(npm run test:*)" / "Read(~/**)" whose ( ) * ~ would break/inject the shell
|
||||
// command if pasted bare. (operator-self-injection only — guests can't reach TUI.)
|
||||
if (allowed.length) toolArgs.push("--allowedTools", ...allowed.map(shq));
|
||||
if (process.env.CLAUDE_MCP_CONFIG) toolArgs.push("--mcp-config", shq(process.env.CLAUDE_MCP_CONFIG));
|
||||
} else {
|
||||
toolArgs = ["--strict-mcp-config", "--disallowedTools", shq("mcp__*")];
|
||||
}
|
||||
|
||||
// Effort: pass --effort EXPLICITLY. Without it, the pane's claude inherits a
|
||||
// HOME-dependent effortLevel — real-home mode inherits the operator's
|
||||
// ~/.claude/settings.json (whatever they set for their own interactive use),
|
||||
// env-token scratch mode inherits claude's built-in default (prepareTuiHome never
|
||||
// writes effortLevel) — so latency silently depends on which HOME mode
|
||||
// resolveTuiHome() picked AND on an unrelated operator setting. Pinning it here
|
||||
// removes both. Measured (docs/plans/2026-07-13-tui-latency): explicit low cuts
|
||||
// direct-spawn TTFT p50 10.35s → 6.17s (−40%) and collapses the spread ~15×;
|
||||
// banner-verified to stay on the subscription pool (`· Claude Max`).
|
||||
// OCP_TUI_EFFORT=inherit restores the pre-flag argv byte-for-byte (no --effort).
|
||||
// An unknown value falls back to the default rather than reaching claude's argv:
|
||||
// a typo'd --effort value must not risk a spawn-time usage error in the pane.
|
||||
const EFFORT_LEVELS = ["low", "medium", "high", "xhigh", "max"]; // claude 2.1.207 --help
|
||||
const effortRaw = (process.env.OCP_TUI_EFFORT || "low").trim().toLowerCase();
|
||||
let effortArgs;
|
||||
if (effortRaw === "inherit") {
|
||||
effortArgs = [];
|
||||
} else if (EFFORT_LEVELS.includes(effortRaw)) {
|
||||
effortArgs = ["--effort", effortRaw];
|
||||
} else {
|
||||
console.error(`[tui] invalid OCP_TUI_EFFORT=${JSON.stringify(process.env.OCP_TUI_EFFORT)}; using "low" (valid: ${EFFORT_LEVELS.join("|")}, or "inherit" to omit the flag)`);
|
||||
effortArgs = ["--effort", "low"];
|
||||
}
|
||||
|
||||
// --settings registers the MessageDisplay hook. Omitted entirely when streaming is off,
|
||||
// so the OFF argv is byte-for-byte the pre-streaming argv.
|
||||
const settingsArgs = stream && stream.settings ? ["--settings", shq(stream.settings)] : [];
|
||||
|
||||
return [
|
||||
envPrefix,
|
||||
shq(claudeBin),
|
||||
"--model", shq(model),
|
||||
"--session-id", sessionId,
|
||||
...toolArgs,
|
||||
...effortArgs,
|
||||
...settingsArgs,
|
||||
].join(" ");
|
||||
}
|
||||
|
||||
// Is a pane alive AND still sitting at its input bar? Used by the warm pool to decide,
|
||||
// at hand-out time, whether a pre-booted pane is still usable (a dead/degraded pane must
|
||||
// become a MISS → cold path, never a hung turn). capture-pane exits non-zero when the
|
||||
// session no longer exists, so this covers "pane gone" and "pane not ready" in one call.
|
||||
export function tuiPaneHealthy(tmux, tmuxName) {
|
||||
const r = tmux(["capture-pane", "-p", "-t", tmuxName]);
|
||||
if (!r || r.status !== 0 || typeof r.stdout !== "string") return false;
|
||||
return tuiInputReady(r.stdout);
|
||||
}
|
||||
|
||||
// Pool pane names carry a "p" marker after the port-scoped prefix:
|
||||
// turn pane: ocp-tui-<port>-<8hex> (unchanged)
|
||||
// pool pane: ocp-tui-<port>-p<8hex>
|
||||
// Purely for operator legibility (`tmux ls` shows which panes are warm). It is NOT the
|
||||
// reaper's exemption mechanism — that is the exact-name spare set (see the POOL/REAPER
|
||||
// INVARIANT above), so a pooled-LOOKING orphan is still reaped. Both shapes start with
|
||||
// sessionPrefixForPort(port), so both remain reapable as "ours", and neither can match
|
||||
// LEGACY_SESSION_NAME_RE.
|
||||
export function poolPaneName(port, sessionId) {
|
||||
return sessionPrefixForPort(port) + "p" + sessionId.slice(0, 8);
|
||||
}
|
||||
|
||||
// Boot ONE interactive `claude` pane and wait for its input bar. Shared by the cold
|
||||
// request path (runTuiTurn) and the warm pool (lib/tui/pool.mjs) so a pooled pane is
|
||||
// spawned with byte-for-byte the same argv, HOME, cwd and trust preparation as a
|
||||
// cold-booted one — the pool must not become a second, drifting spawn path.
|
||||
//
|
||||
// Each pane gets its OWN fresh randomUUID() --session-id, fixed at boot. That is what
|
||||
// keeps a pooled pane single-use-safe: its transcript holds exactly one exchange.
|
||||
//
|
||||
// requireReady: the cold path tolerates a readiness timeout (it falls through and lets
|
||||
// the paste-verify decide — pre-existing behaviour, unchanged). The POOL sets it, because
|
||||
// a pane that never reached its input bar is worthless as a warm pane and must not be
|
||||
// enlisted: throw, let the pool count a bootFailure, and leave the request path to
|
||||
// cold-boot as usual.
|
||||
// bootMs: max wait for the input bar. Defaults to BOOT_MS (the REQUEST path's cap, which is
|
||||
// deliberately tight — a client is blocked on it). The POOL passes POOL_BOOT_MS instead: a
|
||||
// background pre-boot has nobody waiting on it, and capping it at the request-path's 4 s
|
||||
// made real refills fail (observed live: a refill booting alongside an in-flight turn took
|
||||
// >4 s and was discarded, silently lowering the hit rate). Slow != broken for a pre-boot.
|
||||
// `sessionId` / `name` (both optional): the caller may supply the pane's identity instead of
|
||||
// letting bootTuiPane mint it. The POOL does, because it must know the tmux session's NAME
|
||||
// before this function runs — the session is created synchronously below, well before the
|
||||
// readiness wait returns, so a pool that only learned the name on resolve could neither spare
|
||||
// the session from the reaper nor kill it on shutdown. Supplying BOTH also keeps the name's
|
||||
// hex suffix equal to the session-id's, so `tmux ls` correlates to the transcript file.
|
||||
// `streamDir` (optional, OCP_TUI_STREAM): install claude's MessageDisplay hook on this pane.
|
||||
// Done HERE, at boot — not at turn time — and that is the whole reason streaming survives the
|
||||
// WARM POOL: the hook script + settings file are STATIC (one pair per streamDir), and the only
|
||||
// per-turn thing, the sink path, is derived from the pane's own --session-id, which is fixed
|
||||
// right here. So a pre-booted pane already carries its hook and its own sink and streams exactly
|
||||
// like a cold-booted one; nothing request-specific is ever baked into the spawn.
|
||||
export async function bootTuiPane({
|
||||
model, claudeBin, home, realHome, cwd, port, entrypointMode = "cli",
|
||||
tmux = defaultTmux, sessionId = null, name = null, requireReady = false, bootMs = BOOT_MS,
|
||||
streamDir = null,
|
||||
}) {
|
||||
const sid = sessionId || randomUUID();
|
||||
// Port-scoped session name (F7 fix) — see sessionPrefixForPort / reapStaleTuiSessions
|
||||
// for why this instance's own listen port is the namespace discriminator.
|
||||
const tmuxName = name || (sessionPrefixForPort(port) + sid.slice(0, 8));
|
||||
const ehome = home || process.env.HOME; // HOME claude runs under (scratch or real)
|
||||
const rhome = realHome || process.env.HOME; // real home (OAuth + onboarded config source)
|
||||
|
||||
// Env-token-only mode: the env token is set AND claude runs in an isolated home
|
||||
// (ehome !== rhome). In that case the scratch home must be CREDENTIAL-FREE (no
|
||||
// .credentials.json) so the env token — passed by buildTuiCmd — is the only credential
|
||||
// and is therefore authoritative (interactive claude otherwise PREFERS a credentials.json,
|
||||
// shadowing the env token; proven live on PI231). server.mjs derives TUI_HOME via
|
||||
// resolveTuiHome() so this isolated home is the DEFAULT once CLAUDE_CODE_OAUTH_TOKEN is set.
|
||||
const envTokenMode = !!process.env.CLAUDE_CODE_OAUTH_TOKEN && ehome !== rhome;
|
||||
|
||||
// Ensure scratch cwd exists, then prepare the (scratch or real) HOME + trust the
|
||||
// cwd — before claude boots.
|
||||
if (!existsSync(cwd)) mkdirSync(cwd, { recursive: true });
|
||||
prepareTuiHome(rhome, ehome, cwd, { envTokenMode });
|
||||
|
||||
// Streaming sink for THIS pane (see the streamDir note above). rmSync first so a
|
||||
// re-used session-id can never replay a previous turn's deltas.
|
||||
let streamFile = null, streamSettings = null;
|
||||
if (streamDir) {
|
||||
streamFile = streamFilePath(streamDir, sid);
|
||||
streamSettings = prepareStreamHook(streamDir);
|
||||
try { rmSync(streamFile, { force: true }); } catch { /* start from a fresh sink */ }
|
||||
}
|
||||
|
||||
// Minimal env for spawnSync (tmux itself). The pane's claude env comes exclusively
|
||||
// from the `env` prefix string built inside buildTuiCmd — tmux does NOT forward the
|
||||
// spawning process's env to the pane, so the {env} here is intentionally minimal.
|
||||
const env = { ...process.env };
|
||||
env.HOME = ehome; // tmux needs HOME; all claude-specific vars go via buildTuiCmd prefix
|
||||
|
||||
// Boot the interactive session inside tmux, rooted at the scratch cwd.
|
||||
// Capture the result: if tmux new-session fails (status !== 0) there is no PTY, no
|
||||
// interactive spawn — abort BEFORE the boot wait rather than paste into a non-existent
|
||||
// session or issue a billing request without a verified interactive context.
|
||||
const spawnResult = tmux(
|
||||
["new-session", "-d", "-s", tmuxName, "-x", "220", "-y", "50", "-c", cwd,
|
||||
buildTuiCmd(claudeBin, model, sid, ehome, entrypointMode,
|
||||
streamFile ? { file: streamFile, settings: streamSettings } : null)],
|
||||
{ env },
|
||||
);
|
||||
if (!spawnResult || spawnResult.status !== 0) {
|
||||
throw new Error("tui_spawn_failed: tmux session not created");
|
||||
}
|
||||
|
||||
// Wait until claude's input bar is actually ready (not a blind sleep).
|
||||
// bootMs is the MAX readiness wait, not a fixed delay.
|
||||
const ready = await pollUntil(() => tuiInputReady(tuiCapturePane(tmux, tmuxName)),
|
||||
{ timeoutMs: bootMs, intervalMs: READY_POLL_MS });
|
||||
if (!ready) {
|
||||
if (requireReady) {
|
||||
try { tmux(["kill-session", "-t", tmuxName]); } catch { /* already gone */ }
|
||||
throw new Error("tui_pane_not_ready: input bar did not appear within " + bootMs + "ms");
|
||||
}
|
||||
// Cold path (pre-existing behaviour): readiness timed out; rely on paste-verify.
|
||||
console.error("[tui] input_not_ready", tmuxName);
|
||||
}
|
||||
return { name: tmuxName, sessionId: sid, model, ehome, streamFile, bootedAt: Date.now() };
|
||||
}
|
||||
|
||||
// Full per-request TUI lifecycle:
|
||||
// 1. Take a WARM pane from the pool if one is available for this model (opt-in;
|
||||
// OCP_TUI_POOL_SIZE=0 => always null => steps 2-3 below are exactly today's path).
|
||||
// A pooled pane is SINGLE-USE: it already carries its own fresh --session-id, it
|
||||
// serves this one turn, and it is killed in the finally like any other pane.
|
||||
// 2. On a MISS: pre-trust the scratch cwd, boot an interactive `claude` in a fresh tmux
|
||||
// session in the scratch cwd, poll capture-pane until the `? for shortcuts` input bar
|
||||
// appears (bootTuiPane). BOOT_MS is the max wait, not a fixed delay.
|
||||
// 3. Write prompt to a 0600 temp file (no shell injection from prompt content).
|
||||
// 4. Paste the prompt via tmux load-buffer + paste-buffer -p (bracketed paste) —
|
||||
// reliable for large multi-line prompts where send-keys -l is not (issue #130).
|
||||
// Poll-verify the prompt landed in the input (placeholder gone / [Pasted text]);
|
||||
// fast-fail with tui_paste_not_landed if it never lands (prevents the 120s
|
||||
// wallclock "stuck typing" hang). Then submit with a SEPARATE Enter key event.
|
||||
// 5. Block on the native JSONL transcript (located by THIS pane's session-id) until
|
||||
// terminal marker or wall-clock cap.
|
||||
// 6. Always teardown: kill session + rm temp dir (even on throw), and kick a background
|
||||
// pool refill so the next request finds a warm pane.
|
||||
// Returns { text, entrypoint } from readTuiTranscript (entrypoint is the billing-pool
|
||||
// classifier, e.g. "cli", or null if the transcript did not include a turn_duration).
|
||||
//
|
||||
// STREAMING (OCP_TUI_STREAM, default off). Pass `onDelta` and `streamDir`, and the pane's
|
||||
// MessageDisplay hook (installed by bootTuiPane; see lib/tui/stream.mjs) appends each raw
|
||||
// delta payload to the pane's own sink. This driver polls that sink and invokes onDelta(payload)
|
||||
// per fire while the turn is still generating. A WARM pane already carries its sink from boot
|
||||
// (pane.streamFile), so the pooled and cold paths stream identically.
|
||||
//
|
||||
// `streamDir` IS PASSED TO THE COLD BOOT UNCONDITIONALLY (not gated on `onDelta`) — F4 fix. The
|
||||
// spawn argv is this project's billing-classification surface: a caller with OCP_TUI_STREAM on
|
||||
// but THIS particular request non-streaming (stream:false) must still get the SAME argv whether
|
||||
// it lands on a pool HIT or a cold-boot MISS, because a pre-booted pool pane cannot know in
|
||||
// advance whether the request it will eventually serve wants streaming — it installs the hook
|
||||
// unconditionally whenever the pool is warming at all (see server.mjs's bootPane closure). Gating
|
||||
// the cold boot's hook install on `onDelta` made a stream:false request's argv depend on whether
|
||||
// it happened to hit the pool or miss it — the exact drift this surface cannot tolerate. Whether
|
||||
// the hook is actually POLLED is a separate, correctly-scoped decision: see `streaming` below,
|
||||
// gated on onDelta && streamFile, so a non-streaming turn never reads its own sink even though
|
||||
// the hook is running.
|
||||
//
|
||||
// The transcript stays AUTHORITATIVE regardless: it is still the terminal-turn signal, still the
|
||||
// source of the returned `text`, and still the input to the caller's honesty gates. The delta
|
||||
// stream is a low-latency MIRROR of it, never a replacement, and the caller asserts the two
|
||||
// agree. With onDelta AND streamDir both omitted, nothing here changes: no poll, no hook.
|
||||
//
|
||||
// `abortSignal` (optional): aborts the transcript wait, so a client that disconnects mid-turn
|
||||
// tears the pane down NOW (the finally below) instead of holding the pane — and therefore the
|
||||
// caller's semaphore slot — until the turn or the wallclock cap ends.
|
||||
export async function runTuiTurn({
|
||||
prompt,
|
||||
model,
|
||||
claudeBin,
|
||||
home,
|
||||
realHome,
|
||||
cwd,
|
||||
port,
|
||||
wallclockMs = 120000,
|
||||
entrypointMode = "cli",
|
||||
tmux = defaultTmux,
|
||||
pool = null, // TuiPanePool | null — null (default) === today's cold-boot-only path
|
||||
onPane = null, // optional observer: ({ warm }) => void, for logging/metrics
|
||||
onDelta = null, // (payload) => void — invoked per MessageDisplay hook fire, mid-turn
|
||||
streamDir = null, // hook sink dir, passed to the COLD boot UNCONDITIONALLY (F4 — see above);
|
||||
// a warm pane brings its own, fixed at its own boot
|
||||
abortSignal = null,
|
||||
}) {
|
||||
// 1. Warm pane, or cold boot. A MISS is never an error — it is exactly today's path.
|
||||
let pane = pool ? pool.acquire(model) : null;
|
||||
const warm = !!pane;
|
||||
// Kick the refill IMMEDIATELY (not after the turn): the replacement pane then boots
|
||||
// CONCURRENTLY with this turn and is warm by the time the next request arrives. Also
|
||||
// runs on a MISS — acquire() has just retargeted the pool to this model, so the miss
|
||||
// that cold-boots today warms the pool for the next caller. Fire-and-forget; it takes
|
||||
// no TuiSemaphore slot (see pool.refill's SLOT ACCOUNTING note).
|
||||
if (pool) pool.refill();
|
||||
if (onPane) { try { onPane({ warm }); } catch { /* observer must never break a turn */ } }
|
||||
if (!pane) {
|
||||
// streamDir passed AS-IS (not gated on onDelta) — F4: see the STREAMING comment above.
|
||||
pane = await bootTuiPane({ model, claudeBin, home, realHome, cwd, port, entrypointMode, tmux,
|
||||
streamDir });
|
||||
}
|
||||
const tmuxName = pane.name;
|
||||
const sessionId = pane.sessionId; // THIS pane's own session-id — one session, one turn
|
||||
const ehome = pane.ehome || home || process.env.HOME;
|
||||
|
||||
// Streaming state is read off the PANE, not recomputed here — a warm pane fixed its sink at
|
||||
// boot, and a cold one just did the same above. If the pool was booted WITHOUT a streamDir
|
||||
// while onDelta is set, streamFile is null and the turn degrades to buffered: correct, just
|
||||
// not fast. (server.mjs wires the same streamDir into both paths so that cannot happen.)
|
||||
const streamFile = pane.streamFile || null;
|
||||
const streaming = !!(onDelta && streamFile);
|
||||
const streamCursor = { consumed: 0 };
|
||||
let streamStopped = false;
|
||||
let pollTimer = null;
|
||||
// Drain every complete line appended since the last drain. Never throws into the turn: a
|
||||
// malformed line is skipped by parseDeltaChunk, and an onDelta that throws is contained.
|
||||
const drainDeltas = () => {
|
||||
if (!streaming) return;
|
||||
let text;
|
||||
try { text = readFileSync(streamFile, "utf8"); } catch { return; } // absent until the first fire
|
||||
const { deltas, consumed } = parseDeltaChunk(text, streamCursor.consumed);
|
||||
streamCursor.consumed = consumed;
|
||||
for (const d of deltas) {
|
||||
try { onDelta(d); } catch { /* a sink error must never abort the turn */ }
|
||||
}
|
||||
};
|
||||
|
||||
// Write prompt to a temp file (mode 0600) so the content never touches argv.
|
||||
const tmpDir = mkdtempSync(`${tmpdir()}/ocp-tui-`);
|
||||
const promptFile = `${tmpDir}/prompt.txt`;
|
||||
writeFileSync(promptFile, prompt, { mode: 0o600 });
|
||||
|
||||
try {
|
||||
// 3. Paste the prompt via a tmux PASTE BUFFER with bracketed paste (-p), NOT
|
||||
// `send-keys -l`. send-keys of a large multi-line prompt is unreliable: the
|
||||
// embedded newlines arrive as separate key events (effectively repeated Enter),
|
||||
// so a big OpenClaw-style prompt never lands and the turn hangs to the wallclock
|
||||
// (issue #130 — reproduced at ~300 lines; fixed by bracketed paste). load-buffer
|
||||
// reads the file directly (no shell arg limit, no `"$(cat)"`), and paste-buffer -p
|
||||
// wraps it in bracketed-paste markers so claude ingests it atomically as ONE paste
|
||||
// ("[Pasted text #N +M lines]"). -d deletes the buffer afterward. Buffer name is the
|
||||
// per-session tmuxName, so concurrent turns never collide.
|
||||
tmux(["load-buffer", "-b", tmuxName, promptFile]);
|
||||
tmux(["paste-buffer", "-b", tmuxName, "-t", tmuxName, "-p", "-d"]);
|
||||
|
||||
// Verify the prompt POSITIVELY landed before submitting; poll (a large bracketed paste
|
||||
// takes a beat to render the "[Pasted text]" indicator). This is load-bearing: firing
|
||||
// Enter before the paste renders submits an empty box → the turn hangs to the wallclock
|
||||
// (issue #130). Fast-fail if it never lands → deterministic error in seconds.
|
||||
const landed = await pollUntil(() => tuiPromptLanded(tuiCapturePane(tmux, tmuxName), prompt),
|
||||
{ timeoutMs: PASTE_VERIFY_MS, intervalMs: READY_POLL_MS });
|
||||
if (!landed) {
|
||||
throw new Error("tui_paste_not_landed: prompt did not reach claude's input within " + PASTE_VERIFY_MS + "ms");
|
||||
}
|
||||
|
||||
// Submit (separate Enter key event).
|
||||
tmux(["send-keys", "-t", tmuxName, "Enter"]);
|
||||
|
||||
// 5a. Streaming only: start polling the hook sink. Runs CONCURRENTLY with the
|
||||
// transcript wait below — the deltas are what make the answer visible while the
|
||||
// turn is still generating; the transcript is what makes it authoritative.
|
||||
if (streaming) {
|
||||
const loop = () => {
|
||||
if (streamStopped) return;
|
||||
drainDeltas();
|
||||
pollTimer = setTimeout(loop, STREAM_POLL_MS);
|
||||
};
|
||||
pollTimer = setTimeout(loop, STREAM_POLL_MS);
|
||||
}
|
||||
|
||||
// 5b. Block on the native transcript (resolved by THIS pane's session-id) until terminal.
|
||||
// Returns { text, entrypoint, truncated } from readTuiTranscript.
|
||||
const result = await readTuiTranscript({ home: ehome, sessionId, wallclockMs, abortSignal });
|
||||
|
||||
// 5c. FINAL drain. The terminal marker can land between two poll ticks, so the last
|
||||
// delta(s) may still be unread — without this the tail would be missing from the
|
||||
// stream and every turn would need a transcript top-up.
|
||||
streamStopped = true;
|
||||
if (pollTimer) clearTimeout(pollTimer);
|
||||
drainDeltas();
|
||||
return result;
|
||||
} finally {
|
||||
// 6. Teardown — always, even on throw (including an abortSignal disconnect, which is
|
||||
// exactly why the pane cannot outlive a client that walked away). A pooled pane is
|
||||
// torn down here exactly like a cold-booted one: SINGLE-USE, never returned (pool.mjs).
|
||||
streamStopped = true;
|
||||
if (pollTimer) clearTimeout(pollTimer);
|
||||
try { tmux(["kill-session", "-t", tmuxName]); } catch { /* already gone */ }
|
||||
try { rmSync(tmpDir, { recursive: true, force: true }); } catch { /* best effort */ }
|
||||
if (streamFile) { try { rmSync(streamFile, { force: true }); } catch { /* best effort */ } }
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,288 @@
|
||||
// TUI-mode real SSE streaming — the `MessageDisplay` hook sink.
|
||||
//
|
||||
// WHAT THIS IS. `claude` fires a **MessageDisplay** hook per rendered block of the
|
||||
// assistant's reply, handing the hook the RAW MARKDOWN SOURCE of an incremental
|
||||
// `delta` on stdin. Registered via `--settings` on the ordinary interactive TUI spawn
|
||||
// (NO -p, NO --bare — the billing pool is untouched), it is the only byte-faithful
|
||||
// incremental source the interactive CLI exposes. Everything here consumes that hook
|
||||
// surface AS EMITTED — forwarding, not inventing.
|
||||
//
|
||||
// ALIGNMENT.md: **Class B**. We consume claude's own hook payload and re-emit it in the
|
||||
// OpenAI chat/completions streaming shapes OCP already speaks (ADR 0006). There is no
|
||||
// `cli.js` citation because no `cli.js` function is being mirrored: the TUI spawn is
|
||||
// OCP-owned surface (ADR 0007), and the hook payload is claude's own published contract.
|
||||
//
|
||||
// THE VERIFIED CONTRACT (docs/plans/2026-07-13-tui-latency/streaming-spike.md, and
|
||||
// independently reproduced on claude 2.1.207 / sonnet-4-6 / banner `· Claude Max`):
|
||||
//
|
||||
// payload (stdin, one JSON object per fire):
|
||||
// { hook_event_name:"MessageDisplay", session_id, transcript_path, prompt_id, cwd,
|
||||
// turn_id, message_id, index, final, delta }
|
||||
//
|
||||
// - deltas carry the raw markdown source (`## `, `**`, ```javascript all present)
|
||||
// - concat(deltas of one message) === T, byte-exactly (T = extractLatestAssistantText)
|
||||
// - T.startsWith(concat(deltas[0..n])) at EVERY n (prefix-stable)
|
||||
// - block-level granularity (~5-7 fires per answer), NOT token-level
|
||||
// - only `text` blocks fire it — thinking blocks are excluded (what OCP wants)
|
||||
//
|
||||
// ⚠️ THE HOOK IS SYNCHRONOUS. The hook's source sets `forceSyncExecution: true` —
|
||||
// `claude` BLOCKS on every fire. The hook script must therefore write and exit, doing
|
||||
// NO work inline. Measured cost of the script below: p50 7.2 ms / p90 14.7 ms per fire,
|
||||
// i.e. ~50 ms added blocking across a whole ~7-delta turn against a 6-10 s turn. That is
|
||||
// noise, so a plain append is the right sink — a FIFO would be faster on paper but a FIFO
|
||||
// blocks its writer until a reader attaches, which would hand `claude` a way to hang.
|
||||
//
|
||||
// WARM-POOL COMPATIBILITY (load-bearing — a warm pane pool is a separate in-flight PR).
|
||||
// The hook script and the settings file are BOTH STATIC: one copy per stream dir, written
|
||||
// once, never per-request. The per-turn destination is carried in the PANE'S OWN ENV as
|
||||
// `OCP_TUI_STREAM_FILE` (verified live: a hook inherits the pane's environment), and the
|
||||
// path is derived from the session-id — which for a pre-booted pane is fixed at BOOT.
|
||||
// Nothing about a request is baked into the settings file at spawn time, so a pane booted
|
||||
// before its request arrives streams exactly the same way.
|
||||
import { writeFileSync, mkdirSync, renameSync } from "node:fs";
|
||||
import { detectTuiUpstreamError } from "./transcript.mjs";
|
||||
|
||||
// Default holdback before the first byte is released to the client. See TuiDeltaAssembler.
|
||||
export const DEFAULT_HOLDBACK_CHARS = 100;
|
||||
|
||||
// Resolve OCP_TUI_STREAM_HOLDBACK to a SAFE value. The whole C-1 auth-banner guarantee rests
|
||||
// on the holdback being at least the default banner detector's max message length — which is
|
||||
// exactly DEFAULT_HOLDBACK_CHARS. So this is a FLOOR, not a hint: a smaller value (or a NaN
|
||||
// typo like "unlimited"/"5MB") would let a real banner fragment release before the terminal
|
||||
// detector could classify the whole message, silently reopening the leak the assembler exists
|
||||
// to prevent. The env var's own doc says "Only raise it"; this enforces that instead of trusting
|
||||
// it. Returns { value, clamped } so the caller can warn when it had to clamp — a silent floor is
|
||||
// less honest than a noticed one.
|
||||
export function resolveStreamHoldback(raw, floor = DEFAULT_HOLDBACK_CHARS) {
|
||||
const parsed = parseInt(raw ?? "", 10);
|
||||
if (!Number.isFinite(parsed)) return { value: floor, clamped: raw != null && String(raw).trim() !== "" };
|
||||
if (parsed < floor) return { value: floor, clamped: true };
|
||||
return { value: parsed, clamped: false };
|
||||
}
|
||||
|
||||
// The hook script. POSIX sh, no interpreter startup beyond /bin/sh, one fork (`cat`).
|
||||
//
|
||||
// - `printf` is a shell BUILTIN in sh/dash/bash, so the newline costs no fork.
|
||||
// - the `{ cat; printf '\n'; } >>` group opens the file ONCE and appends both writes
|
||||
// through the same O_APPEND fd, so a payload and its terminator can never be split
|
||||
// by another writer. (They never race anyway: one file per pane, and MessageDisplay
|
||||
// is synchronous within a pane.)
|
||||
// - a payload JSON can never contain a literal newline — JSON.stringify escapes them —
|
||||
// so "one line == one payload" holds, and a torn write is always a trailing partial
|
||||
// line, which parseDeltaChunk() leaves unconsumed until it completes.
|
||||
// - NO OCP_TUI_STREAM_FILE (e.g. a pane booted with streaming off, or any other claude
|
||||
// session that happens to load this settings file) => swallow stdin and exit 0. The
|
||||
// hook must NEVER fail or block: claude is waiting on it.
|
||||
export const HOOK_SCRIPT = `#!/bin/sh
|
||||
# OCP TUI streaming sink — claude fires this per MessageDisplay block and BLOCKS on it.
|
||||
# Write and exit. Never do work here.
|
||||
[ -n "\$OCP_TUI_STREAM_FILE" ] || exec cat >/dev/null
|
||||
{ cat; printf '\\n'; } >> "\$OCP_TUI_STREAM_FILE"
|
||||
`;
|
||||
|
||||
// The --settings payload registering the hook. Static: no per-request data.
|
||||
export function buildStreamSettings(hookScriptPath) {
|
||||
return { hooks: { MessageDisplay: [{ hooks: [{ type: "command", command: hookScriptPath }] }] } };
|
||||
}
|
||||
|
||||
export const hookScriptPath = (streamDir) => `${streamDir}/md-hook.sh`;
|
||||
export const streamSettingsPath = (streamDir) => `${streamDir}/settings.json`;
|
||||
// One file per session-id. For a pre-booted (warm) pane the session-id is fixed at boot,
|
||||
// so this path is knowable at boot — which is what keeps the pool compatible.
|
||||
export const streamFilePath = (streamDir, sessionId) => `${streamDir}/${sessionId}.jsonl`;
|
||||
|
||||
// Atomic write: temp file + rename (same-directory, same-filesystem, so rename is atomic on
|
||||
// POSIX). A process killed mid-`writeFileSync` leaves the TEMP file half-written, never the
|
||||
// real path — `path` always names either the old complete content or the new complete
|
||||
// content, never a torn one. That matters specifically for md-hook.sh: it is SYNCHRONOUS
|
||||
// (claude blocks on every fire), so a truncated script would still pass `existsSync`, still
|
||||
// get exec'd, and fail/hang on every single MessageDisplay fire with no operator-visible
|
||||
// symptom short of streaming going silently dead (F7's streamZeroDeltaTurns is the backstop
|
||||
// for exactly that). Mirrors ensureTuiCwdTrusted's tmp+renameSync pattern in session.mjs.
|
||||
function writeFileAtomic(path, content, mode) {
|
||||
const tmp = `${path}.${process.pid}.tmp`;
|
||||
writeFileSync(tmp, content, { mode });
|
||||
renameSync(tmp, path);
|
||||
}
|
||||
|
||||
// Write the static hook script + settings file into `streamDir`. UNCONDITIONAL, not
|
||||
// write-if-missing: these files persist across OCP restarts at `streamDir`, so a host that
|
||||
// booted once under an older version and never had its stream dir cleared would otherwise be
|
||||
// silently stuck on a stale HOOK_SCRIPT / buildStreamSettings() forever — no future OCP
|
||||
// upgrade could ever reach it. Safe to call every boot: the content is static (no per-request
|
||||
// data), so a same-content rewrite is the overwhelmingly common case and costs two tiny
|
||||
// atomic writes, not a per-turn expense. Returns the settings path to hand to `claude
|
||||
// --settings`.
|
||||
export function prepareStreamHook(streamDir) {
|
||||
mkdirSync(streamDir, { recursive: true });
|
||||
const script = hookScriptPath(streamDir);
|
||||
const settings = streamSettingsPath(streamDir);
|
||||
writeFileAtomic(script, HOOK_SCRIPT, 0o700);
|
||||
writeFileAtomic(settings, JSON.stringify(buildStreamSettings(script), null, 2), 0o600);
|
||||
return settings;
|
||||
}
|
||||
|
||||
// Parse newly-appended sink lines. `consumed` is the number of COMPLETE lines already
|
||||
// taken; only lines terminated by "\n" are complete, so a payload caught mid-write stays
|
||||
// unconsumed until its terminator lands. Returns the fresh MessageDisplay payloads plus
|
||||
// the new consumed count. Pure — the caller owns the cursor.
|
||||
export function parseDeltaChunk(text, consumed = 0) {
|
||||
const lines = String(text ?? "").split("\n");
|
||||
const complete = lines.slice(0, -1); // the tail after the last "\n" is a partial line
|
||||
const deltas = [];
|
||||
for (const line of complete.slice(consumed)) {
|
||||
const t = line.trim();
|
||||
if (!t) continue;
|
||||
try {
|
||||
const o = JSON.parse(t);
|
||||
if (o && o.hook_event_name === "MessageDisplay" && typeof o.delta === "string") deltas.push(o);
|
||||
} catch { /* not ours / not parseable — skip, never throw into the request path */ }
|
||||
}
|
||||
return { deltas, consumed: complete.length };
|
||||
}
|
||||
|
||||
// ── The assembler: hook deltas → client bytes, with the honesty gates intact ──
|
||||
//
|
||||
// Two jobs, both load-bearing.
|
||||
//
|
||||
// 1. THE AUTH-BANNER HOLDBACK (C-1 / issue #133 must survive streaming).
|
||||
// The interactive CLI renders an auth failure as ordinary assistant TEXT — so an
|
||||
// expired-credential turn fires MessageDisplay with the BANNER as its delta, and a
|
||||
// naive forwarder would stream "Please run /login · API Error: 401 …" to the client as
|
||||
// a normal answer, exactly the silent-error case C-1 exists to prevent.
|
||||
// detectTuiUpstreamError() classifies a WHOLE message, so it cannot be run per-delta.
|
||||
// Instead we HOLD BACK the first `holdbackChars` characters. The default detector only
|
||||
// ever fires on a message of <= 100 chars (TUI_ERR_MAX_LEN — real banners are 69 and 73),
|
||||
// so once the TRIMMED accumulation EXCEEDS 100 chars the final text cannot be a banner by
|
||||
// that detector's own length rule, and releasing is safe. An answer that never exceeds the
|
||||
// holdback is simply delivered whole at terminal — i.e. exactly today's buffered
|
||||
// behaviour, gates and all.
|
||||
// THE GUARANTEE HAS TWO HALVES, both required — neither alone is sufficient:
|
||||
// (i) Nothing is emitted for a message until its trimmed accumulation exceeds the
|
||||
// detector's max banner length. This is what keeps the FIRST message of a turn
|
||||
// safe: a banner-length message can never clear the holdback.
|
||||
// (ii) Once a message boundary follows an emit (`restartedAfterEmit`), push() stops
|
||||
// emitting ENTIRELY for the rest of the turn — a SECOND message (e.g. an
|
||||
// auth-failure banner rendered mid-turn, after tool-using prose already streamed)
|
||||
// gets zero bytes forwarded, not just a fresh holdback of its own. finalize() then
|
||||
// refuses the whole turn (SSE error frame, no cache) precisely because the first
|
||||
// message's bytes are unretractable and unverifiable against T. Without this half,
|
||||
// (i) alone only protects the FIRST message per turn — see F1.
|
||||
// ⚠️ Soundness is w.r.t. the DEFAULT detector. An operator who REPLACES it via
|
||||
// CLAUDE_TUI_ERROR_PATTERNS with a pattern that can match a longer message must raise
|
||||
// OCP_TUI_STREAM_HOLDBACK past their longest banner; server.mjs warns at boot. That is the
|
||||
// one case (i) does not cover — (ii) still applies regardless. Even past both, the
|
||||
// terminal gate still refuses to cache a banner and still ends the stream on an SSE error
|
||||
// frame rather than finish_reason:"stop" — the holdback is the first of two layers, not
|
||||
// the only one.
|
||||
//
|
||||
// 2. MESSAGE SCOPING (keeps `concat === T` the RIGHT assertion).
|
||||
// The transcript's T is extractLatestAssistantText() — the LAST text-bearing assistant
|
||||
// entry, not every assistant entry. A tool-using turn therefore has TWO messages
|
||||
// (prose → tool_use → answer) and T is only the second. So the assembler scopes to the
|
||||
// CURRENT message_id: when a new message_id appears and NOTHING has been emitted yet,
|
||||
// the held text is DISCARDED — the transcript is about to discard it too, so this keeps
|
||||
// us byte-identical to the buffered path instead of streaming prose the buffered path
|
||||
// would have dropped. When a new message_id appears AFTER we have already emitted, the
|
||||
// bytes are gone and cannot be retracted: finalize() then reports !ok and the caller
|
||||
// fails the turn loudly (SSE error frame, no cache, counted on /health). Fail-loud is
|
||||
// the correct posture — a proxy that silently serves text the transcript disagrees with
|
||||
// is the exact class of bug ALIGNMENT.md exists to prevent.
|
||||
// Sentinel for "no message seen yet". Deliberately not null/undefined — see the constructor.
|
||||
const NO_MESSAGE_YET = Symbol("no-message-yet");
|
||||
|
||||
export class TuiDeltaAssembler {
|
||||
constructor({ holdbackChars = DEFAULT_HOLDBACK_CHARS, detectError = detectTuiUpstreamError } = {}) {
|
||||
this.holdbackChars = holdbackChars;
|
||||
this.detectError = detectError;
|
||||
this.emitted = ""; // bytes ALREADY written to the client — unretractable
|
||||
this.pending = ""; // held back, not yet written
|
||||
this.released = false;
|
||||
// NOT null: a payload may legitimately carry message_id === null, and if the sentinel were
|
||||
// also null the FIRST such payload would compare equal to it, register no boundary, and
|
||||
// leave `messages` at 0 — which used to disarm the restartedAfterEmit guard below entirely.
|
||||
// A unique object is === to nothing a JSON payload can produce, so the first fire ALWAYS
|
||||
// registers as message 1, whatever its message_id is (or isn't).
|
||||
this.messageId = NO_MESSAGE_YET;
|
||||
this.deltas = 0; // hook fires seen
|
||||
this.messages = 0; // distinct message_ids seen
|
||||
this.restartedAfterEmit = false;
|
||||
}
|
||||
|
||||
// All hook bytes for the CURRENT message (emitted + still held).
|
||||
get full() { return this.emitted + this.pending; }
|
||||
|
||||
// Feed one MessageDisplay payload. Returns the text to emit NOW, or null (held back).
|
||||
push(payload) {
|
||||
const delta = payload && typeof payload.delta === "string" ? payload.delta : "";
|
||||
const mid = payload ? payload.message_id : null;
|
||||
if (mid !== this.messageId) {
|
||||
this.messageId = mid;
|
||||
this.messages++;
|
||||
if (this.emitted === "") {
|
||||
this.pending = ""; // safe: the transcript will drop this message too
|
||||
} else {
|
||||
// A boundary while bytes are ALREADY out is unrecoverable, full stop — the count of
|
||||
// messages seen so far is irrelevant. The old `else if (this.messages > 1)` guard was
|
||||
// the sole reason a null-message_id first payload could disarm F1: it left `messages`
|
||||
// at 0, so the real boundary evaluated 1 > 1 === false and never armed. The invariant
|
||||
// is "a boundary occurred while emitted !== ''", and that is exactly what this says.
|
||||
this.restartedAfterEmit = true; // unrecoverable — finalize() will refuse the turn
|
||||
}
|
||||
}
|
||||
this.deltas++;
|
||||
// F1: once a message boundary has followed an emit, the turn is ALREADY unrecoverable —
|
||||
// finalize() will refuse it (see restartedAfterEmit above). `this.released` stays true
|
||||
// from the FIRST message's release and, uncorrected, lets every later message's deltas
|
||||
// stream straight through unfiltered — exactly the auth-banner-mid-turn leak this class
|
||||
// exists to prevent. Stop emitting HERE, permanently, for the rest of the turn: there is
|
||||
// nothing left to gain from continuing to forward bytes for a turn that will be refused,
|
||||
// and every byte forwarded now is one more the client cannot be told to un-see.
|
||||
if (this.restartedAfterEmit) return null;
|
||||
if (!delta) return null;
|
||||
|
||||
if (this.released) {
|
||||
this.emitted += delta;
|
||||
return delta;
|
||||
}
|
||||
this.pending += delta;
|
||||
// Release only once the TRIMMED accumulation is past the banner detector's reach.
|
||||
// detectTuiUpstreamError() trims before measuring length (TUI_ERR_MAX_LEN is a trimmed-
|
||||
// length bound), so gating release on the UNTRIMMED pending.length let a run of >
|
||||
// holdbackChars whitespace trim down to "" — detectError("") sees nothing to classify,
|
||||
// returns null, and release fires with the holdback never having actually screened
|
||||
// anything. Trimming here keeps both sides of the check talking about the same string.
|
||||
if (this.pending.trim().length > this.holdbackChars && this.detectError(this.pending) == null) {
|
||||
const out = this.pending;
|
||||
this.pending = "";
|
||||
this.released = true;
|
||||
this.emitted += out;
|
||||
return out;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Reconcile against the AUTHORITATIVE transcript text T. Call only AFTER the truncation
|
||||
// and auth-banner gates have passed. Returns:
|
||||
// { ok:true, tail, exact } — tail is the remaining text to emit (may be ""). `exact`
|
||||
// is concat(deltas) === T; when false we still serve exactly
|
||||
// T, having topped up from the transcript, and the caller
|
||||
// counts a topUp.
|
||||
// { ok:false, ... } — what we already emitted is NOT a prefix of T. The client
|
||||
// holds bytes the transcript disagrees with; the caller must
|
||||
// NOT cache and must end the stream on an SSE error frame.
|
||||
finalize(T) {
|
||||
const text = typeof T === "string" ? T : "";
|
||||
const full = this.full;
|
||||
if (!text.startsWith(this.emitted)) {
|
||||
return { ok: false, tail: null, exact: false, emitted: this.emitted.length, transcript: text.length };
|
||||
}
|
||||
return {
|
||||
ok: true,
|
||||
tail: text.slice(this.emitted.length),
|
||||
exact: full === text,
|
||||
emitted: this.emitted.length,
|
||||
transcript: text.length,
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,301 @@
|
||||
// Transcript reader for TUI-mode. Reads claude's native JSONL session transcript
|
||||
// and returns the latest assistant turn's text once the turn is terminal.
|
||||
//
|
||||
// Authority: claude CLI v2.1.157 — interactive session transcript at
|
||||
// <HOME>/.claude/projects/<CWD with every "/" -> "-">/<--session-id>.jsonl
|
||||
// Completion marker: a line {"type":"system","subtype":"turn_duration",...}.
|
||||
// See docs/superpowers/specs/2026-05-30-tui-mode-production-design.md §4.
|
||||
import { readFileSync, existsSync, readdirSync } from "node:fs";
|
||||
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
// Locate a session's transcript by its UUID across every projects subdir, without
|
||||
// reconstructing the encoded cwd. Robust to whatever encoding claude applies.
|
||||
// Returns the path, or null if not present yet (it appears once the turn starts).
|
||||
// TODO: add a CI fixture-contract test (a captured real transcript) so schema drift
|
||||
// in the claude JSONL format fails loudly rather than silently degrading.
|
||||
export function findTranscriptPath(home, sessionId) {
|
||||
if (!home || !sessionId) return null;
|
||||
const root = `${home}/.claude/projects`;
|
||||
let dirs;
|
||||
try { dirs = readdirSync(root); } catch { return null; }
|
||||
for (const d of dirs) {
|
||||
const candidate = `${root}/${d}/${sessionId}.jsonl`;
|
||||
if (existsSync(candidate)) return candidate;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Parse NDJSON text into objects; skip blank lines and partial/forming lines
|
||||
// (the live transcript is read mid-write, so the last line may be incomplete).
|
||||
export function parseTranscriptLines(text) {
|
||||
const out = [];
|
||||
for (const line of text.split("\n")) {
|
||||
const t = line.trim();
|
||||
if (!t) continue;
|
||||
try { out.push(JSON.parse(t)); } catch { /* partial line being written */ }
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// A line marks the assistant turn complete when EITHER:
|
||||
// (a) {type:"system", subtype:"turn_duration"} — emitted by newer claude builds
|
||||
// (e.g. 2.1.159), OR
|
||||
// (b) {type:"assistant"} whose message.stop_reason is a FINAL reason
|
||||
// ("end_turn" / "stop_sequence" / "max_tokens"). This is the API-level
|
||||
// end-of-turn signal, present across claude builds whose transcripts do NOT
|
||||
// emit turn_duration (e.g. 2.1.114 — verified live on the cloud host). Without
|
||||
// it OCP can't detect completion on those builds and hangs to the wallclock,
|
||||
// then returns only partial text (issue #130, cloud/server-side symptom).
|
||||
//
|
||||
// stop_reason "tool_use" is deliberately NOT terminal: the model is mid-turn (it will
|
||||
// run a tool and continue with a later assistant entry). Matching on a FINAL
|
||||
// stop_reason — not on the mere presence of a tool_use — keeps tool-using turns intact.
|
||||
// (The v3.17.1 narrowing dropped a buggy "tool_use is terminal" rule; this restores
|
||||
// cross-version completion detection without bringing that bug back.)
|
||||
const TERMINAL_STOP_REASONS = new Set(["end_turn", "stop_sequence", "max_tokens"]);
|
||||
export function isTerminalLine(obj) {
|
||||
if (!obj || typeof obj !== "object") return false;
|
||||
if (obj.type === "system" && obj.subtype === "turn_duration") return true;
|
||||
if (obj.type === "assistant" && obj.message && typeof obj.message === "object") {
|
||||
return TERMINAL_STOP_REASONS.has(obj.message.stop_reason);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Text of the LAST assistant turn: concatenate its text content blocks
|
||||
// (ignore thinking/tool_use blocks). Later assistant entries overwrite earlier.
|
||||
// Fixture-confirmed shape: top-level type:"assistant", message.content[] array.
|
||||
//
|
||||
// Scoping: this returns the FINAL text-bearing assistant entry in the whole file,
|
||||
// not "text since the matching user line" (spec §4.2). Those are equivalent ONLY
|
||||
// under OCP's one-session-per-request model (a fresh --session-id => a fresh
|
||||
// transcript holding one logical exchange). If a future warm-pool ever reuses a
|
||||
// session WITHOUT a fresh session-id / clear, earlier-turn text could leak — that
|
||||
// author must add user-line scoping here. See spec §7.2.
|
||||
//
|
||||
// STATUS (warm pool, lib/tui/pool.mjs — the "future warm-pool" this note anticipated):
|
||||
// the pool does NOT reuse sessions, so the precondition above still holds and no
|
||||
// user-line scoping was added. Each pooled pane is booted with its OWN fresh
|
||||
// randomUUID() --session-id (bootTuiPane) and is SINGLE-USE: it serves exactly one turn
|
||||
// and is then killed and replaced. One session still means one logical exchange, so the
|
||||
// last assistant entry is still that request's answer.
|
||||
// The warning therefore stands UNCHANGED for anyone who later wants a pane to serve a
|
||||
// SECOND turn (or to reset one with /clear and reuse it): that is a leak, and it needs
|
||||
// user-line scoping HERE before it can be safe. Do not relax pool.mjs's single-use rule
|
||||
// without doing that work first.
|
||||
export function extractLatestAssistantText(events) {
|
||||
let text = "";
|
||||
for (const ev of events) {
|
||||
if (!ev || ev.type !== "assistant") continue;
|
||||
const content = ev.message && ev.message.content;
|
||||
if (!Array.isArray(content)) continue;
|
||||
const parts = content
|
||||
.filter((b) => b && b.type === "text" && typeof b.text === "string")
|
||||
.map((b) => b.text);
|
||||
if (parts.length) text = parts.join("");
|
||||
}
|
||||
return text;
|
||||
}
|
||||
|
||||
// Returns the entrypoint string (e.g. "cli") used for the billing-pool assertion,
|
||||
// or null if absent. Lets callers assert the subscription-classified path.
|
||||
//
|
||||
// Resolution order (C-3, issue #133):
|
||||
// 1. PREFER the turn_duration system line's `entrypoint` — the authoritative
|
||||
// end-of-turn classifier emitted by builds that produce turn_duration
|
||||
// (e.g. claude-2.1.104/2.1.157 on PI231).
|
||||
// 2. FALL BACK to the `entrypoint` field on ANY ordinary transcript line
|
||||
// (assistant / user / attachment / system) — present on BOTH emitting and
|
||||
// non-emitting builds. Some claude builds (e.g. certain Mac mini transcripts)
|
||||
// do NOT emit a turn_duration line at all; reading ONLY turn_duration made the
|
||||
// caller's tui_entrypoint_mismatch assertion (server.mjs) get got:null every
|
||||
// turn and go blind. The entrypoint value is identical across line types within
|
||||
// a single interactive session (fixture-confirmed: every line in
|
||||
// complete-haiku.jsonl carrying `entrypoint` reads "cli"), so the fallback
|
||||
// yields the same classifier. Last-writer-wins on the fallback.
|
||||
export function verifyEntrypoint(events) {
|
||||
let fallback = null;
|
||||
for (const ev of events) {
|
||||
if (!ev || typeof ev !== "object") continue;
|
||||
if (ev.type === "system" && ev.subtype === "turn_duration" && ev.entrypoint != null) {
|
||||
return ev.entrypoint; // authoritative — short-circuit
|
||||
}
|
||||
if (ev.entrypoint != null) fallback = ev.entrypoint;
|
||||
}
|
||||
return fallback;
|
||||
}
|
||||
|
||||
// ── C-1: honest AUTH-FAILURE banner detection (issue #133) ───────────────
|
||||
// When the interactive `claude` CLI hits an in-session error it does NOT crash —
|
||||
// it renders the error as ordinary assistant text in the transcript. The specific
|
||||
// failure C-1 exists to catch is R-1: EXPIRED / INVALID credentials, where every
|
||||
// turn comes back as the same one-line auth-failure banner and OCP, none the wiser,
|
||||
// caches that banner (server.mjs setCachedResponse), shares it via singleflight, and
|
||||
// records a model SUCCESS — so a hard auth error is silently served (and cached for
|
||||
// the 5-min TTL) as a real answer. The two live-reproduced banners on PI231
|
||||
// (2026-06-10) are:
|
||||
// "Please run /login · API Error: 401 Invalid authentication credentials" (69 chars)
|
||||
// "Failed to authenticate. API Error: 401 Invalid authentication credentials" (73 chars)
|
||||
//
|
||||
// WHY THE SCOPE IS NARROW (conservatism — the load-bearing design choice).
|
||||
// An earlier generalised rule (^<short-prefix>?API Error:\s*\d{3}\b.*$) was TOO
|
||||
// BROAD: its unbounded `.*` tail let any short prefix + "API Error: NNN" + an
|
||||
// arbitrarily long sentence match, so it KILLED legitimate long answers that merely
|
||||
// DISCUSS an API error (e.g. "API Error: 500 happened because the server was
|
||||
// overloaded. To fix this, retry with exponential backoff …"). That is the worst
|
||||
// outcome: a false-positive costs the user a missing answer AND a double-burn retry,
|
||||
// whereas the rare false-negative (caching one transient error for the 5-min TTL) is
|
||||
// cheap and self-healing. So C-1 is reframed from "detect ANY API error" to "detect
|
||||
// a claude-CLI AUTHENTICATION-FAILURE banner", and when unsure it PASSES (does not
|
||||
// kill). Transient 5xx server errors are deliberately NOT detected — they are not the
|
||||
// R-1 case and the conservative choice is to let them through.
|
||||
//
|
||||
// THE SIGNAL — a turn is an auth-failure banner only if ALL of these hold over the
|
||||
// WHOLE trimmed assistant text (a conjunction; any one failing => PASS):
|
||||
// 1. SHORT whole-message. Real banners are one short line (the two live samples are
|
||||
// 69 and 73 chars). Cap = TUI_ERR_MAX_LEN (100) — headroom over 73 for a
|
||||
// slightly longer future banner, while still rejecting multi-sentence prose. A
|
||||
// long answer that happens to discuss auth (no code chars, e.g. 226 chars) is
|
||||
// rejected on length alone.
|
||||
// 2. Contains "API Error: 4\d{2}" — auth failures are 4xx (401/403). This rejects
|
||||
// transient 5xx ("API Error: 500/503 …") and bare "HTTP 401 means unauthorized."
|
||||
// (no "API Error:" core).
|
||||
// 3. Contains an auth KEYWORD — authenticat | /login | credential (case-insensitive).
|
||||
// This rejects answers that quote a 4xx but are not auth banners, e.g.
|
||||
// "To debug a 401: the server returns API Error: 401 Unauthorized …"
|
||||
// ("Unauthorized" is authoriz-, not authenticat-; no /login, no credential).
|
||||
// 4. Contains NO backtick or quote char (` ' "). A real CLI banner is plain text;
|
||||
// backticked/quoted text signals an answer that is QUOTING the error rather than
|
||||
// being the banner, e.g. "You'll see `API Error: 401` … run /login to fix it."
|
||||
// (75 chars — passes 1-3 but is excluded here). This is the conservative tie-
|
||||
// breaker for short instructional answers.
|
||||
//
|
||||
// Worked matrix (all required cases pass — see test-features.mjs C-1 block):
|
||||
// KILL: "Please run /login · API Error: 401 Invalid authentication credentials"
|
||||
// KILL: "Failed to authenticate. API Error: 401 Invalid authentication credentials"
|
||||
// PASS: "API Error: 500 happened because the server was overloaded. …" (not 4xx)
|
||||
// PASS: "Failed to parse the config. Here are the API Error: 401 details …" (too long + no auth-kw)
|
||||
// PASS: "To debug a 401: … API Error: 401 Unauthorized, then you refresh …" (no auth-kw)
|
||||
// PASS: "Here is the handler … It logs the string API Error: 503 …" (not 4xx)
|
||||
// PASS: "You'll see `API Error: 401` … run /login to fix it." (has backtick)
|
||||
// PASS: "HTTP 401 means unauthorized." (no API Error core)
|
||||
// PASS: "The capital of France is Paris." (nothing matches)
|
||||
//
|
||||
// OPERATOR OVERRIDE (unchanged): CLAUDE_TUI_ERROR_PATTERNS lets an operator REPLACE
|
||||
// the default auth-banner detector with their own newline- or `||`-separated JS regex
|
||||
// source strings (each auto-anchored ^…$ over the trimmed text, case-insensitive). A
|
||||
// non-empty override uses ONLY those regexes (the narrowed default is bypassed); an
|
||||
// empty / whitespace-only override DISABLES detection entirely (escape hatch).
|
||||
|
||||
// Whole-message length cap for the default auth-banner detector. Real banners are
|
||||
// 69/73 chars; 100 gives headroom while still rejecting multi-sentence prose.
|
||||
const TUI_ERR_MAX_LEN = 100;
|
||||
// 4xx "API Error:" core — auth failures are 4xx (401/403), never 5xx.
|
||||
const TUI_ERR_4XX = /API Error:\s*4\d{2}\b/i;
|
||||
// Auth keyword — the message must be about authentication, not just quote a 4xx.
|
||||
const TUI_ERR_AUTH_KW = /authenticat|\/login|credential/i;
|
||||
// Code/quote chars — their presence signals prose QUOTING an error, not the banner.
|
||||
const TUI_ERR_CODE_CHAR = /[`'"]/;
|
||||
|
||||
// Default detector: returns true iff `trimmed` IS a claude-CLI auth-failure banner
|
||||
// (all four signals above). Conservative — any signal failing => false (PASS).
|
||||
function isDefaultAuthFailureBanner(trimmed) {
|
||||
if (trimmed.length > TUI_ERR_MAX_LEN) return false; // 1. short whole-message
|
||||
if (!TUI_ERR_4XX.test(trimmed)) return false; // 2. 4xx API Error core
|
||||
if (!TUI_ERR_AUTH_KW.test(trimmed)) return false; // 3. auth keyword
|
||||
if (TUI_ERR_CODE_CHAR.test(trimmed)) return false; // 4. no code/quote chars
|
||||
return true;
|
||||
}
|
||||
|
||||
// Compile an OPERATOR-SUPPLIED pattern set (override path only). Each source is
|
||||
// anchored ^…$ over the trimmed text and matched case-insensitively (`s` so `.` spans
|
||||
// a multi-line banner). A pattern that fails to compile is skipped (never throws into
|
||||
// the request path).
|
||||
function compileTuiErrorPatterns(raw) {
|
||||
const sources = String(raw).split(/\r?\n|\|\|/).map((s) => s.trim()).filter(Boolean);
|
||||
const out = [];
|
||||
for (const src of sources) {
|
||||
try { out.push(new RegExp(`^(?:${src})$`, "is")); } catch { /* skip bad pattern */ }
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// Returns the matched banner text (the trimmed assistant text) if `text` IS a claude-
|
||||
// CLI auth-failure banner in its entirety, else null. `patternsRaw` defaults to
|
||||
// process.env.CLAUDE_TUI_ERROR_PATTERNS:
|
||||
// - undefined → narrowed default auth-banner detector (isDefaultAuthFailureBanner).
|
||||
// - non-empty → operator regex override REPLACES the default.
|
||||
// - empty/ws → detection disabled (escape hatch).
|
||||
export function detectTuiUpstreamError(text, patternsRaw = process.env.CLAUDE_TUI_ERROR_PATTERNS) {
|
||||
if (typeof text !== "string") return null;
|
||||
const trimmed = text.trim();
|
||||
if (!trimmed) return null;
|
||||
if (patternsRaw == null) {
|
||||
return isDefaultAuthFailureBanner(trimmed) ? trimmed : null;
|
||||
}
|
||||
// Operator override path: empty/whitespace disables; otherwise use only their regexes.
|
||||
const patterns = compileTuiErrorPatterns(patternsRaw);
|
||||
if (patterns.length === 0) return null;
|
||||
for (const re of patterns) {
|
||||
if (re.test(trimmed)) return trimmed;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Block until the session transcript is terminal (turn_duration / final
|
||||
// stop_reason) or the wall-clock cap elapses, polling the file (no fs.watch —
|
||||
// robust over NFS / editors). Returns { text, entrypoint, truncated }:
|
||||
// - text: latest assistant text.
|
||||
// - entrypoint: billing-pool classifier (see verifyEntrypoint), or null.
|
||||
// - truncated: FALSE when a terminal marker was reached (the turn completed);
|
||||
// TRUE when the wall-clock cap was hit with partial text but NO
|
||||
// terminal marker (the turn is INCOMPLETE — what we have is a
|
||||
// cut-off prefix). (C-2, issue #133.)
|
||||
//
|
||||
// Why `truncated` matters: previously the terminal-marker path and the
|
||||
// cap-with-partial-text path BOTH returned `{text, entrypoint}` identically, so
|
||||
// callClaudeTui could not tell a complete answer from a truncated one and cached +
|
||||
// returned the partial as finish_reason:stop (silent success). The caller now
|
||||
// throws on `truncated` so a cut-off turn is neither cached nor counted as success.
|
||||
// The field is additive — existing call sites that ignore it keep working.
|
||||
//
|
||||
// On cap with NO text at all, still throws (unchanged) — there is nothing to return.
|
||||
//
|
||||
// No quiescence heuristic by design: a long Opus thinking turn stalls transcript
|
||||
// growth and a "file stable for N s" rule would false-abort it (spec §4.3).
|
||||
// Resolution: pass an explicit `transcriptPath` (used by unit tests), OR pass
|
||||
// `home` + `sessionId` to resolve by glob each poll (production) — the transcript
|
||||
// file does not exist until the turn starts, so resolution happens inside the loop.
|
||||
// `abortSignal` (optional): when it fires, stop waiting and throw TuiAbortError. The one
|
||||
// caller that passes it is the STREAMING TUI path, which ties it to the client's socket:
|
||||
// a client that disconnects mid-turn should not leave the pane running (and the caller's
|
||||
// concurrency slot held) until the turn or the 120s cap ends. runTuiTurn's finally does the
|
||||
// teardown. Omitted => the loop is byte-for-byte the pre-streaming loop.
|
||||
export async function readTuiTranscript({ transcriptPath: p, home, sessionId, wallclockMs = 120000, pollMs = 250, abortSignal = null }) {
|
||||
const deadline = Date.now() + wallclockMs;
|
||||
let lastText = "";
|
||||
let lastEntrypoint = null;
|
||||
while (Date.now() < deadline) {
|
||||
if (abortSignal && abortSignal.aborted) {
|
||||
const err = new Error("tui_aborted: client disconnected before the turn completed");
|
||||
err.name = "TuiAbortError";
|
||||
throw err;
|
||||
}
|
||||
const resolved = p || findTranscriptPath(home, sessionId);
|
||||
if (resolved && existsSync(resolved)) {
|
||||
const events = parseTranscriptLines(readFileSync(resolved, "utf8"));
|
||||
lastText = extractLatestAssistantText(events) || lastText;
|
||||
const ep = verifyEntrypoint(events);
|
||||
if (ep != null) lastEntrypoint = ep;
|
||||
// Terminal marker reached → the turn is COMPLETE.
|
||||
if (events.some(isTerminalLine)) return { text: lastText, entrypoint: lastEntrypoint, truncated: false };
|
||||
}
|
||||
await sleep(pollMs);
|
||||
}
|
||||
// Cap elapsed with no terminal marker. If we have partial text, flag it truncated
|
||||
// so the caller rejects it (don't cache / don't count as success). No text at all
|
||||
// → throw (nothing to return).
|
||||
if (lastText) return { text: lastText, entrypoint: lastEntrypoint, truncated: true };
|
||||
throw new Error("tui_transcript_timeout: no assistant text within wallclock cap");
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
# OCP Constitution
|
||||
<!-- Created by spec-kit integration (github/spec-kit v0.7.3) -->
|
||||
<!--
|
||||
NOTE TO CLAUDE SESSIONS: This file is the spec-kit constitution for the OCP repo.
|
||||
It coexists with — and is subordinate to — the project-specific constraints in
|
||||
~/ocp/AGENTS.md. Always read AGENTS.md first for memory policies, protected files,
|
||||
and development discipline rules. This constitution covers spec-driven development
|
||||
principles for new features.
|
||||
-->
|
||||
|
||||
## Core Principles
|
||||
|
||||
### I. Spec-First Development
|
||||
All non-trivial features begin with a specification in `specs/` before any code is written.
|
||||
Use `/speckit-specify` to create the spec, `/speckit-plan` for the implementation plan,
|
||||
`/speckit-tasks` to generate the task list, then `/speckit-implement` to execute.
|
||||
|
||||
### II. Server Integrity (NON-NEGOTIABLE)
|
||||
`server.mjs`, `models.json`, `package.json`, and `keys.mjs` are protected files.
|
||||
No spec-driven workflow may modify them without explicit approval from the project
|
||||
maintainer. If a spec calls for changes to these files, halt and escalate.
|
||||
|
||||
### III. Test-First
|
||||
Implementation tasks include tests before code. The `/speckit-implement` command
|
||||
must follow TDD for any logic touching the proxy or model-routing paths.
|
||||
|
||||
### IV. Additive by Default
|
||||
Features should be additive — new endpoints, new model entries, new config flags —
|
||||
not modifications to existing contracts unless the spec explicitly justifies the
|
||||
breaking change and the plan documents a migration path.
|
||||
|
||||
### V. Cross-Device Compatibility
|
||||
OCP runs on multiple machines. Specs and plans committed to `specs/` serve as the
|
||||
cross-device handoff artifact. Keep `specs/NNN/tasks.md` updated as the canonical
|
||||
work-state file.
|
||||
|
||||
## Constraints
|
||||
|
||||
- Follow CC 开发铁律 v1.3 (Iron Rules) as loaded via `/cc-rules` in CLAUDE.md.
|
||||
- One PR per reviewable unit (Iron Rule 11 IDR).
|
||||
- Independent review required before merge (Iron Rule 10).
|
||||
- Pre-brainstorm prior-art search required (Iron Rule 12).
|
||||
|
||||
## Governance
|
||||
|
||||
This constitution is subordinate to `AGENTS.md` for project-specific memory policy
|
||||
and to CC 开发铁律 for development discipline. Amendments require a PR reviewed by
|
||||
the project maintainer.
|
||||
|
||||
**Version**: 1.0.0 | **Ratified**: 2026-04-21 | **Last Amended**: 2026-04-21
|
||||
+64
@@ -0,0 +1,64 @@
|
||||
{
|
||||
"$schema": "./models.schema.json",
|
||||
"version": 1,
|
||||
"models": [
|
||||
{
|
||||
"id": "claude-opus-4-8",
|
||||
"displayName": "Claude Opus 4.8",
|
||||
"openclawName": "Claude Opus 4.8 (via CLI)",
|
||||
"reasoning": true,
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
{
|
||||
"id": "claude-opus-4-7",
|
||||
"displayName": "Claude Opus 4.7",
|
||||
"openclawName": "Claude Opus 4.7 (via CLI)",
|
||||
"reasoning": true,
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
{
|
||||
"id": "claude-opus-4-6",
|
||||
"displayName": "Claude Opus 4.6",
|
||||
"openclawName": "Claude Opus 4.6 (via CLI)",
|
||||
"reasoning": true,
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
{
|
||||
"id": "claude-sonnet-5",
|
||||
"displayName": "Claude Sonnet 5",
|
||||
"openclawName": "Claude Sonnet 5 (via CLI)",
|
||||
"reasoning": true,
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
{
|
||||
"id": "claude-sonnet-4-6",
|
||||
"displayName": "Claude Sonnet 4.6",
|
||||
"openclawName": "Claude Sonnet 4.6 (via CLI)",
|
||||
"reasoning": true,
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 16384
|
||||
},
|
||||
{
|
||||
"id": "claude-haiku-4-5-20251001",
|
||||
"displayName": "Claude Haiku 4.5",
|
||||
"openclawName": "Claude Haiku 4.5 (via CLI)",
|
||||
"reasoning": false,
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": 8192
|
||||
}
|
||||
],
|
||||
"aliases": {
|
||||
"opus": "claude-opus-4-8",
|
||||
"sonnet": "claude-sonnet-5",
|
||||
"haiku": "claude-haiku-4-5-20251001"
|
||||
},
|
||||
"legacyAliases": {
|
||||
"claude-opus-4": "claude-opus-4-7",
|
||||
"claude-haiku-4": "claude-haiku-4-5-20251001",
|
||||
"claude-haiku-4-5": "claude-haiku-4-5-20251001"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,934 @@
|
||||
#!/usr/bin/env bash
|
||||
# ocp — OpenClaw Proxy CLI
|
||||
# Usage: ocp <command> [args]
|
||||
#
|
||||
# Talks to the local claude-proxy at http://127.0.0.1:3456
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
PROXY="http://127.0.0.1:3456"
|
||||
|
||||
# Auth args for multi-key mode: reads from OCP_ADMIN_KEY env or ~/.ocp/admin-key file
|
||||
# Using a bash array preserves word boundaries — no eval needed.
|
||||
_AUTH_ARGS=()
|
||||
if [[ -n "${OCP_ADMIN_KEY:-}" ]]; then
|
||||
_AUTH_ARGS=(-H "Authorization: Bearer $OCP_ADMIN_KEY")
|
||||
elif [[ -f "$HOME/.ocp/admin-key" ]]; then
|
||||
_AUTH_ARGS=(-H "Authorization: Bearer $(cat "$HOME/.ocp/admin-key")")
|
||||
fi
|
||||
|
||||
# Wrapper: curl with optional auth
|
||||
_curl() {
|
||||
curl "${_AUTH_ARGS[@]}" "$@"
|
||||
}
|
||||
|
||||
_json() { python3 -m json.tool 2>/dev/null || cat; }
|
||||
|
||||
_bar() {
|
||||
local pct=$1 width=20
|
||||
local filled=$(( pct * width / 100 ))
|
||||
local empty=$(( width - filled ))
|
||||
printf '['
|
||||
printf '█%.0s' $(seq 1 $filled 2>/dev/null) || true
|
||||
printf '░%.0s' $(seq 1 $empty 2>/dev/null) || true
|
||||
printf '] %d%%' "$pct"
|
||||
}
|
||||
|
||||
# ── usage ────────────────────────────────────────────────────────────────
|
||||
cmd_usage_help() {
|
||||
cat <<'EOF'
|
||||
ocp usage — Show Claude plan usage limits
|
||||
|
||||
Displays current session utilization, weekly limits, extra usage status,
|
||||
and proxy request statistics. Data is fetched from the Anthropic API
|
||||
via a minimal probe call (cached for 5 minutes).
|
||||
|
||||
Usage: ocp usage [--by-key]
|
||||
|
||||
Options:
|
||||
--by-key Show per-key usage statistics
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_usage() {
|
||||
if [[ "${1:-}" == "--by-key" ]]; then
|
||||
local data
|
||||
data=$(_curl -sf --max-time 15 "$PROXY/api/usage" 2>&1) || { echo "Error: proxy unreachable or usage API not available"; exit 1; }
|
||||
echo "$data" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
by_key = d.get('byKey', [])
|
||||
if not by_key:
|
||||
print('No usage data yet.')
|
||||
else:
|
||||
print('Usage by Key')
|
||||
print('─────────────────────────────────────────────────────────────────')
|
||||
hdr = f' {\"Key\":<20} {\"Reqs\":>5} {\"OK\":>4} {\"Err\":>4} {\"Avg Time\":>9} {\"Last Request\":<20}'
|
||||
print(hdr)
|
||||
print(' ' + '─' * (len(hdr) - 2))
|
||||
for k in by_key:
|
||||
avg_t = f'{k[\"avg_elapsed_ms\"]/1000:.1f}s' if k['avg_elapsed_ms'] else '-'
|
||||
print(f' {k[\"key_name\"]:<20} {k[\"requests\"]:>5} {k[\"successes\"]:>4} {k[\"errors\"]:>4} {avg_t:>9} {k[\"last_request\"]:<20}')
|
||||
"
|
||||
return
|
||||
fi
|
||||
|
||||
local data
|
||||
data=$(curl -sf --max-time 15 "$PROXY/usage" 2>&1) || { echo "Error: proxy unreachable"; exit 1; }
|
||||
|
||||
echo "$data" | python3 -c "
|
||||
import sys, json
|
||||
|
||||
d = json.loads(sys.stdin.read())
|
||||
p = d['plan']
|
||||
s = p['currentSession']
|
||||
w = p['weeklyLimits']['allModels']
|
||||
e = p['extraUsage']
|
||||
px = d['proxy']
|
||||
models = d.get('models', {})
|
||||
|
||||
print('Plan Usage Limits')
|
||||
print('─────────────────────────────────────')
|
||||
print(f' Current session {s[\"percent\"]:>4} used')
|
||||
print(f' Resets in {s[\"resetsIn\"]} ({s[\"resetsAtHuman\"]})')
|
||||
print()
|
||||
print(f' Weekly (all models) {w[\"percent\"]:>4} used')
|
||||
print(f' Resets in {w[\"resetsIn\"]} ({w[\"resetsAtHuman\"]})')
|
||||
print()
|
||||
status_icon = 'on' if e['status'] == 'allowed' else 'off'
|
||||
print(f' Extra usage {status_icon}')
|
||||
print()
|
||||
|
||||
if models:
|
||||
print('Model Stats (since proxy start)')
|
||||
print('─────────────────────────────────────────────────────────────────────')
|
||||
hdr = f' {\"Model\":<25} {\"Reqs\":>5} {\"OK\":>4} {\"Err\":>4} {\"Avg Time\":>9} {\"Max Time\":>9} {\"Avg Prompt\":>11} {\"Max Prompt\":>11}'
|
||||
print(hdr)
|
||||
print(' ' + '─' * (len(hdr) - 2))
|
||||
total_reqs = 0
|
||||
for model in sorted(models):
|
||||
m = models[model]
|
||||
total_reqs += m['requests']
|
||||
avg_t = f'{m[\"avgElapsed\"]/1000:.0f}s' if m['avgElapsed'] else '-'
|
||||
max_t = f'{m[\"maxElapsed\"]/1000:.0f}s' if m['maxElapsed'] else '-'
|
||||
avg_p = f'{m[\"avgPromptChars\"]/1000:.0f}K' if m['avgPromptChars'] else '-'
|
||||
max_p = f'{m[\"maxPromptChars\"]/1000:.0f}K' if m['maxPromptChars'] else '-'
|
||||
print(f' {model:<25} {m[\"requests\"]:>5} {m[\"successes\"]:>4} {m[\"errors\"]:>4} {avg_t:>9} {max_t:>9} {avg_p:>11} {max_p:>11}')
|
||||
print(f' {\"Total\":<25} {total_reqs:>5}')
|
||||
else:
|
||||
print(' No model requests yet.')
|
||||
|
||||
print()
|
||||
print(f'Proxy: up {px[\"uptime\"]} | {px[\"totalRequests\"]} reqs | {px[\"activeRequests\"]} active | {px[\"errors\"]} err | {px[\"timeouts\"]} timeout')
|
||||
"
|
||||
}
|
||||
|
||||
# ── status ───────────────────────────────────────────────────────────────
|
||||
cmd_status_help() {
|
||||
cat <<'EOF'
|
||||
ocp status — Quick combined overview (usage + health)
|
||||
|
||||
Usage: ocp status
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_status() {
|
||||
curl -sf --max-time 15 "$PROXY/status" | _json
|
||||
}
|
||||
|
||||
# ── health ───────────────────────────────────────────────────────────────
|
||||
cmd_health_help() {
|
||||
cat <<'EOF'
|
||||
ocp health — Proxy health and diagnostics
|
||||
|
||||
Shows proxy status, version, uptime, auth, config, sessions, and recent errors.
|
||||
|
||||
Usage: ocp health
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_health() {
|
||||
curl -sf --max-time 10 "$PROXY/health" | _json
|
||||
}
|
||||
|
||||
# ── logs ─────────────────────────────────────────────────────────────────
|
||||
cmd_logs_help() {
|
||||
cat <<'EOF'
|
||||
ocp logs — Show recent proxy log entries
|
||||
|
||||
Usage: ocp logs [N] [LEVEL]
|
||||
|
||||
Arguments:
|
||||
N Number of entries to show (default: 20, max: 200)
|
||||
LEVEL Filter level: error, warn, info, all (default: error)
|
||||
|
||||
Examples:
|
||||
ocp logs Last 20 errors
|
||||
ocp logs 50 all Last 50 entries of any level
|
||||
ocp logs 10 warn Last 10 warnings
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_logs() {
|
||||
local n=${1:-20}
|
||||
local level=${2:-error}
|
||||
curl -sf --max-time 10 "$PROXY/logs?n=$n&level=$level" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
entries = d.get('entries', [])
|
||||
if not entries:
|
||||
print(f'No {d.get(\"level\",\"\")} log entries.')
|
||||
else:
|
||||
for e in entries:
|
||||
if 'raw' in e:
|
||||
print(e['raw'][:200])
|
||||
else:
|
||||
ts = e.get('ts','')[:19]
|
||||
ev = e.get('event','?')
|
||||
lvl = e.get('level','')
|
||||
model = e.get('model','')
|
||||
extra = ' '.join(f'{k}={v}' for k,v in e.items() if k not in ('ts','event','level','model'))
|
||||
parts = [ts, lvl.upper(), ev]
|
||||
if model: parts.append(model)
|
||||
if extra: parts.append(extra)
|
||||
print(' | '.join(parts))
|
||||
"
|
||||
}
|
||||
|
||||
# ── models ───────────────────────────────────────────────────────────────
|
||||
cmd_models_help() {
|
||||
cat <<'EOF'
|
||||
ocp models — List available Claude models
|
||||
|
||||
Usage: ocp models
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_models() {
|
||||
curl -sf --max-time 5 "$PROXY/v1/models" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
for m in d.get('data', []):
|
||||
print(f\" {m['id']}\")
|
||||
"
|
||||
}
|
||||
|
||||
# ── sessions ─────────────────────────────────────────────────────────────
|
||||
cmd_sessions_help() {
|
||||
cat <<'EOF'
|
||||
ocp sessions — List active CLI sessions
|
||||
|
||||
Usage: ocp sessions
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_sessions() {
|
||||
curl -sf --max-time 5 "$PROXY/sessions" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
sessions = d.get('sessions', [])
|
||||
if not sessions:
|
||||
print('No active sessions.')
|
||||
else:
|
||||
for s in sessions:
|
||||
print(f\" {s['id'][:16]}... model={s['model']} msgs={s['messages']} last={s['lastUsed']}\")
|
||||
"
|
||||
}
|
||||
|
||||
# ── clear ────────────────────────────────────────────────────────────────
|
||||
cmd_clear_help() {
|
||||
cat <<'EOF'
|
||||
ocp clear — Clear all active CLI sessions
|
||||
|
||||
Usage: ocp clear
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_clear() {
|
||||
local result
|
||||
result=$(curl -sf --max-time 5 -X DELETE "$PROXY/sessions")
|
||||
local count
|
||||
count=$(echo "$result" | python3 -c "import sys,json; print(json.loads(sys.stdin.read())['cleared'])")
|
||||
echo "Cleared $count sessions."
|
||||
}
|
||||
|
||||
# ── keys ────────────────────────────────────────────────────────────────
|
||||
cmd_keys_help() {
|
||||
cat <<'EOF'
|
||||
ocp keys — Manage API keys (multi-key auth mode)
|
||||
|
||||
Usage:
|
||||
ocp keys List all keys
|
||||
ocp keys add <name> Create a new key
|
||||
ocp keys revoke <name|id> Revoke a key
|
||||
|
||||
Examples:
|
||||
ocp keys add wife-laptop
|
||||
ocp keys add son-ipad
|
||||
ocp keys revoke wife-laptop
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_keys() {
|
||||
case "${1:-}" in
|
||||
add)
|
||||
if [[ -z "${2:-}" ]]; then echo "Usage: ocp keys add <name>"; return 1; fi
|
||||
local result
|
||||
result=$(_curl -sf --max-time 5 -X POST "$PROXY/api/keys" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "{\"name\": \"$2\"}" 2>&1) || { echo "Error: proxy unreachable or unauthorized"; exit 1; }
|
||||
echo "$result" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
if 'key' in d:
|
||||
print(f'✓ Key created for \"{d[\"name\"]}\"')
|
||||
print(f'')
|
||||
print(f' API Key: {d[\"key\"]}')
|
||||
print(f'')
|
||||
print(f' Copy this key now — you won\\'t see it again.')
|
||||
print(f' Configure in IDE: OPENAI_API_KEY={d[\"key\"]}')
|
||||
else:
|
||||
print(f'✗ {d.get(\"error\", \"Unknown error\")}')
|
||||
"
|
||||
;;
|
||||
revoke)
|
||||
if [[ -z "${2:-}" ]]; then echo "Usage: ocp keys revoke <name|id>"; return 1; fi
|
||||
_curl -sf --max-time 5 -X DELETE "$PROXY/api/keys/$2" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
if d.get('revoked'):
|
||||
print(f'✓ Key \"{d[\"idOrName\"]}\" revoked.')
|
||||
else:
|
||||
print(f'✗ Key not found or already revoked.')
|
||||
"
|
||||
;;
|
||||
--help|-h)
|
||||
cmd_keys_help
|
||||
;;
|
||||
"")
|
||||
_curl -sf --max-time 5 "$PROXY/api/keys" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
keys = d.get('keys', [])
|
||||
if not keys:
|
||||
print('No API keys configured.')
|
||||
print('Create one: ocp keys add <name>')
|
||||
else:
|
||||
print('API Keys')
|
||||
print('─────────────────────────────────────────────────')
|
||||
for k in keys:
|
||||
status = '✗ revoked' if k['revoked'] else '✓ active'
|
||||
print(f' {k[\"name\"]:<20} {k[\"keyPreview\"]:<20} {status} {k[\"created_at\"]}')
|
||||
" 2>/dev/null || { echo "Error: proxy unreachable or key management not available"; exit 1; }
|
||||
;;
|
||||
*)
|
||||
echo "Unknown subcommand: $1"; cmd_keys_help; return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# ── connect ─────────────────────────────────────────────────────────────
|
||||
cmd_connect_help() {
|
||||
cat <<'EOF'
|
||||
ocp connect — Connect this machine to a remote OCP as a LAN client
|
||||
|
||||
Configures OPENAI_BASE_URL (and optionally OPENAI_API_KEY) in your shell
|
||||
rc file so tools like Claude Code point to a remote OCP instance.
|
||||
|
||||
Usage:
|
||||
ocp connect <host-ip> [--port PORT] [--key API_KEY]
|
||||
|
||||
Arguments:
|
||||
host-ip IP address of the machine running OCP
|
||||
--port PORT Port OCP listens on (default: 3456)
|
||||
--key API_KEY API key to use (prompted if remote requires auth)
|
||||
|
||||
Examples:
|
||||
ocp connect 192.168.1.10
|
||||
ocp connect 192.168.1.10 --port 8080
|
||||
ocp connect 192.168.1.10 --key sk-abc123
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_connect() {
|
||||
local host="" port=3456 key=""
|
||||
|
||||
# Parse args
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--port) port="${2:?'--port requires a value'}"; shift 2 ;;
|
||||
--key) key="${2:?'--key requires a value'}"; shift 2 ;;
|
||||
--help|-h) cmd_connect_help; return 0 ;;
|
||||
-*) echo "Unknown option: $1"; cmd_connect_help; return 1 ;;
|
||||
*) host="$1"; shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ -z "$host" ]]; then
|
||||
echo "Error: host IP is required."
|
||||
echo ""
|
||||
cmd_connect_help
|
||||
return 1
|
||||
fi
|
||||
|
||||
if ! [[ "$host" =~ ^[a-zA-Z0-9._-]+$ ]]; then
|
||||
echo "Error: invalid host '$host'"
|
||||
return 1
|
||||
fi
|
||||
|
||||
local base_url="http://$host:$port"
|
||||
|
||||
echo "OCP Connect"
|
||||
echo "─────────────────────────────────────"
|
||||
echo " Remote: $base_url"
|
||||
echo ""
|
||||
|
||||
# Step 1: Test connectivity via /health
|
||||
echo " Checking connectivity..."
|
||||
local health_json
|
||||
health_json=$(curl -sf --max-time 10 "$base_url/health" 2>/dev/null) || {
|
||||
echo " ✗ Cannot reach $base_url/health"
|
||||
echo " Make sure OCP is running on $host and bound to 0.0.0.0 (LAN mode)."
|
||||
return 1
|
||||
}
|
||||
echo " ✓ Connected"
|
||||
echo ""
|
||||
|
||||
# Step 2: Show remote info
|
||||
local remote_version auth_mode
|
||||
remote_version=$(echo "$health_json" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); print(d.get('version','?'))" 2>/dev/null || echo "?")
|
||||
auth_mode=$(echo "$health_json" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); print(d.get('authMode','none'))" 2>/dev/null || echo "none")
|
||||
|
||||
echo " Remote OCP v$remote_version (auth: $auth_mode)"
|
||||
echo ""
|
||||
|
||||
# Step 3: Determine if key is needed
|
||||
local needs_key=0
|
||||
if [[ "$auth_mode" != "none" ]]; then
|
||||
needs_key=1
|
||||
fi
|
||||
|
||||
if [[ $needs_key -eq 1 && -z "$key" ]]; then
|
||||
echo " Remote requires authentication."
|
||||
echo " Ask the admin to run: ocp keys add <name>"
|
||||
printf " Enter API key (or press Enter to skip): "
|
||||
read -rs key </dev/tty
|
||||
echo
|
||||
if [[ -z "$key" ]]; then
|
||||
echo ""
|
||||
echo " ✗ No key provided — cannot connect to an auth-required remote without a key."
|
||||
return 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# Step 4: Test API access via /v1/models
|
||||
echo " Testing API access..."
|
||||
local models_out models_ok=0
|
||||
if [[ -n "$key" ]]; then
|
||||
models_out=$(curl -sf --max-time 10 \
|
||||
-H "Authorization: Bearer $key" \
|
||||
"$base_url/v1/models" 2>/dev/null) && models_ok=1
|
||||
else
|
||||
models_out=$(curl -sf --max-time 10 "$base_url/v1/models" 2>/dev/null) && models_ok=1
|
||||
fi
|
||||
|
||||
if [[ $models_ok -eq 0 ]]; then
|
||||
echo " ✗ API access failed — key may be invalid or revoked."
|
||||
return 1
|
||||
fi
|
||||
local model_count
|
||||
model_count=$(echo "$models_out" | python3 -c "import sys,json; print(len(json.loads(sys.stdin.read()).get('data',[])))" 2>/dev/null || echo "?")
|
||||
echo " ✓ API accessible ($model_count models available)"
|
||||
echo ""
|
||||
|
||||
# Step 5: Detect shell rc file
|
||||
local rc_file
|
||||
if [[ "${SHELL:-}" == */zsh ]]; then
|
||||
rc_file="$HOME/.zshrc"
|
||||
else
|
||||
rc_file="$HOME/.bashrc"
|
||||
fi
|
||||
|
||||
# Step 6: Remove any previously written OCP LAN lines
|
||||
if [[ -f "$rc_file" ]]; then
|
||||
local tmp_rc
|
||||
tmp_rc=$(mktemp)
|
||||
if python3 - "$rc_file" "$tmp_rc" <<'PYEOF'
|
||||
import sys
|
||||
src, dst = sys.argv[1], sys.argv[2]
|
||||
lines = open(src).readlines()
|
||||
out = []
|
||||
skip = False
|
||||
for line in lines:
|
||||
s = line.rstrip('\n')
|
||||
if s == '# OCP LAN (added by ocp connect)':
|
||||
skip = True
|
||||
continue
|
||||
if skip and (s == '' or s.startswith('export OPENAI_BASE_URL=') or s.startswith('export OPENAI_API_KEY=')):
|
||||
continue
|
||||
skip = False
|
||||
out.append(line)
|
||||
open(dst, 'w').writelines(out)
|
||||
PYEOF
|
||||
then
|
||||
cp "$tmp_rc" "$rc_file"
|
||||
fi
|
||||
rm -f "$tmp_rc"
|
||||
fi
|
||||
|
||||
# Step 7: Append new config
|
||||
{
|
||||
echo ""
|
||||
echo "# OCP LAN (added by ocp connect)"
|
||||
echo "export OPENAI_BASE_URL=$base_url/v1"
|
||||
if [[ -n "$key" ]]; then
|
||||
echo "export OPENAI_API_KEY=$key"
|
||||
fi
|
||||
} >> "$rc_file"
|
||||
|
||||
echo " Written to $rc_file:"
|
||||
echo " OPENAI_BASE_URL=$base_url/v1"
|
||||
if [[ -n "$key" ]]; then
|
||||
echo " OPENAI_API_KEY=${key:0:8}..."
|
||||
fi
|
||||
echo ""
|
||||
|
||||
# Step 8: Quick smoke test — send a minimal chat completion
|
||||
echo " Running smoke test..."
|
||||
local chat_payload='{"model":"claude-haiku-4-5-20251001","messages":[{"role":"user","content":"Reply with OK only."}],"max_tokens":10}'
|
||||
local chat_out chat_ok=0
|
||||
if [[ -n "$key" ]]; then
|
||||
chat_out=$(curl -sf --max-time 30 \
|
||||
-H "Authorization: Bearer $key" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "$chat_payload" \
|
||||
"$base_url/v1/chat/completions" 2>/dev/null) && chat_ok=1
|
||||
else
|
||||
chat_out=$(curl -sf --max-time 30 \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "$chat_payload" \
|
||||
"$base_url/v1/chat/completions" 2>/dev/null) && chat_ok=1
|
||||
fi
|
||||
|
||||
if [[ $chat_ok -eq 1 ]]; then
|
||||
local reply
|
||||
reply=$(echo "$chat_out" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); print(d['choices'][0]['message']['content'].strip())" 2>/dev/null || echo "(response received)")
|
||||
echo " ✓ Smoke test passed: $reply"
|
||||
else
|
||||
echo " ⚠ Smoke test failed (proxy is reachable but chat completion did not succeed)."
|
||||
echo " The env vars have still been written. Check the remote OCP logs."
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo " Done. Reload your shell to apply:"
|
||||
echo " source $rc_file"
|
||||
}
|
||||
|
||||
# ── lan ─────────────────────────────────────────────────────────────────
|
||||
cmd_lan_help() {
|
||||
cat <<'EOF'
|
||||
ocp lan — Quick LAN mode setup guide
|
||||
|
||||
Shows current network configuration and connection instructions
|
||||
for other devices on the same network.
|
||||
|
||||
Usage: ocp lan
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_lan() {
|
||||
local ip
|
||||
ip=$(ipconfig getifaddr en0 2>/dev/null || ipconfig getifaddr en1 2>/dev/null || hostname -I 2>/dev/null | awk '{print $1}' || echo "unknown")
|
||||
local port=3456
|
||||
|
||||
echo "OCP LAN Setup"
|
||||
echo "─────────────────────────────────────"
|
||||
echo ""
|
||||
echo " Your IP: $ip"
|
||||
echo " Port: $port"
|
||||
echo ""
|
||||
echo " For IDE users, set:"
|
||||
echo " OPENAI_BASE_URL=http://$ip:$port/v1"
|
||||
echo " OPENAI_API_KEY=<your-key> (if auth enabled)"
|
||||
echo ""
|
||||
echo " Dashboard: http://$ip:$port/dashboard"
|
||||
echo ""
|
||||
|
||||
if curl -sf --max-time 2 "http://$ip:$port/health" > /dev/null 2>&1; then
|
||||
echo " Status: ✓ LAN-accessible"
|
||||
else
|
||||
echo " Status: ✗ Not LAN-accessible (bound to localhost only)"
|
||||
echo ""
|
||||
echo " To enable LAN mode, set env var and restart:"
|
||||
echo " CLAUDE_BIND=0.0.0.0"
|
||||
echo " ocp restart"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── restart ──────────────────────────────────────────────────────────────
|
||||
cmd_restart_help() {
|
||||
cat <<'EOF'
|
||||
ocp restart — Restart the proxy or gateway
|
||||
|
||||
Usage:
|
||||
ocp restart Restart the Claude proxy service
|
||||
ocp restart gateway Restart the OpenClaw gateway
|
||||
(briefly disconnects all Telegram/Discord bots)
|
||||
|
||||
Note (macOS): restart does a full launchctl bootout + bootstrap, NOT
|
||||
`kickstart -k`. bootout+bootstrap re-reads the plist's EnvironmentVariables,
|
||||
so an env change you made (e.g. CLAUDE_BIND, CLAUDE_CODE_OAUTH_TOKEN) actually
|
||||
takes effect. `kickstart -k` only re-execs the process and reuses launchd's
|
||||
cached env, so env edits would be silently ignored. (Linux systemctl already
|
||||
re-reads its EnvironmentFile on restart.)
|
||||
EOF
|
||||
}
|
||||
|
||||
# macOS only: reload a launchd agent via bootout + bootstrap so plist
|
||||
# EnvironmentVariables are re-read (kickstart -k would reuse the cached env).
|
||||
# Args: <uid> <label> <plist-path>. Returns 0 iff bootstrap succeeds.
|
||||
_launchd_reload() {
|
||||
local uid="$1" label="$2" plist="$3"
|
||||
[[ -f "$plist" ]] || return 1
|
||||
# bootout may legitimately fail if the agent is not currently loaded — that's fine,
|
||||
# we only require the subsequent bootstrap to succeed (the load that re-reads env).
|
||||
launchctl bootout "gui/$uid/$label" 2>/dev/null || true
|
||||
launchctl bootstrap "gui/$uid" "$plist" 2>/dev/null
|
||||
}
|
||||
|
||||
cmd_restart() {
|
||||
if [[ "${1:-}" == "gateway" ]]; then
|
||||
echo "Restarting gateway..."
|
||||
openclaw gateway restart 2>&1
|
||||
else
|
||||
echo "Restarting proxy..."
|
||||
# Try current service name, then legacy, then manual restart.
|
||||
# macOS: bootout+bootstrap (re-reads plist EnvironmentVariables — see cmd_restart_help).
|
||||
# Linux: systemctl --user restart already re-reads its EnvironmentFile.
|
||||
local uid
|
||||
uid=$(id -u)
|
||||
if _launchd_reload "$uid" "dev.ocp.proxy" "$HOME/Library/LaunchAgents/dev.ocp.proxy.plist"; then
|
||||
true
|
||||
elif _launchd_reload "$uid" "ai.openclaw.proxy" "$HOME/Library/LaunchAgents/ai.openclaw.proxy.plist"; then
|
||||
true
|
||||
elif systemctl --user restart ocp-proxy 2>/dev/null; then
|
||||
true
|
||||
elif systemctl --user restart openclaw-proxy 2>/dev/null; then
|
||||
true
|
||||
else
|
||||
echo "Service restart failed, trying kill + restart..."
|
||||
pkill -f "server.mjs.*CLAUDE_PROXY" 2>/dev/null || pkill -f "claude-proxy/server.mjs" 2>/dev/null || true
|
||||
sleep 1
|
||||
local self_r script_dir
|
||||
self_r="${BASH_SOURCE[0]}"
|
||||
while [[ -L "$self_r" ]]; do self_r="$(readlink "$self_r")"; done
|
||||
script_dir="$(cd "$(dirname "$self_r")" && pwd)"
|
||||
# env -u strips test-only key-store redirection vars (A4): if the invoking shell had
|
||||
# NODE_ENV=test + OCP_DIR_OVERRIDE exported (e.g. from a debugging session), this manual
|
||||
# fallback would otherwise inherit them and start the daemon against a scratch/empty key
|
||||
# store — a silent auth outage in AUTH_MODE=multi. The plist/systemd paths strip these via
|
||||
# plist-merge's NEVER_PRESERVE; this covers the one direct-launch path OCP controls.
|
||||
DISABLE_AUTOUPDATER=1 env -u NODE_ENV -u OCP_DIR_OVERRIDE nohup node "$script_dir/server.mjs" >> "$HOME/.ocp/logs/proxy.log" 2>&1 &
|
||||
fi
|
||||
sleep 3
|
||||
if curl -sf --max-time 5 "$PROXY/health" > /dev/null 2>&1; then
|
||||
echo "✓ Proxy restarted successfully."
|
||||
cmd_usage
|
||||
else
|
||||
echo "✗ Proxy not responding after restart."
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
# ── settings ─────────────────────────────────────────────────────────────
|
||||
cmd_settings_help() {
|
||||
cat <<'EOF'
|
||||
ocp settings — View or update proxy settings at runtime
|
||||
|
||||
Usage:
|
||||
ocp settings Show all tunable settings
|
||||
ocp settings <key> <value> Update a setting (no restart needed)
|
||||
|
||||
Tunable keys:
|
||||
timeout Overall request timeout (ms) [30000 - 600000]
|
||||
firstByteTimeout Base first-byte timeout (ms) [15000 - 300000]
|
||||
maxConcurrent Max concurrent claude processes [1 - 32]
|
||||
sessionTTL Session idle expiry (ms) [60000 - 86400000]
|
||||
maxPromptChars Prompt truncation limit (chars) [10000 - 1000000]
|
||||
tiers.opus.base Opus first-byte timeout base (ms) [30000 - 600000]
|
||||
tiers.opus.perChar Opus per-char timeout (ms/char) [0 - 0.01]
|
||||
tiers.sonnet.base Sonnet first-byte timeout base (ms) [30000 - 600000]
|
||||
tiers.sonnet.perChar Sonnet per-char timeout (ms/char) [0 - 0.01]
|
||||
tiers.haiku.base Haiku first-byte timeout base (ms) [15000 - 300000]
|
||||
tiers.haiku.perChar Haiku per-char timeout (ms/char) [0 - 0.01]
|
||||
|
||||
Examples:
|
||||
ocp settings maxPromptChars 200000
|
||||
ocp settings tiers.sonnet.base 150000
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_settings() {
|
||||
if [[ -z "${1:-}" ]]; then
|
||||
# GET — show all settings
|
||||
curl -sf --max-time 5 "$PROXY/settings" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
print('OCP Settings')
|
||||
print('─────────────────────────────────────')
|
||||
for k in ('timeout','firstByteTimeout','maxConcurrent','sessionTTL','maxPromptChars'):
|
||||
v = d.get(k, {})
|
||||
val = v.get('value', '?')
|
||||
unit = v.get('unit', '')
|
||||
desc = v.get('desc', '')
|
||||
print(f' {k:<20} {val:>8} {unit:<6} {desc}')
|
||||
t = d.get('tiers', {})
|
||||
print()
|
||||
print('Timeout Tiers (first-byte):')
|
||||
for tier in ('opus','sonnet','haiku'):
|
||||
info = t.get(tier, {})
|
||||
print(f' {tier:<8} base={info.get(\"base\",\"?\"):>6}ms perChar={info.get(\"perPromptChar\",\"?\")}')
|
||||
"
|
||||
elif [[ "${1:-}" == "--help" || "${1:-}" == "-h" ]]; then
|
||||
cmd_settings_help
|
||||
elif [[ -z "${2:-}" ]]; then
|
||||
echo "Usage: ocp settings <key> <value>"
|
||||
echo "Run 'ocp settings --help' for available keys."
|
||||
return 1
|
||||
else
|
||||
# PATCH — update a setting
|
||||
local key="$1"
|
||||
local value="$2"
|
||||
local result
|
||||
result=$(curl -s --max-time 5 -X PATCH "$PROXY/settings" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "{\"$key\": $value}" 2>&1)
|
||||
if echo "$result" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); errs=d.get('errors',[]); exit(1 if errs else 0)" 2>/dev/null; then
|
||||
echo "✓ $key = $value"
|
||||
else
|
||||
echo "$result" | python3 -c "
|
||||
import sys, json
|
||||
d = json.loads(sys.stdin.read())
|
||||
for e in d.get('errors', []):
|
||||
print(f'✗ {e}')
|
||||
" 2>/dev/null || echo "✗ $result"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
# ── update ──────────────────────────────────────────────────────────────
|
||||
cmd_update_help() {
|
||||
cat <<'EOF'
|
||||
ocp update — Smart upgrade dispatcher
|
||||
|
||||
Runs `ocp doctor` internally to choose the right path:
|
||||
• Patch bump (same minor): light path (git pull + npm install + restart)
|
||||
• Cross-minor (e.g. v3.10→v3.14): full path with snapshot + post-flight
|
||||
• Old version (< v3.4.0): fresh-install (asks first; AI passes --yes)
|
||||
|
||||
Usage:
|
||||
ocp update Smart auto-pick path
|
||||
ocp update --check Show available updates, don't apply
|
||||
ocp update --dry-run Preview the plan, don't mutate
|
||||
ocp update --target v3.13.0 Pin a specific version
|
||||
ocp update --yes Skip y/N prompts (AI agents pass this)
|
||||
ocp update --rollback Restore the most recent upgrade snapshot
|
||||
ocp update --rollback --list List available snapshots
|
||||
ocp update --rollback <path> Restore a specific snapshot
|
||||
ocp update --rollback --dry-run Preview rollback plan
|
||||
ocp update --rollback --gc Delete old snapshots (keep last 5, or <30 days)
|
||||
ocp update --rollback --gc --dry-run Preview what would be deleted
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_update() {
|
||||
local script_dir self
|
||||
self="${BASH_SOURCE[0]}"
|
||||
while [[ -L "$self" ]]; do self="$(readlink "$self")"; done
|
||||
script_dir="$(cd "$(dirname "$self")" && pwd)"
|
||||
|
||||
# Pass through --check fast path (existing behaviour)
|
||||
if [[ "${1:-}" == "--check" ]]; then
|
||||
cd "$script_dir"
|
||||
git fetch origin main --quiet 2>/dev/null || true
|
||||
local local_ver remote_ver
|
||||
local_ver=$(python3 -c "import json; print(json.load(open('package.json'))['version'])" 2>/dev/null || echo "?")
|
||||
remote_ver=$(git show origin/main:package.json 2>/dev/null | python3 -c "import sys,json; print(json.load(sys.stdin)['version'])" 2>/dev/null || echo "?")
|
||||
local behind
|
||||
behind=$(git rev-list HEAD..origin/main --count 2>/dev/null || echo "?")
|
||||
echo "OCP Update Check"
|
||||
echo "─────────────────────────────────────"
|
||||
echo " Current: v$local_ver"
|
||||
echo " Latest: v$remote_ver"
|
||||
if [[ "$behind" == "0" ]]; then
|
||||
echo " Status: ✓ Up to date"
|
||||
else
|
||||
echo " Status: $behind commit(s) behind"
|
||||
echo " Run 'ocp update' to apply."
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Rollback path
|
||||
if [[ "${1:-}" == "--rollback" ]]; then
|
||||
shift
|
||||
exec node "$script_dir/scripts/upgrade.mjs" --rollback "$@"
|
||||
fi
|
||||
|
||||
# Doctor-driven path selection
|
||||
local kind
|
||||
kind=$(node "$script_dir/scripts/doctor.mjs" --json 2>/dev/null | python3 -c "import sys,json; print(json.load(sys.stdin)['next_action']['kind'])" 2>/dev/null || echo "unknown")
|
||||
|
||||
case "$kind" in
|
||||
noop)
|
||||
echo "Already at latest. Nothing to do."
|
||||
return 0
|
||||
;;
|
||||
update)
|
||||
_cmd_update_light "$script_dir"
|
||||
;;
|
||||
upgrade|fresh_install)
|
||||
exec node "$script_dir/scripts/upgrade.mjs" "$@"
|
||||
;;
|
||||
fix_oauth|fix_service)
|
||||
echo "Pre-upgrade check failed: $kind"
|
||||
echo "Run \`ocp doctor\` for details and ai_executable steps."
|
||||
return 1
|
||||
;;
|
||||
*)
|
||||
echo "Unknown doctor kind: $kind. Run \`ocp doctor --json\` to inspect."
|
||||
return 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Existing light-path body extracted into a helper so cmd_update can call it conditionally.
|
||||
_cmd_update_light() {
|
||||
local script_dir="$1"
|
||||
echo "Updating OCP (light path)..."
|
||||
cd "$script_dir"
|
||||
local old_ver new_ver
|
||||
old_ver=$(python3 -c "import json; print(json.load(open('package.json'))['version'])" 2>/dev/null || echo "?")
|
||||
echo " Pulling latest from GitHub..."
|
||||
if ! git pull origin main --ff-only 2>&1 | sed 's/^/ /'; then
|
||||
echo " ✗ Git pull failed. Resolve conflicts manually in: $script_dir"
|
||||
return 1
|
||||
fi
|
||||
new_ver=$(python3 -c "import json; print(json.load(open('package.json'))['version'])" 2>/dev/null || echo "?")
|
||||
if [[ "$old_ver" == "$new_ver" ]]; then
|
||||
echo " ✓ Already at latest (v$new_ver)"
|
||||
else
|
||||
echo " ✓ Updated v$old_ver → v$new_ver"
|
||||
fi
|
||||
|
||||
# Sync plugin (existing logic preserved)
|
||||
local ext_dir="$HOME/.openclaw/extensions/ocp"
|
||||
if [[ -d "$ext_dir" && -d "$script_dir/ocp-plugin" ]]; then
|
||||
echo " Syncing OCP plugin..."
|
||||
cp "$script_dir/ocp-plugin/index.js" "$ext_dir/index.js" 2>/dev/null
|
||||
cp "$script_dir/ocp-plugin/package.json" "$ext_dir/package.json" 2>/dev/null
|
||||
cp "$script_dir/ocp-plugin/openclaw.plugin.json" "$ext_dir/openclaw.plugin.json" 2>/dev/null
|
||||
echo " ✓ Plugin synced"
|
||||
fi
|
||||
|
||||
if command -v node >/dev/null 2>&1 && [[ -f "$script_dir/scripts/sync-openclaw.mjs" ]]; then
|
||||
echo " Syncing OpenClaw registry..."
|
||||
node "$script_dir/scripts/sync-openclaw.mjs" 2>&1 | sed 's/^/ /' || echo " ⚠ OpenClaw sync failed (non-fatal)"
|
||||
fi
|
||||
|
||||
echo " Restarting proxy..."
|
||||
cmd_restart > /dev/null 2>&1
|
||||
}
|
||||
|
||||
cmd_doctor_help() {
|
||||
cat <<'EOF'
|
||||
ocp doctor — Health & upgrade-readiness check
|
||||
|
||||
Runs a series of checks (Node version, git state, service health,
|
||||
OAuth token, plist customisation, OpenClaw provider) and emits either
|
||||
human-readable PASS/WARN/FAIL output or a JSON next_action that
|
||||
AI agents can execute.
|
||||
|
||||
Usage:
|
||||
ocp doctor Human-readable output
|
||||
ocp doctor --json JSON for AI agents and ocp update internal use
|
||||
ocp doctor --check oauth Fast path: OAuth check only
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_doctor() {
|
||||
local script_dir self
|
||||
self="${BASH_SOURCE[0]}"
|
||||
while [[ -L "$self" ]]; do self="$(readlink "$self")"; done
|
||||
script_dir="$(cd "$(dirname "$self")" && pwd)"
|
||||
exec node "$script_dir/scripts/doctor.mjs" "$@"
|
||||
}
|
||||
|
||||
# ── help ─────────────────────────────────────────────────────────────────
|
||||
cmd_help() {
|
||||
cat <<'EOF'
|
||||
ocp — OpenClaw Proxy CLI
|
||||
|
||||
Usage: ocp <command> [args]
|
||||
|
||||
Commands:
|
||||
usage Plan usage limits (--by-key for per-key stats)
|
||||
status Quick overview (usage + health)
|
||||
health Proxy diagnostics
|
||||
settings View or update tunable settings
|
||||
logs [N] [level] Recent logs (default: 20, error)
|
||||
models Available models
|
||||
sessions Active sessions
|
||||
clear Clear all sessions
|
||||
keys Manage API keys (add/list/revoke)
|
||||
lan LAN mode setup guide
|
||||
connect <ip> Connect to a remote OCP (sets env vars in rc file)
|
||||
restart Restart proxy
|
||||
restart gateway Restart gateway
|
||||
update Update OCP to latest version
|
||||
update --check Check for updates
|
||||
|
||||
Run 'ocp <command> --help' for details on a specific command.
|
||||
EOF
|
||||
}
|
||||
|
||||
# ── dispatch ─────────────────────────────────────────────────────────────
|
||||
# Check for --help/-h on any subcommand: ocp <cmd> --help
|
||||
subcmd="${1:-help}"
|
||||
shift 2>/dev/null || true
|
||||
|
||||
# Global --help
|
||||
if [[ "$subcmd" == "--help" || "$subcmd" == "-h" || "$subcmd" == "help" ]]; then
|
||||
cmd_help
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Per-subcommand --help
|
||||
for arg in "$@"; do
|
||||
if [[ "$arg" == "--help" || "$arg" == "-h" ]]; then
|
||||
fn="cmd_${subcmd}_help"
|
||||
if declare -f "$fn" > /dev/null 2>&1; then
|
||||
"$fn"
|
||||
else
|
||||
echo "No help available for '$subcmd'."
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
done
|
||||
|
||||
case "$subcmd" in
|
||||
usage) cmd_usage "${1:-}" ;;
|
||||
status) cmd_status ;;
|
||||
health) cmd_health ;;
|
||||
settings) cmd_settings "${1:-}" "${2:-}" ;;
|
||||
logs) cmd_logs "${1:-20}" "${2:-error}" ;;
|
||||
models) cmd_models ;;
|
||||
sessions) cmd_sessions ;;
|
||||
clear) cmd_clear ;;
|
||||
keys) cmd_keys "${1:-}" "${2:-}" ;;
|
||||
lan) cmd_lan ;;
|
||||
connect) cmd_connect "$@" ;;
|
||||
restart) cmd_restart "${1:-}" ;;
|
||||
doctor) cmd_doctor "$@" ;;
|
||||
update) cmd_update "$@" ;;
|
||||
*) echo "Unknown command: $subcmd"; echo ""; cmd_help; exit 1 ;;
|
||||
esac
|
||||
Executable
+736
@@ -0,0 +1,736 @@
|
||||
#!/usr/bin/env bash
|
||||
# ocp-connect — Lightweight client script to connect to a remote OCP instance
|
||||
# No dependencies beyond curl and python3 (available on most Linux/Mac systems)
|
||||
#
|
||||
# Install:
|
||||
# curl -fsSL https://raw.githubusercontent.com/dtzp555-max/ocp/main/ocp-connect -o ocp-connect
|
||||
# chmod +x ocp-connect
|
||||
#
|
||||
# Or run directly:
|
||||
# curl -fsSL https://raw.githubusercontent.com/dtzp555-max/ocp/main/ocp-connect | bash -s -- <host-ip> --key <key>
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
OCP_CONNECT_VERSION="1.3.0"
|
||||
|
||||
show_version() {
|
||||
echo "ocp-connect $OCP_CONNECT_VERSION"
|
||||
}
|
||||
|
||||
show_help() {
|
||||
cat <<'EOF'
|
||||
ocp-connect — Connect this machine to a remote OCP (Open Claude Proxy)
|
||||
|
||||
Configures OPENAI_BASE_URL and OPENAI_API_KEY in your shell rc file
|
||||
so tools like Claude Code, Cline, Aider, etc. point to the remote OCP.
|
||||
|
||||
Usage:
|
||||
ocp-connect <host-ip> [options]
|
||||
|
||||
Options:
|
||||
--port PORT Port OCP listens on (default: 3456)
|
||||
--key API_KEY API key (prompted interactively if remote requires auth)
|
||||
--version Print version and exit
|
||||
--help, -h Show this help
|
||||
|
||||
Examples:
|
||||
ocp-connect 192.168.1.10
|
||||
ocp-connect 192.168.1.10 --port 8080
|
||||
ocp-connect 192.168.1.10 --key ocp_abc123
|
||||
|
||||
What it does:
|
||||
1. Tests connectivity to the remote OCP
|
||||
2. Verifies your API key (if auth is enabled)
|
||||
3. Writes OPENAI_BASE_URL and OPENAI_API_KEY to ~/.bashrc or ~/.zshrc
|
||||
4. Sets system-level env vars (launchctl on macOS, systemd on Linux)
|
||||
5. Configures OpenClaw automatically; prints setup hints for other IDEs
|
||||
(Cline, Continue.dev, Cursor — manual configuration required)
|
||||
6. Runs a smoke test to confirm everything works
|
||||
|
||||
After running, reload your shell: source ~/.bashrc (or ~/.zshrc)
|
||||
EOF
|
||||
}
|
||||
|
||||
configure_ides() {
|
||||
local base_url="$1" key="$2" models_out="$3"
|
||||
|
||||
# --- OpenClaw ---
|
||||
local oc_config="$HOME/.openclaw/openclaw.json"
|
||||
local oc_found=false
|
||||
if command -v openclaw &>/dev/null || [[ -f "$oc_config" ]]; then
|
||||
oc_found=true
|
||||
fi
|
||||
|
||||
if $oc_found; then
|
||||
echo " IDE Configuration"
|
||||
echo " ─────────────────────────────────────"
|
||||
echo " Detected: OpenClaw ($oc_config)"
|
||||
echo ""
|
||||
printf " Configure OpenClaw to use this OCP? [Y/n] "
|
||||
local oc_answer
|
||||
{ read -r oc_answer </dev/tty; } 2>/dev/null || oc_answer="y"
|
||||
oc_answer="${oc_answer:-y}"
|
||||
|
||||
if [[ "$oc_answer" =~ ^[Yy]$ ]]; then
|
||||
# Ask for provider name
|
||||
printf " Provider name (models show as <name>/model-id) [ocp]: "
|
||||
local provider_name
|
||||
{ read -r provider_name </dev/tty; } 2>/dev/null || provider_name=""
|
||||
provider_name="${provider_name:-ocp}"
|
||||
# Sanitize: only allow alphanumeric, dash, underscore
|
||||
provider_name=$(echo "$provider_name" | tr -cd 'a-zA-Z0-9_-')
|
||||
[[ -z "$provider_name" ]] && provider_name="ocp"
|
||||
|
||||
# Ask for priority
|
||||
echo ""
|
||||
echo " How should OCP models be configured?"
|
||||
echo " 1) Primary — use OCP by default, keep existing models as backup"
|
||||
echo " 2) Backup — keep current primary, add OCP as additional option"
|
||||
echo ""
|
||||
printf " Choice [1]: "
|
||||
local priority_choice
|
||||
{ read -r priority_choice </dev/tty; } 2>/dev/null || priority_choice="1"
|
||||
priority_choice="${priority_choice:-1}"
|
||||
|
||||
echo ""
|
||||
echo " Writing OpenClaw config..."
|
||||
|
||||
# Use python3 to safely manipulate JSON
|
||||
local py_ok=0
|
||||
python3 - "$oc_config" "$base_url" "$key" "$provider_name" "$priority_choice" "$models_out" <<'PYEOF' && py_ok=1
|
||||
import sys, json, os
|
||||
|
||||
config_path = sys.argv[1]
|
||||
base_url = sys.argv[2]
|
||||
api_key = sys.argv[3]
|
||||
provider_name = sys.argv[4]
|
||||
priority = sys.argv[5] # "1" = primary, "2" = backup
|
||||
models_json_str = sys.argv[6]
|
||||
|
||||
# Parse models from OCP /v1/models response
|
||||
try:
|
||||
models_data = json.loads(models_json_str)
|
||||
model_ids = [m["id"] for m in models_data.get("data", [])]
|
||||
except:
|
||||
model_ids = ["claude-opus-4-6", "claude-sonnet-4-6", "claude-haiku-4"]
|
||||
|
||||
# Build provider entry
|
||||
provider = {
|
||||
"baseUrl": base_url + "/v1",
|
||||
"api": "openai-completions",
|
||||
"authHeader": bool(api_key),
|
||||
"models": []
|
||||
}
|
||||
|
||||
# Model metadata mapping. Prefix match on the model FAMILY (claude-opus / -sonnet /
|
||||
# -haiku), not a pinned version. A version-pinned prefix like "claude-sonnet-4"
|
||||
# silently misses "claude-sonnet-5" and falls through to the non-reasoning /
|
||||
# 8k-output default (PR #152 review) — every future Sonnet/Opus/Haiku bump would
|
||||
# re-trip it. Family prefixes classify any versioned ID correctly with no per-model
|
||||
# edit. (ADR 0003: models.json is the SPOT for model existence; /v1/models does not
|
||||
# expose reasoning/maxTokens, so family classification stays here.)
|
||||
model_meta = {
|
||||
"claude-opus": {"name": "Claude Opus (OCP)", "reasoning": True, "maxTokens": 16384},
|
||||
"claude-sonnet": {"name": "Claude Sonnet (OCP)", "reasoning": True, "maxTokens": 16384},
|
||||
"claude-haiku": {"name": "Claude Haiku (OCP)", "reasoning": False, "maxTokens": 8192},
|
||||
}
|
||||
|
||||
def get_model_meta(mid):
|
||||
"""Match model metadata by prefix."""
|
||||
for prefix, meta in sorted(model_meta.items(), key=lambda x: -len(x[0])):
|
||||
if mid.startswith(prefix):
|
||||
return meta
|
||||
return {"name": mid + " (OCP)", "reasoning": False, "maxTokens": 8192}
|
||||
|
||||
for mid in model_ids:
|
||||
meta = get_model_meta(mid)
|
||||
provider["models"].append({
|
||||
"id": mid,
|
||||
"name": meta["name"],
|
||||
"reasoning": meta["reasoning"],
|
||||
"input": ["text"],
|
||||
"cost": {"input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0},
|
||||
"contextWindow": 200000,
|
||||
"maxTokens": meta["maxTokens"]
|
||||
})
|
||||
|
||||
# Load or create config
|
||||
if os.path.exists(config_path):
|
||||
with open(config_path, "r") as f:
|
||||
config = json.load(f)
|
||||
else:
|
||||
os.makedirs(os.path.dirname(config_path), exist_ok=True)
|
||||
config = {}
|
||||
|
||||
# Ensure models.providers exists
|
||||
config.setdefault("models", {})
|
||||
config["models"].setdefault("mode", "merge")
|
||||
config["models"].setdefault("providers", {})
|
||||
|
||||
# Remove any previous OCP provider with the same name
|
||||
config["models"]["providers"][provider_name] = provider
|
||||
|
||||
# Set up auth profile if key is provided
|
||||
if api_key:
|
||||
config.setdefault("auth", {})
|
||||
config["auth"].setdefault("profiles", {})
|
||||
config["auth"]["profiles"][provider_name + ":default"] = {
|
||||
"provider": provider_name,
|
||||
"mode": "api_key"
|
||||
}
|
||||
|
||||
# Configure agent defaults — add model aliases
|
||||
config.setdefault("agents", {})
|
||||
config["agents"].setdefault("defaults", {})
|
||||
config["agents"]["defaults"].setdefault("models", {})
|
||||
|
||||
# Build alias map (family prefix match — version-agnostic, see model_meta note)
|
||||
alias_prefixes = {
|
||||
"claude-opus": "Claude Opus",
|
||||
"claude-sonnet": "Claude Sonnet",
|
||||
"claude-haiku": "Claude Haiku",
|
||||
}
|
||||
|
||||
for mid in model_ids:
|
||||
full_id = provider_name + "/" + mid
|
||||
alias = mid # fallback
|
||||
for prefix, name in sorted(alias_prefixes.items(), key=lambda x: -len(x[0])):
|
||||
if mid.startswith(prefix):
|
||||
alias = name
|
||||
break
|
||||
config["agents"]["defaults"]["models"][full_id] = {"alias": alias}
|
||||
|
||||
# Handle primary/backup
|
||||
if priority == "1":
|
||||
# OCP as primary — pick the best model (prefer the latest Sonnet for daily use,
|
||||
# tracking the `sonnet` alias default in models.json; fall back across versions).
|
||||
_sonnet_pref = ["claude-sonnet-5", "claude-sonnet-4-6"]
|
||||
primary_model = next(
|
||||
(provider_name + "/" + m for m in _sonnet_pref if m in model_ids),
|
||||
provider_name + "/" + model_ids[0],
|
||||
)
|
||||
config["agents"]["defaults"].setdefault("model", {})
|
||||
config["agents"]["defaults"]["model"]["primary"] = primary_model
|
||||
# Keep existing fallbacks
|
||||
config["agents"]["defaults"]["model"].setdefault("fallbacks", [])
|
||||
|
||||
# If backup (priority == "2"), don't change the primary — just add models to the list
|
||||
|
||||
# Update agent list entries that use old provider name patterns
|
||||
# (only update if agents.list exists and has entries using old OCP-like providers)
|
||||
if "list" in config.get("agents", {}):
|
||||
for agent in config["agents"]["list"]:
|
||||
agent_model = agent.get("model", {})
|
||||
if priority == "1" and agent_model.get("primary", "").startswith("claude-local/"):
|
||||
# Migrate from claude-local to new provider
|
||||
old_model_id = agent_model["primary"].split("/", 1)[1]
|
||||
if old_model_id in model_ids:
|
||||
agent_model["primary"] = provider_name + "/" + old_model_id
|
||||
# Also update subagents if they use claude-local
|
||||
sub = agent.get("subagents", config["agents"]["defaults"].get("subagents", {}))
|
||||
# Don't modify subagents in agent entries — they inherit from defaults
|
||||
|
||||
# Update defaults subagents model if using claude-local
|
||||
if priority == "1":
|
||||
sub = config["agents"]["defaults"].get("subagents", {})
|
||||
if sub.get("model", "").startswith("claude-local/"):
|
||||
old_id = sub["model"].split("/", 1)[1]
|
||||
if old_id in model_ids:
|
||||
sub["model"] = provider_name + "/" + old_id
|
||||
|
||||
with open(config_path, "w") as f:
|
||||
json.dump(config, f, indent=2, ensure_ascii=False)
|
||||
f.write("\n")
|
||||
|
||||
# === B1 fix: seed per-agent auth-profiles.json for OpenClaw multi-agent setups ===
|
||||
# OpenClaw's per-agent auth loader reads <agentDir>/auth-profiles.json (NOT the
|
||||
# root openclaw.json's auth.profiles section). Without a real key in each agent's
|
||||
# agentDir, OpenClaw rejects the lane with "No API key found for provider X".
|
||||
# See https://github.com/dtzp555-max/ocp/issues/12 for the full investigation.
|
||||
_seeded = []
|
||||
_failed = []
|
||||
_skipped_anonymous = []
|
||||
_profile_key = provider_name + ":default"
|
||||
# OpenClaw stores per-agent auth in <openclaw_root>/agents/<id>/agent/ when
|
||||
# the agent has no explicit `agentDir` field — derive that path so the
|
||||
# default `main` agent (which never sets agentDir) is also seeded.
|
||||
_openclaw_root = os.path.dirname(config_path)
|
||||
for _agent in config.get("agents", {}).get("list", []):
|
||||
_agent_dir = _agent.get("agentDir")
|
||||
if not _agent_dir:
|
||||
_agent_id = _agent.get("id")
|
||||
if not _agent_id:
|
||||
continue
|
||||
_agent_dir = os.path.join(_openclaw_root, "agents", _agent_id, "agent")
|
||||
if not api_key:
|
||||
# Anonymous mode is incompatible with OpenClaw per-agent auth (empty key
|
||||
# is dropped by OpenClaw's pi-auth-credentials). Per OCP issue #12 §14
|
||||
# decision: take Path C (require --key for OpenClaw multi-agent setups).
|
||||
_skipped_anonymous.append(_agent_dir)
|
||||
continue
|
||||
_profiles_path = os.path.join(_agent_dir, "auth-profiles.json")
|
||||
try:
|
||||
os.makedirs(_agent_dir, exist_ok=True)
|
||||
except OSError as _e:
|
||||
_failed.append((_profiles_path, "mkdir: " + str(_e)))
|
||||
continue
|
||||
if os.path.exists(_profiles_path):
|
||||
try:
|
||||
with open(_profiles_path) as _f:
|
||||
_ap = json.load(_f)
|
||||
except json.JSONDecodeError:
|
||||
# Corrupted JSON — back up and rebuild from scratch.
|
||||
_bak = _profiles_path + ".bak"
|
||||
try:
|
||||
os.rename(_profiles_path, _bak)
|
||||
print(f" ⚠ {_profiles_path} was corrupt; backed up to {_bak}")
|
||||
except OSError:
|
||||
pass
|
||||
_ap = {"version": 1, "profiles": {}}
|
||||
except OSError as _e:
|
||||
# Read failure (permissions, disk) — skip to AVOID clobbering
|
||||
# the user's existing profile (which may hold other providers' keys).
|
||||
_failed.append((_profiles_path, "read: " + str(_e)))
|
||||
continue
|
||||
else:
|
||||
_ap = {"version": 1, "profiles": {}}
|
||||
_ap.setdefault("version", 1)
|
||||
_ap.setdefault("profiles", {})
|
||||
_ap["profiles"][_profile_key] = {
|
||||
"type": "api_key",
|
||||
"provider": provider_name,
|
||||
"key": api_key
|
||||
}
|
||||
try:
|
||||
with open(_profiles_path, "w") as _f:
|
||||
json.dump(_ap, _f, indent=2, ensure_ascii=False)
|
||||
_f.write("\n")
|
||||
os.chmod(_profiles_path, 0o600)
|
||||
_seeded.append(_profiles_path)
|
||||
except OSError as _e:
|
||||
_failed.append((_profiles_path, "write: " + str(_e)))
|
||||
|
||||
# Report — all three sections are independent (mixed scenarios are surfaced).
|
||||
if _seeded:
|
||||
print(f" ✓ Per-agent auth profile seeded ({len(_seeded)}):")
|
||||
for _p in _seeded:
|
||||
print(f" • {_p}")
|
||||
if _failed:
|
||||
print(f" ⚠ Per-agent auth profile FAILED ({len(_failed)}):")
|
||||
for _p, _err in _failed:
|
||||
print(f" • {_p}: {_err}")
|
||||
print(" OpenClaw will report \"No API key found\" for the failed agents.")
|
||||
print(" Fix the underlying error (permissions / disk) and re-run ocp-connect.")
|
||||
if _skipped_anonymous:
|
||||
print(f" ⚠ OpenClaw multi-agent mode detected ({len(_skipped_anonymous)} agents).")
|
||||
print(" Anonymous mode does not work for OpenClaw per-agent auth.")
|
||||
print(" Re-run with: ocp-connect <host> --key ocp_xxx")
|
||||
PYEOF
|
||||
|
||||
if [[ $py_ok -eq 1 ]]; then
|
||||
echo " ✓ OpenClaw configured"
|
||||
echo " Provider: $provider_name"
|
||||
echo " Models:"
|
||||
# List models from the already-fetched models_out
|
||||
echo "$models_out" | python3 -c "
|
||||
import sys,json
|
||||
d=json.loads(sys.stdin.read())
|
||||
pn='$provider_name'
|
||||
for m in d.get('data',[]):
|
||||
print(' • ' + pn + '/' + m['id'])
|
||||
" 2>/dev/null
|
||||
if [[ "$priority_choice" == "1" ]]; then
|
||||
echo " Priority: PRIMARY (default model)"
|
||||
else
|
||||
echo " Priority: BACKUP (available in model selector)"
|
||||
fi
|
||||
echo ""
|
||||
echo " Restart OpenClaw to apply: openclaw gateway restart"
|
||||
else
|
||||
echo " ⚠ Failed to write OpenClaw config. You can configure it manually."
|
||||
fi
|
||||
else
|
||||
echo " Skipped OpenClaw configuration."
|
||||
fi
|
||||
echo ""
|
||||
fi
|
||||
|
||||
# --- Other IDEs: print manual instructions ---
|
||||
local other_ides_shown=false
|
||||
|
||||
# Collect VS Code extension list once (reused by Cline and Continue.dev checks)
|
||||
local _vscode_exts=""
|
||||
if command -v code &>/dev/null; then
|
||||
_vscode_exts=$(code --list-extensions 2>/dev/null || true)
|
||||
fi
|
||||
if [[ -z "$_vscode_exts" && -d "$HOME/.vscode/extensions" ]]; then
|
||||
_vscode_exts=$(ls "$HOME/.vscode/extensions/" 2>/dev/null || true)
|
||||
fi
|
||||
|
||||
# Extract model IDs from /v1/models response for use in hints
|
||||
local _model_ids
|
||||
_model_ids=$(echo "$models_out" | python3 -c "
|
||||
import sys,json
|
||||
try:
|
||||
d=json.loads(sys.stdin.read())
|
||||
print(', '.join(m['id'] for m in d.get('data', [])))
|
||||
except Exception:
|
||||
print('claude-sonnet-4-6, claude-opus-4-6, claude-haiku-4-5-20251001')
|
||||
" 2>/dev/null || echo "claude-sonnet-4-6, claude-opus-4-6, claude-haiku-4-5-20251001")
|
||||
|
||||
# Key display: truncate if longer than 16 chars to avoid screenshot leakage,
|
||||
# show explicit "(none)" in anonymous mode so users don't paste blank fields.
|
||||
local _key_display
|
||||
if [[ -z "$key" ]]; then
|
||||
_key_display="(none — anonymous mode; most external IDEs require a non-empty API Key)"
|
||||
elif [[ ${#key} -gt 16 ]]; then
|
||||
_key_display="${key:0:8}...${key: -4}"
|
||||
else
|
||||
_key_display="$key"
|
||||
fi
|
||||
|
||||
# Detect Cline (VS Code extension saoudrizwan.claude-dev, see issue #12)
|
||||
if echo "$_vscode_exts" | grep -qiE 'cline|saoudrizwan\.claude-dev'; then
|
||||
if ! $other_ides_shown; then
|
||||
echo " Other IDEs detected:"
|
||||
other_ides_shown=true
|
||||
fi
|
||||
echo " • Cline: VSCode → Cline panel → Settings → API Provider = \"OpenAI Compatible\""
|
||||
echo " Base URL: $base_url/v1"
|
||||
echo " API Key: $_key_display"
|
||||
echo " Model ID: $_model_ids"
|
||||
fi
|
||||
|
||||
# Detect Continue.dev (extension ID continue.continue or config file, see issue #12)
|
||||
if echo "$_vscode_exts" | grep -qi 'continue\.continue' \
|
||||
|| [[ -f "$HOME/.continue/config.yaml" ]] \
|
||||
|| [[ -f "$HOME/.continue/config.json" ]]; then
|
||||
if ! $other_ides_shown; then
|
||||
echo " Other IDEs detected:"
|
||||
other_ides_shown=true
|
||||
fi
|
||||
echo " • Continue.dev: edit ~/.continue/config.yaml — add under \`models:\` (top-level key):"
|
||||
echo " models:"
|
||||
echo " - name: OCP Sonnet"
|
||||
echo " provider: openai"
|
||||
echo " model: claude-sonnet-4-6"
|
||||
echo " apiBase: $base_url/v1"
|
||||
echo " apiKey: $_key_display"
|
||||
echo " (other model IDs: $_model_ids)"
|
||||
fi
|
||||
|
||||
# Detect Cursor (command, ~/.cursor dir, or /Applications/Cursor.app, see issue #12)
|
||||
if command -v cursor &>/dev/null \
|
||||
|| [[ -d "$HOME/.cursor" ]] \
|
||||
|| [[ -d "/Applications/Cursor.app" ]]; then
|
||||
if ! $other_ides_shown; then
|
||||
echo " Other IDEs detected:"
|
||||
other_ides_shown=true
|
||||
fi
|
||||
echo " • Cursor: Cmd+Shift+P → 'Cursor Settings' → Models →"
|
||||
echo " OpenAI API Key: $_key_display"
|
||||
echo " Override OpenAI Base URL: $base_url/v1"
|
||||
echo " Custom OpenAI Models: $_model_ids"
|
||||
fi
|
||||
|
||||
# Detect opencode (https://opencode.ai, SST team CLI, see issue #12)
|
||||
if command -v opencode &>/dev/null \
|
||||
|| [[ -x "$HOME/.opencode/bin/opencode" ]] \
|
||||
|| [[ -d "$HOME/.local/share/opencode" ]]; then
|
||||
if ! $other_ides_shown; then
|
||||
echo " Other IDEs detected:"
|
||||
other_ides_shown=true
|
||||
fi
|
||||
echo " • opencode: not yet auto-configured by ocp-connect (PR follow-up)."
|
||||
echo " Run \`opencode providers login openai\` and provide:"
|
||||
echo " Base URL: $base_url/v1"
|
||||
echo " API Key: $_key_display"
|
||||
echo " Available model IDs: $_model_ids"
|
||||
fi
|
||||
|
||||
if $other_ides_shown; then
|
||||
echo ""
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
local host="" port=3456 key=""
|
||||
|
||||
# Parse args
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--port) port="${2:?'--port requires a value'}"; shift 2 ;;
|
||||
--key) key="${2:?'--key requires a value'}"
|
||||
[[ -z "$key" ]] && { echo "Error: --key cannot be empty (omit --key entirely for anonymous mode)"; exit 1; }
|
||||
shift 2 ;;
|
||||
--version) show_version; exit 0 ;;
|
||||
--help|-h) show_help; exit 0 ;;
|
||||
-*) echo "Unknown option: $1"; show_help; exit 1 ;;
|
||||
*) host="$1"; shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ -z "$host" ]]; then
|
||||
echo "Error: host IP is required."
|
||||
echo ""
|
||||
show_help
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! [[ "$host" =~ ^[a-zA-Z0-9._-]+$ ]]; then
|
||||
echo "Error: invalid host '$host'"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check dependencies
|
||||
for cmd in curl python3; do
|
||||
if ! command -v "$cmd" &>/dev/null; then
|
||||
echo "Error: '$cmd' is required but not found."
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
local base_url="http://$host:$port"
|
||||
|
||||
echo "OCP Connect v$OCP_CONNECT_VERSION"
|
||||
echo "─────────────────────────────────────"
|
||||
echo " Remote: $base_url"
|
||||
echo ""
|
||||
|
||||
# Step 1: Test connectivity via /health
|
||||
echo " Checking connectivity..."
|
||||
local health_json
|
||||
health_json=$(curl -sf --max-time 10 "$base_url/health" 2>/dev/null) || {
|
||||
echo " ✗ Cannot reach $base_url/health"
|
||||
echo " Make sure OCP is running on $host and bound to 0.0.0.0 (LAN mode)."
|
||||
exit 1
|
||||
}
|
||||
echo " ✓ Connected"
|
||||
echo ""
|
||||
|
||||
# Step 2: Show remote info
|
||||
local remote_version auth_mode
|
||||
remote_version=$(echo "$health_json" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); print(d.get('version','?'))" 2>/dev/null || echo "?")
|
||||
auth_mode=$(echo "$health_json" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); print(d.get('authMode','none'))" 2>/dev/null || echo "none")
|
||||
|
||||
echo " Remote OCP v$remote_version (auth: $auth_mode)"
|
||||
echo ""
|
||||
|
||||
# Step 2.5: auto-discover anonymous key from /health (issue #12 §14 Path A).
|
||||
# The server advertises anonymousKey in /health ONLY when the admin has set
|
||||
# PROXY_ADVERTISE_ANON_KEY=1 (default off — /health is unauthenticated, so
|
||||
# advertising exposes the shared key to any LAN-reachable device; issue #109).
|
||||
# Localhost callers always receive it regardless. When the field is absent,
|
||||
# ocp-connect falls back to anonymous access / interactive --key (step 3 below).
|
||||
if [[ -z "$key" ]]; then
|
||||
local anon_key
|
||||
anon_key=$(echo "$health_json" | python3 -c "
|
||||
import sys, json
|
||||
try:
|
||||
d = json.loads(sys.stdin.read())
|
||||
k = d.get('anonymousKey')
|
||||
except Exception:
|
||||
k = None
|
||||
print(k if k else '')
|
||||
" 2>/dev/null || echo "")
|
||||
if [[ -n "$anon_key" ]]; then
|
||||
key="$anon_key"
|
||||
local _anon_display="$anon_key"
|
||||
if [[ ${#anon_key} -gt 16 ]]; then
|
||||
_anon_display="${anon_key:0:8}...${anon_key: -4}"
|
||||
fi
|
||||
echo " ⓘ Using server-advertised anonymous key: $_anon_display"
|
||||
echo " (set by admin via PROXY_ANONYMOUS_KEY; see issue #12 §14 Path A)"
|
||||
echo ""
|
||||
fi
|
||||
fi
|
||||
|
||||
# Step 3: Determine if key is needed
|
||||
if [[ "$auth_mode" != "none" && -z "$key" ]]; then
|
||||
# Try anonymous access first (zero-config: server may allow it)
|
||||
if curl -sf --max-time 5 "$base_url/v1/models" >/dev/null 2>&1; then
|
||||
echo " Server allows anonymous access — no key needed."
|
||||
echo ""
|
||||
else
|
||||
echo " Remote requires authentication."
|
||||
echo " Ask the OCP admin to run: ocp keys add <name>"
|
||||
printf " Enter API key (or press Enter to skip): "
|
||||
{ read -rs key </dev/tty; } 2>/dev/null || key=""
|
||||
echo
|
||||
if [[ -z "$key" ]]; then
|
||||
echo ""
|
||||
echo " ✗ No key provided — cannot connect to an auth-required remote without a key."
|
||||
echo " Use: ocp-connect $host --key <key>"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# Step 4: Test API access via /v1/models
|
||||
echo " Testing API access..."
|
||||
local models_out models_ok=0
|
||||
if [[ -n "$key" ]]; then
|
||||
models_out=$(curl -sf --max-time 10 \
|
||||
-H "Authorization: Bearer $key" \
|
||||
"$base_url/v1/models" 2>/dev/null) && models_ok=1
|
||||
else
|
||||
models_out=$(curl -sf --max-time 10 "$base_url/v1/models" 2>/dev/null) && models_ok=1
|
||||
fi
|
||||
|
||||
if [[ $models_ok -eq 0 ]]; then
|
||||
echo " ✗ API access failed — key may be invalid or revoked."
|
||||
exit 1
|
||||
fi
|
||||
local model_count
|
||||
model_count=$(echo "$models_out" | python3 -c "import sys,json; print(len(json.loads(sys.stdin.read()).get('data',[])))" 2>/dev/null || echo "?")
|
||||
echo " ✓ API accessible ($model_count models available)"
|
||||
echo ""
|
||||
|
||||
# Step 5: Detect shell rc files and OS
|
||||
local rc_files=()
|
||||
local is_mac=false
|
||||
[[ "$(uname)" == "Darwin" ]] && is_mac=true
|
||||
|
||||
# Write to all relevant rc files
|
||||
if [[ "${SHELL:-}" == */fish ]]; then
|
||||
echo " Note: fish shell detected. Writing to ~/.bashrc — add to fish config manually."
|
||||
rc_files+=("$HOME/.bashrc")
|
||||
elif $is_mac; then
|
||||
# macOS: default shell since Catalina (2019) is zsh.
|
||||
# Always write ~/.zshrc (create if absent — zsh tolerates an empty file).
|
||||
# Only write ~/.bashrc if it already exists (don't surprise users with new files).
|
||||
[[ -f "$HOME/.bashrc" ]] && rc_files+=("$HOME/.bashrc")
|
||||
# zshrc: always include on macOS; create the file if it doesn't exist yet
|
||||
[[ -f "$HOME/.zshrc" ]] || touch "$HOME/.zshrc"
|
||||
rc_files+=("$HOME/.zshrc")
|
||||
else
|
||||
# Linux / other: write to whichever rc files already exist or match current shell
|
||||
[[ -f "$HOME/.bashrc" || "${SHELL:-}" == */bash ]] && rc_files+=("$HOME/.bashrc")
|
||||
[[ -f "$HOME/.zshrc" || "${SHELL:-}" == */zsh ]] && rc_files+=("$HOME/.zshrc")
|
||||
# If neither exists, fall back to creating one for the current shell
|
||||
[[ ${#rc_files[@]} -eq 0 ]] && rc_files+=("$HOME/.${SHELL##*/}rc")
|
||||
fi
|
||||
|
||||
# Step 6: Remove any previously written OCP LAN lines (idempotent) from all rc files
|
||||
for rc_file in "${rc_files[@]}"; do
|
||||
if [[ -f "$rc_file" ]]; then
|
||||
local tmp_rc
|
||||
tmp_rc=$(mktemp)
|
||||
if python3 - "$rc_file" "$tmp_rc" <<'PYEOF'
|
||||
import sys
|
||||
src, dst = sys.argv[1], sys.argv[2]
|
||||
lines = open(src).readlines()
|
||||
out = []
|
||||
skip = False
|
||||
for line in lines:
|
||||
s = line.rstrip('\n')
|
||||
if s == '# OCP LAN (added by ocp connect)':
|
||||
skip = True
|
||||
continue
|
||||
if skip and (s == '' or s.startswith('export OPENAI_BASE_URL=') or s.startswith('export OPENAI_API_KEY=')):
|
||||
continue
|
||||
skip = False
|
||||
out.append(line)
|
||||
open(dst, 'w').writelines(out)
|
||||
PYEOF
|
||||
then
|
||||
cp "$tmp_rc" "$rc_file"
|
||||
fi
|
||||
rm -f "$tmp_rc"
|
||||
fi
|
||||
done
|
||||
|
||||
# Step 7: Append new config to all rc files
|
||||
for rc_file in "${rc_files[@]}"; do
|
||||
{
|
||||
echo ""
|
||||
echo "# OCP LAN (added by ocp connect)"
|
||||
echo "export OPENAI_BASE_URL='$base_url/v1'"
|
||||
if [[ -n "$key" ]]; then
|
||||
echo "export OPENAI_API_KEY='$key'"
|
||||
fi
|
||||
} >> "$rc_file"
|
||||
chmod 600 "$rc_file" 2>/dev/null || true
|
||||
done
|
||||
|
||||
echo " Shell config:"
|
||||
for rc_file in "${rc_files[@]}"; do
|
||||
echo " ✓ $(basename "$rc_file")"
|
||||
done
|
||||
echo " OPENAI_BASE_URL=$base_url/v1"
|
||||
if [[ -n "$key" ]]; then
|
||||
echo " OPENAI_API_KEY=${key:0:8}..."
|
||||
fi
|
||||
|
||||
# Step 7b: System-level env vars (for IDEs, daemons, GUI apps)
|
||||
if $is_mac; then
|
||||
# macOS: launchctl setenv makes vars visible to all GUI apps and launchd services
|
||||
launchctl setenv OPENAI_BASE_URL "$base_url/v1" 2>/dev/null
|
||||
if [[ -n "$key" ]]; then
|
||||
launchctl setenv OPENAI_API_KEY "$key" 2>/dev/null
|
||||
fi
|
||||
echo ""
|
||||
echo " System-level (launchctl):"
|
||||
echo " ✓ OPENAI_BASE_URL set for GUI apps and daemons"
|
||||
echo " Note: launchctl vars reset on reboot. Add to Login Items or re-run ocp-connect."
|
||||
else
|
||||
# Linux: write to environment.d for systemd user services
|
||||
local env_dir="$HOME/.config/environment.d"
|
||||
mkdir -p "$env_dir" 2>/dev/null
|
||||
{
|
||||
echo "OPENAI_BASE_URL=$base_url/v1"
|
||||
if [[ -n "$key" ]]; then
|
||||
echo "OPENAI_API_KEY=$key"
|
||||
fi
|
||||
} > "$env_dir/ocp.conf"
|
||||
chmod 600 "$env_dir/ocp.conf" 2>/dev/null || true
|
||||
echo ""
|
||||
echo " System-level (systemd):"
|
||||
echo " ✓ $env_dir/ocp.conf"
|
||||
echo " Applies to systemd user services after re-login."
|
||||
fi
|
||||
echo ""
|
||||
|
||||
# Step 7c: Interactive IDE configuration
|
||||
configure_ides "$base_url" "$key" "$models_out"
|
||||
|
||||
# Step 8: Quick smoke test
|
||||
echo " Running smoke test..."
|
||||
# Pick the first available model from /v1/models
|
||||
local smoke_model
|
||||
smoke_model=$(echo "$models_out" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); print(d['data'][0]['id'])" 2>/dev/null || echo "claude-haiku-4-5-20251001")
|
||||
local chat_payload="{\"model\":\"$smoke_model\",\"messages\":[{\"role\":\"user\",\"content\":\"Reply with OK only.\"}],\"max_tokens\":10}"
|
||||
local chat_out chat_ok=0
|
||||
if [[ -n "$key" ]]; then
|
||||
chat_out=$(curl -sf --max-time 30 \
|
||||
-H "Authorization: Bearer $key" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "$chat_payload" \
|
||||
"$base_url/v1/chat/completions" 2>/dev/null) && chat_ok=1
|
||||
else
|
||||
chat_out=$(curl -sf --max-time 30 \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "$chat_payload" \
|
||||
"$base_url/v1/chat/completions" 2>/dev/null) && chat_ok=1
|
||||
fi
|
||||
|
||||
if [[ $chat_ok -eq 1 ]]; then
|
||||
local reply
|
||||
reply=$(echo "$chat_out" | python3 -c "import sys,json; d=json.loads(sys.stdin.read()); print(d['choices'][0]['message']['content'].strip())" 2>/dev/null || echo "(response received)")
|
||||
echo " ✓ Smoke test passed: $reply"
|
||||
echo " Note: smoke test only verifies OCP is reachable and the key is valid."
|
||||
echo " It does not verify your IDE/agent end-to-end. To verify OpenClaw works,"
|
||||
echo " restart it (\`openclaw gateway restart\`) and send a test message to your bot."
|
||||
else
|
||||
echo " ⚠ Smoke test failed (proxy is reachable but chat completion did not succeed)."
|
||||
echo " The env vars have still been written. Check the remote OCP logs."
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo " Done. Reload your shell to apply:"
|
||||
for rc_file in "${rc_files[@]}"; do
|
||||
echo " source $rc_file"
|
||||
done
|
||||
}
|
||||
|
||||
main "$@"
|
||||
@@ -0,0 +1,314 @@
|
||||
/**
|
||||
* OCP Plugin — registers /ocp as a native slash command in OpenClaw gateway.
|
||||
* Calls the local claude-proxy and formats the response.
|
||||
*
|
||||
* Port resolution (in priority order):
|
||||
* 1. OCP_PROXY_URL env (full URL, e.g. http://10.0.0.5:3456)
|
||||
* 2. CLAUDE_PROXY_PORT env (port only; localhost assumed)
|
||||
* 3. Fallback: http://127.0.0.1:3456 (OCP server source default since v1.0)
|
||||
*
|
||||
* If a particular host's OCP plist injects a non-default CLAUDE_PROXY_PORT,
|
||||
* the OpenClaw launchd plist for that host must also inject the same
|
||||
* CLAUDE_PROXY_PORT into the plugin's env, or the plugin will fall back to
|
||||
* 3456 and miss the server.
|
||||
*/
|
||||
const PROXY = process.env.OCP_PROXY_URL
|
||||
|| (process.env.CLAUDE_PROXY_PORT ? `http://127.0.0.1:${process.env.CLAUDE_PROXY_PORT}` : "http://127.0.0.1:3456");
|
||||
|
||||
// Wrap output in monospace code block for Telegram/Discord alignment
|
||||
function mono(text) { return "```\n" + text + "\n```"; }
|
||||
|
||||
async function fetchJSON(path) {
|
||||
const resp = await fetch(`${PROXY}${path}`, { signal: AbortSignal.timeout(15000) });
|
||||
if (!resp.ok) throw new Error(`proxy ${resp.status}: ${resp.statusText}`);
|
||||
return resp.json();
|
||||
}
|
||||
|
||||
function bar(pct, width = 16) {
|
||||
const filled = Math.round(pct * width);
|
||||
return "█".repeat(filled) + "░".repeat(width - filled);
|
||||
}
|
||||
|
||||
function fmtMs(ms) {
|
||||
return ms >= 1000 ? `${(ms / 1000).toFixed(0)}s` : `${ms}ms`;
|
||||
}
|
||||
|
||||
function fmtChars(c) {
|
||||
return c >= 1000 ? `${(c / 1000).toFixed(0)}K` : `${c}`;
|
||||
}
|
||||
|
||||
// ── Subcommand handlers ─────────────────────────────────────────────────
|
||||
|
||||
async function cmdUsage() {
|
||||
const d = await fetchJSON("/usage");
|
||||
const plan = d.plan || {};
|
||||
const s = plan.currentSession || {};
|
||||
const w = plan.weeklyLimits?.allModels || {};
|
||||
const e = plan.extraUsage || {};
|
||||
const px = d.proxy;
|
||||
const models = d.models || {};
|
||||
|
||||
let out = "";
|
||||
|
||||
// Show subscription info if available
|
||||
if (plan.subscription) {
|
||||
out += `Plan: ${plan.subscription} (${plan.rateLimitTier || "default"})\n`;
|
||||
}
|
||||
|
||||
if (s.utilization !== null && s.utilization !== undefined) {
|
||||
// Full plan usage data available (API key mode)
|
||||
out += "Plan Usage Limits\n";
|
||||
out += "─────────────────────────────\n";
|
||||
out += `Current session ${bar(s.utilization)} ${s.percent}\n`;
|
||||
out += ` Resets in ${s.resetsIn} (${s.resetsAtHuman})\n\n`;
|
||||
out += `Weekly (all) ${bar(w.utilization)} ${w.percent}\n`;
|
||||
out += ` Resets in ${w.resetsIn} (${w.resetsAtHuman})\n\n`;
|
||||
out += `Extra usage ${e.status === "allowed" ? "on" : "off"}\n\n`;
|
||||
} else {
|
||||
// No API key — show note
|
||||
out += "Session/weekly %: claude.ai/settings\n";
|
||||
out += "(Anthropic OAuth API doesn't expose limits yet)\n\n";
|
||||
}
|
||||
|
||||
const modelNames = Object.keys(models).sort();
|
||||
if (modelNames.length > 0) {
|
||||
out += "Model Stats\n";
|
||||
const hdr = `${"Model".padEnd(14)} ${"Req".padStart(4)} ${"OK".padStart(3)} ${"Er".padStart(3)} ${"AvgT".padStart(5)} ${"MaxT".padStart(5)} ${"AvgP".padStart(5)} ${"MaxP".padStart(5)}`;
|
||||
out += hdr + "\n";
|
||||
out += "─".repeat(hdr.length) + "\n";
|
||||
let total = 0;
|
||||
for (const name of modelNames) {
|
||||
const m = models[name];
|
||||
total += m.requests;
|
||||
const short = name.replace("claude-", "").replace("-4-5-20251001", "").replace("-4-6", "");
|
||||
out += `${short.padEnd(14)} ${String(m.requests).padStart(4)} ${String(m.successes).padStart(3)} ${String(m.errors).padStart(3)} ${fmtMs(m.avgElapsed).padStart(5)} ${fmtMs(m.maxElapsed).padStart(5)} ${fmtChars(m.avgPromptChars).padStart(5)} ${fmtChars(m.maxPromptChars).padStart(5)}\n`;
|
||||
}
|
||||
out += `${"Total".padEnd(14)} ${String(total).padStart(4)}\n`;
|
||||
}
|
||||
|
||||
out += `\nProxy: up ${px.uptime} | ${px.totalRequests} reqs | ${px.errors} err | ${px.timeouts} timeout`;
|
||||
return out;
|
||||
}
|
||||
|
||||
async function cmdHealth() {
|
||||
const d = await fetchJSON("/health");
|
||||
let out = `Status: ${d.status} | v${d.version}\n`;
|
||||
out += `Uptime: ${d.uptimeHuman}\n`;
|
||||
out += `Auth: ${d.auth?.ok ? "ok" : d.auth?.message || "unknown"}\n`;
|
||||
out += `Binary: ${d.claudeBinaryOk ? "ok" : "missing"}\n`;
|
||||
out += `Sessions: ${d.sessions?.length || 0} active\n`;
|
||||
out += `Requests: ${d.stats?.totalRequests || 0} total, ${d.stats?.activeRequests || 0} active\n`;
|
||||
out += `Errors: ${d.stats?.errors || 0} | Timeouts: ${d.stats?.timeouts || 0}\n`;
|
||||
if (d.recentErrors?.length) {
|
||||
out += "\nRecent errors:\n";
|
||||
for (const e of d.recentErrors.slice(-3)) {
|
||||
out += ` ${e.time?.slice(11, 19) || "?"} ${e.message}\n`;
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
async function cmdStatus() {
|
||||
const d = await fetchJSON("/status");
|
||||
const icon = d.proxy?.status === "ok" ? "🟢" : d.proxy?.status === "degraded" ? "🟡" : "🔴";
|
||||
let out = `${icon} ${d.proxy?.status} | v${d.proxy?.version} | up ${d.proxy?.uptime} | auth ${d.proxy?.auth}\n`;
|
||||
out += `Sessions: ${d.proxy?.activeSessions || 0}\n`;
|
||||
out += `Requests: ${d.requests?.total || 0} | active ${d.requests?.active || 0} | err ${d.requests?.errors || 0} | timeout ${d.requests?.timeouts || 0}\n`;
|
||||
if (d.plan?.currentSession) {
|
||||
out += `\nSession: ${d.plan.currentSession.percent} (resets ${d.plan.currentSession.resetsIn})\n`;
|
||||
out += `Weekly: ${d.plan.weeklyLimits?.allModels?.percent} (resets ${d.plan.weeklyLimits?.allModels?.resetsIn})`;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
async function cmdSettings(args) {
|
||||
if (!args) {
|
||||
const d = await fetchJSON("/settings");
|
||||
let out = "OCP Settings\n─────────────────────────────\n";
|
||||
for (const k of ["timeout", "firstByteTimeout", "maxConcurrent", "sessionTTL", "maxPromptChars"]) {
|
||||
const v = d[k];
|
||||
if (v) out += `${k.padEnd(20)} ${String(v.value).padStart(8)} ${(v.unit || "").padEnd(6)} ${v.desc}\n`;
|
||||
}
|
||||
if (d.tiers) {
|
||||
out += "\nTimeout tiers:\n";
|
||||
for (const t of ["opus", "sonnet", "haiku"]) {
|
||||
const info = d.tiers[t];
|
||||
if (info) out += ` ${t.padEnd(8)} base=${info.base}ms perChar=${info.perPromptChar}\n`;
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// Parse "key value"
|
||||
const parts = args.trim().split(/\s+/);
|
||||
if (parts.length < 2 || parts[0] === "--help" || parts[0] === "-h") {
|
||||
return "Usage: /ocp settings <key> <value>\nKeys: timeout, firstByteTimeout, maxConcurrent, sessionTTL, maxPromptChars, tiers.opus.base, tiers.sonnet.base, tiers.haiku.base, tiers.*.perChar";
|
||||
}
|
||||
const [key, val] = parts;
|
||||
const numVal = Number(val);
|
||||
if (isNaN(numVal)) return `Error: value must be a number, got "${val}"`;
|
||||
|
||||
const resp = await fetch(`${PROXY}/settings`, {
|
||||
method: "PATCH",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ [key]: numVal }),
|
||||
signal: AbortSignal.timeout(5000),
|
||||
});
|
||||
const d = await resp.json();
|
||||
if (d.errors?.length) return `✗ ${d.errors.join("; ")}`;
|
||||
return `✓ ${key} = ${numVal}`;
|
||||
}
|
||||
|
||||
async function cmdModels() {
|
||||
const d = await fetchJSON("/v1/models");
|
||||
return (d.data || []).map((m) => ` ${m.id}`).join("\n") || "No models.";
|
||||
}
|
||||
|
||||
async function cmdSessions() {
|
||||
const d = await fetchJSON("/sessions");
|
||||
if (!d.sessions?.length) return "No active sessions.";
|
||||
return d.sessions.map((s) => ` ${s.id.slice(0, 16)}… model=${s.model} msgs=${s.messages}`).join("\n");
|
||||
}
|
||||
|
||||
async function cmdClear() {
|
||||
const resp = await fetch(`${PROXY}/sessions`, { method: "DELETE", signal: AbortSignal.timeout(5000) });
|
||||
const d = await resp.json();
|
||||
return `Cleared ${d.cleared} sessions.`;
|
||||
}
|
||||
|
||||
async function cmdVersion() {
|
||||
const d = await fetchJSON("/health");
|
||||
return `OCP v${d.version || "?"}\nUptime: ${d.uptimeHuman || "?"}\nNode: ${process.version}\nPlatform: ${process.platform} ${process.arch}`;
|
||||
}
|
||||
|
||||
async function cmdTest() {
|
||||
const t0 = Date.now();
|
||||
try {
|
||||
const resp = await fetch(`${PROXY}/v1/chat/completions`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
model: "claude-haiku-4-5-20251001",
|
||||
messages: [{ role: "user", content: "say ok" }],
|
||||
max_tokens: 5,
|
||||
}),
|
||||
signal: AbortSignal.timeout(30000),
|
||||
});
|
||||
const d = await resp.json();
|
||||
const elapsed = Date.now() - t0;
|
||||
if (d.choices?.[0]?.message?.content) {
|
||||
return `✓ Proxy OK (${elapsed}ms)\n Model: haiku\n Response: "${d.choices[0].message.content.slice(0, 50)}"`;
|
||||
}
|
||||
return `✗ Unexpected response: ${JSON.stringify(d).slice(0, 100)}`;
|
||||
} catch (e) {
|
||||
return `✗ Test failed (${Date.now() - t0}ms): ${e.message}`;
|
||||
}
|
||||
}
|
||||
|
||||
async function cmdRestart(args) {
|
||||
const target = (args || "").trim().toLowerCase();
|
||||
const { execSync } = await import("node:child_process");
|
||||
const uid = typeof process.getuid === "function" ? process.getuid() : 501;
|
||||
const macProxy = `launchctl kickstart -k gui/${uid}/dev.ocp.proxy`;
|
||||
const macGateway = `launchctl kickstart -k gui/${uid}/ai.openclaw.gateway`;
|
||||
try {
|
||||
if (target === "gateway") {
|
||||
execSync(macGateway, { timeout: 15000 });
|
||||
return "✓ Gateway restarted";
|
||||
} else if (target === "all") {
|
||||
execSync(macProxy, { timeout: 15000 });
|
||||
// Gateway restart will kill this plugin too, so do it last
|
||||
execSync(macGateway, { timeout: 15000 });
|
||||
return "✓ Proxy + Gateway restarted";
|
||||
} else {
|
||||
execSync(macProxy, { timeout: 15000 });
|
||||
return "✓ Proxy restarted";
|
||||
}
|
||||
} catch (e) {
|
||||
// Linux: systemd user services
|
||||
try {
|
||||
if (target === "gateway") {
|
||||
execSync("systemctl --user restart openclaw-gateway", { timeout: 15000 });
|
||||
return "✓ Gateway restarted";
|
||||
} else {
|
||||
execSync("systemctl --user restart ocp-proxy", { timeout: 15000 });
|
||||
return "✓ Proxy restarted";
|
||||
}
|
||||
} catch (e2) {
|
||||
return `✗ Restart failed: ${e2.message?.slice(0, 100)}. Run \`ocp restart\` on the server host manually.`;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function cmdLogs(args) {
|
||||
const parts = (args || "").trim().split(/\s+/);
|
||||
const n = parseInt(parts[0]) || 20;
|
||||
const level = parts[1] || "error";
|
||||
const d = await fetchJSON(`/logs?n=${n}&level=${level}`);
|
||||
if (!d.entries?.length) return `No ${level} log entries.`;
|
||||
return d.entries.map((e) => {
|
||||
if (e.raw) return e.raw.slice(0, 120);
|
||||
return `${(e.ts || "").slice(11, 19)} ${(e.level || "").toUpperCase()} ${e.event || "?"} ${e.model || ""}`;
|
||||
}).join("\n");
|
||||
}
|
||||
|
||||
function cmdHelp() {
|
||||
return `OCP Commands
|
||||
─────────────────────────────
|
||||
/ocp usage Plan usage & model stats
|
||||
/ocp status Quick overview
|
||||
/ocp health Proxy diagnostics
|
||||
/ocp settings View tunable settings
|
||||
/ocp settings <k> <v> Update a setting
|
||||
/ocp logs [N] [level] Recent logs (default: 20, error)
|
||||
/ocp models Available models
|
||||
/ocp sessions Active sessions
|
||||
/ocp clear Clear all sessions
|
||||
/ocp restart Restart proxy
|
||||
/ocp restart gateway Restart gateway
|
||||
/ocp restart all Restart both
|
||||
/ocp version Version & platform info
|
||||
/ocp test End-to-end proxy test`;
|
||||
}
|
||||
|
||||
// ── Plugin entry point ──────────────────────────────────────────────────
|
||||
|
||||
export default function (api) {
|
||||
console.log("[ocp] OCP plugin loading, registering /ocp command...");
|
||||
api.registerCommand({
|
||||
name: "ocp",
|
||||
description: "OpenClaw Proxy commands — usage, health, settings, logs, etc.",
|
||||
acceptsArgs: true,
|
||||
requireAuth: true,
|
||||
handler: async (ctx) => {
|
||||
const raw = (ctx.args || "").trim();
|
||||
const spaceIdx = raw.indexOf(" ");
|
||||
const subcmd = spaceIdx === -1 ? raw : raw.slice(0, spaceIdx);
|
||||
const subargs = spaceIdx === -1 ? "" : raw.slice(spaceIdx + 1).trim();
|
||||
|
||||
try {
|
||||
let text;
|
||||
switch (subcmd) {
|
||||
case "usage": text = await cmdUsage(); break;
|
||||
case "health": text = await cmdHealth(); break;
|
||||
case "status": text = await cmdStatus(); break;
|
||||
case "settings": text = await cmdSettings(subargs || null); break;
|
||||
case "models": text = await cmdModels(); break;
|
||||
case "sessions": text = await cmdSessions(); break;
|
||||
case "clear": text = await cmdClear(); break;
|
||||
case "restart": text = await cmdRestart(subargs); break;
|
||||
case "version": text = await cmdVersion(); break;
|
||||
case "test": text = await cmdTest(); break;
|
||||
case "logs": text = await cmdLogs(subargs); break;
|
||||
case "help": case "--help": case "-h": case "":
|
||||
text = cmdHelp(); break;
|
||||
default:
|
||||
text = `Unknown subcommand: ${subcmd}\n\n${cmdHelp()}`;
|
||||
}
|
||||
return { text: mono(text) };
|
||||
} catch (err) {
|
||||
return { text: `OCP error: ${err.message}` };
|
||||
}
|
||||
},
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
{
|
||||
"id": "ocp",
|
||||
"name": "OCP Commands",
|
||||
"description": "Slash commands for the OpenClaw Proxy — /ocp usage, /ocp settings, /ocp health, etc.",
|
||||
"version": "3.16.2",
|
||||
"configSchema": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"properties": {
|
||||
"proxyUrl": {
|
||||
"type": "string",
|
||||
"default": "http://127.0.0.1:3456",
|
||||
"description": "URL of the Claude proxy. Overridable via OCP_PROXY_URL or CLAUDE_PROXY_PORT env."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
{
|
||||
"name": "ocp",
|
||||
"version": "3.16.2",
|
||||
"description": "Slash commands for the OpenClaw Proxy",
|
||||
"main": "index.js",
|
||||
"type": "module",
|
||||
"keywords": ["openclaw", "plugin", "ocp", "proxy"],
|
||||
"license": "MIT",
|
||||
"openclaw": {
|
||||
"type": "plugin",
|
||||
"id": "ocp",
|
||||
"pluginManifest": "openclaw.plugin.json",
|
||||
"extensions": ["./index.js"]
|
||||
}
|
||||
}
|
||||
+9
-7
@@ -1,14 +1,16 @@
|
||||
{
|
||||
"name": "openclaw-claude-proxy",
|
||||
"version": "1.7.1",
|
||||
"description": "OpenAI-compatible proxy that routes requests through Claude CLI \u2014 use your Claude Pro/Max subscription as an OpenClaw model provider",
|
||||
"name": "open-claude-proxy",
|
||||
"version": "3.24.0",
|
||||
"description": "OCP (Open Claude Proxy) — use your Claude Pro/Max subscription as an OpenAI-compatible API for any IDE. Works with Cline, OpenCode, Aider, Continue.dev, OpenClaw, and more.",
|
||||
"type": "module",
|
||||
"bin": {
|
||||
"openclaw-claude-proxy": "./server.mjs"
|
||||
"openclaw-claude-proxy": "./server.mjs",
|
||||
"ocp": "./ocp"
|
||||
},
|
||||
"scripts": {
|
||||
"start": "node server.mjs",
|
||||
"setup": "node setup.mjs"
|
||||
"setup": "node setup.mjs",
|
||||
"test": "node test-features.mjs"
|
||||
},
|
||||
"keywords": [
|
||||
"openclaw",
|
||||
@@ -19,10 +21,10 @@
|
||||
],
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
"node": ">=22.5"
|
||||
},
|
||||
"repository": {
|
||||
"type": "git",
|
||||
"url": "https://github.com/dtzp555-max/openclaw-claude-proxy"
|
||||
"url": "https://github.com/dtzp555-max/ocp"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,311 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* scripts/doctor.mjs — OCP health & upgrade-readiness check.
|
||||
*
|
||||
* Usage:
|
||||
* ocp doctor human-readable PASS/WARN/FAIL
|
||||
* ocp doctor --json machine-readable JSON for AI agents + ocp update
|
||||
* ocp doctor --check oauth fast path: only OAuth check
|
||||
*
|
||||
* Exit codes:
|
||||
* 0 all PASS or WARN-only
|
||||
* 1 any FAIL
|
||||
*/
|
||||
import { readFileSync, existsSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
import { homedir } from "node:os";
|
||||
import { execSync } from "node:child_process";
|
||||
import { DEFAULT_PORT } from "../lib/constants.mjs";
|
||||
|
||||
const SCHEMA_VERSION = "1";
|
||||
|
||||
function semverParts(v) {
|
||||
const m = String(v).replace(/^v/, "").match(/^(\d+)\.(\d+)\.(\d+)/);
|
||||
if (!m) return null;
|
||||
return { major: +m[1], minor: +m[2], patch: +m[3] };
|
||||
}
|
||||
|
||||
function semverCompare(a, b) {
|
||||
const A = semverParts(a), B = semverParts(b);
|
||||
if (!A || !B) return 0;
|
||||
if (A.major !== B.major) return A.major - B.major;
|
||||
if (A.minor !== B.minor) return A.minor - B.minor;
|
||||
return A.patch - B.patch;
|
||||
}
|
||||
|
||||
export async function runDoctor(opts = {}) {
|
||||
const checks = [];
|
||||
const push = (id, level, message, extra = {}) =>
|
||||
checks.push({ id, level, message, ...extra });
|
||||
|
||||
// --- fast path: --check oauth ---
|
||||
if (opts.checkOnly === "oauth") {
|
||||
return runOauthOnly(opts, checks, push);
|
||||
}
|
||||
|
||||
// --- version detection ---
|
||||
const ocpDir = opts.ocpDir || join(homedir(), "ocp");
|
||||
let currentVersion = opts.mockVersion;
|
||||
if (!currentVersion) {
|
||||
try {
|
||||
const pkg = JSON.parse(readFileSync(join(ocpDir, "package.json"), "utf8"));
|
||||
currentVersion = `v${pkg.version}`;
|
||||
} catch {
|
||||
currentVersion = "unknown";
|
||||
}
|
||||
}
|
||||
// Resolve latest from origin/main (cheap: `git show origin/main:package.json`).
|
||||
// Falls back to current_version when network/git unavailable, so kind = noop instead
|
||||
// of recommending a downgrade against a stale hardcoded value.
|
||||
let latestVersion = opts.mockLatest;
|
||||
if (!latestVersion) {
|
||||
// Issue #173: `git show origin/main:...` reads the LOCALLY CACHED remote ref. Without a
|
||||
// fetch first, a machine that hasn't pulled since the last release sees latest == current
|
||||
// and reports noop — new releases were invisible everywhere except the machine that cut
|
||||
// the tag (live repro: Oracle VM, 2026-07-17). Fetch before comparing; on failure
|
||||
// (offline, auth, timeout) fall through to the cached ref — the pre-existing behavior.
|
||||
if (!opts.skipNetwork) {
|
||||
try {
|
||||
execSync(`git -C ${ocpDir} fetch --tags --quiet`, { stdio: ["pipe", "pipe", "pipe"], timeout: 15000 });
|
||||
} catch { /* offline → compare against cached origin/main, as before */ }
|
||||
}
|
||||
try {
|
||||
const out = execSync(`git -C ${ocpDir} show origin/main:package.json 2>/dev/null`, { stdio: ["pipe", "pipe", "pipe"] }).toString();
|
||||
const remotePkg = JSON.parse(out);
|
||||
latestVersion = `v${remotePkg.version}`;
|
||||
} catch {
|
||||
latestVersion = currentVersion;
|
||||
}
|
||||
}
|
||||
push("current_version", "PASS", `current=${currentVersion}`);
|
||||
|
||||
// --- from-version supported? ---
|
||||
const fromSupported = !!semverParts(currentVersion) && semverCompare(currentVersion, "v3.4.0") >= 0;
|
||||
push("from_version_supported", fromSupported ? "PASS" : "FAIL",
|
||||
fromSupported ? "≥ v3.4.0" : `${currentVersion} < v3.4.0; in-place upgrade not supported`);
|
||||
|
||||
// --- service health check (mockable) ---
|
||||
let healthOk = true, oauthOk = true;
|
||||
if (!opts.skipNetwork) {
|
||||
let health;
|
||||
if (opts.mockHealth !== undefined) {
|
||||
health = opts.mockHealth;
|
||||
} else {
|
||||
try {
|
||||
const port = process.env.CLAUDE_PROXY_PORT || String(DEFAULT_PORT);
|
||||
const out = execSync(`curl -sf --max-time 3 http://127.0.0.1:${port}/health`, { stdio: ["pipe", "pipe", "pipe"] }).toString();
|
||||
health = { status: 200, body: JSON.parse(out) };
|
||||
} catch (e) {
|
||||
health = { error: String(e.message || e) };
|
||||
}
|
||||
}
|
||||
if (health.error || health.status !== 200) {
|
||||
healthOk = false;
|
||||
push("service_running", "FAIL", `service unreachable: ${health.error || `status ${health.status}`}`);
|
||||
} else if (!health.body || typeof health.body !== "object") {
|
||||
healthOk = false;
|
||||
push("service_running", "FAIL", "service /health returned 200 but empty/non-JSON body");
|
||||
} else {
|
||||
push("service_running", "PASS", "service responding on /health");
|
||||
const authOk = health.body?.auth?.ok;
|
||||
if (!authOk) {
|
||||
oauthOk = false;
|
||||
push("oauth_ok", "FAIL", `auth.ok=false: ${health.body?.auth?.message || "unknown"}`);
|
||||
} else {
|
||||
push("oauth_ok", "PASS", "OAuth token valid");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- determine next_action.kind (priority: fresh_install > fix_service > fix_oauth > noop > update > upgrade) ---
|
||||
let kind;
|
||||
if (!fromSupported) {
|
||||
kind = "fresh_install";
|
||||
} else if (!opts.skipNetwork && !healthOk) {
|
||||
kind = "fix_service";
|
||||
} else if (!opts.skipNetwork && !oauthOk) {
|
||||
kind = "fix_oauth";
|
||||
} else {
|
||||
const cur = semverParts(currentVersion), lat = semverParts(latestVersion);
|
||||
if (!cur) {
|
||||
kind = "fresh_install";
|
||||
} else if (semverCompare(currentVersion, latestVersion) === 0) {
|
||||
kind = "noop";
|
||||
} else if (lat && cur.major === lat.major && cur.minor === lat.minor) {
|
||||
kind = "update";
|
||||
} else {
|
||||
kind = "upgrade";
|
||||
}
|
||||
}
|
||||
|
||||
// --- next_action shape ---
|
||||
let next_action;
|
||||
if (kind === "fresh_install") {
|
||||
next_action = {
|
||||
kind,
|
||||
human_required: ["claude auth login (only if OAuth becomes invalid after reinstall)"],
|
||||
ai_executable: [
|
||||
`launchctl bootout gui/$(id -u)/ai.openclaw.proxy 2>/dev/null || true`,
|
||||
`launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null || true`,
|
||||
`mv ${join(homedir(), ".ocp")} ${join(homedir(), ".ocp.backup-")}$(date +%s) 2>/dev/null || true`,
|
||||
`rm -rf ${ocpDir}`,
|
||||
`git clone https://github.com/dtzp555-max/ocp ${ocpDir}`,
|
||||
`cd ${ocpDir} && npm install --no-audit --no-fund && node setup.mjs`,
|
||||
`${ocpDir}/ocp doctor`
|
||||
],
|
||||
verify: "ocp doctor expects PASS on all checks"
|
||||
};
|
||||
} else if (kind === "noop") {
|
||||
next_action = { kind, human_required: [], ai_executable: [], verify: "already at latest" };
|
||||
} else if (kind === "fix_oauth") {
|
||||
next_action = {
|
||||
kind,
|
||||
human_required: [],
|
||||
ai_executable: [
|
||||
`cd "$(npm root -g)/@anthropic-ai/claude-code" && node install.cjs`,
|
||||
`launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null || true`,
|
||||
`launchctl bootstrap gui/$(id -u) ${join(homedir(), "Library", "LaunchAgents", "dev.ocp.proxy.plist")}`,
|
||||
`${ocpDir}/ocp doctor`
|
||||
],
|
||||
verify: "ocp doctor expects oauth_ok=PASS",
|
||||
reference: "~/.cc-rules/memory/learnings/ocp_claude_native_binary_postinstall.md"
|
||||
};
|
||||
} else if (kind === "fix_service") {
|
||||
next_action = {
|
||||
kind,
|
||||
human_required: [],
|
||||
ai_executable: [
|
||||
`launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null || true`,
|
||||
`launchctl bootstrap gui/$(id -u) ${join(homedir(), "Library", "LaunchAgents", "dev.ocp.proxy.plist")}`,
|
||||
`${ocpDir}/ocp doctor`
|
||||
],
|
||||
verify: "ocp doctor expects service_running=PASS"
|
||||
};
|
||||
} else {
|
||||
next_action = {
|
||||
kind,
|
||||
human_required: [],
|
||||
ai_executable: [`${ocpDir}/ocp update --yes`],
|
||||
verify: "ocp doctor expects PASS on all checks"
|
||||
};
|
||||
}
|
||||
|
||||
const fail_count = checks.filter(c => c.level === "FAIL").length;
|
||||
const warn_count = checks.filter(c => c.level === "WARN").length;
|
||||
return {
|
||||
schema_version: SCHEMA_VERSION,
|
||||
timestamp: new Date().toISOString(),
|
||||
ready_to_upgrade: fail_count === 0,
|
||||
current_version: currentVersion,
|
||||
latest_version: latestVersion,
|
||||
from_version_supported: fromSupported,
|
||||
fail_count,
|
||||
warn_count,
|
||||
checks,
|
||||
next_action
|
||||
};
|
||||
}
|
||||
|
||||
function runOauthOnly(opts, checks, push) {
|
||||
let healthOk = true, oauthOk = true;
|
||||
let health;
|
||||
if (opts.mockHealth !== undefined) {
|
||||
health = opts.mockHealth;
|
||||
} else {
|
||||
try {
|
||||
const port = process.env.CLAUDE_PROXY_PORT || String(DEFAULT_PORT);
|
||||
const out = execSync(`curl -sf --max-time 3 http://127.0.0.1:${port}/health`, { stdio: ["pipe", "pipe", "pipe"] }).toString();
|
||||
health = { status: 200, body: JSON.parse(out) };
|
||||
} catch (e) {
|
||||
health = { error: String(e.message || e) };
|
||||
}
|
||||
}
|
||||
|
||||
if (health.error || health.status !== 200) {
|
||||
healthOk = false;
|
||||
push("oauth_ok", "FAIL", `service unreachable: ${health.error || `status ${health.status}`}`);
|
||||
} else if (!health.body || typeof health.body !== "object") {
|
||||
healthOk = false;
|
||||
push("oauth_ok", "FAIL", "service /health returned 200 but empty/non-JSON body");
|
||||
} else if (!health.body?.auth?.ok) {
|
||||
oauthOk = false;
|
||||
push("oauth_ok", "FAIL", `auth.ok=false: ${health.body?.auth?.message || "unknown"}`);
|
||||
} else {
|
||||
push("oauth_ok", "PASS", "OAuth token valid");
|
||||
}
|
||||
|
||||
const kind = !healthOk ? "fix_service" : !oauthOk ? "fix_oauth" : "noop";
|
||||
|
||||
let next_action;
|
||||
const ocpDir = opts.ocpDir || join(homedir(), "ocp");
|
||||
if (kind === "noop") {
|
||||
next_action = { kind, human_required: [], ai_executable: [], verify: "OAuth healthy" };
|
||||
} else if (kind === "fix_oauth") {
|
||||
next_action = {
|
||||
kind,
|
||||
human_required: [],
|
||||
ai_executable: [
|
||||
`cd "$(npm root -g)/@anthropic-ai/claude-code" && node install.cjs`,
|
||||
`launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null || true`,
|
||||
`launchctl bootstrap gui/$(id -u) ${join(homedir(), "Library", "LaunchAgents", "dev.ocp.proxy.plist")}`,
|
||||
`${ocpDir}/ocp doctor --check oauth`
|
||||
],
|
||||
verify: "ocp doctor --check oauth expects PASS",
|
||||
reference: "~/.cc-rules/memory/learnings/ocp_claude_native_binary_postinstall.md"
|
||||
};
|
||||
} else {
|
||||
next_action = {
|
||||
kind,
|
||||
human_required: [],
|
||||
ai_executable: [
|
||||
`launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null || true`,
|
||||
`launchctl bootstrap gui/$(id -u) ${join(homedir(), "Library", "LaunchAgents", "dev.ocp.proxy.plist")}`,
|
||||
`${ocpDir}/ocp doctor --check oauth`
|
||||
],
|
||||
verify: "ocp doctor --check oauth expects service_running=PASS"
|
||||
};
|
||||
}
|
||||
|
||||
const fail_count = checks.filter(c => c.level === "FAIL").length;
|
||||
// "skipped" = --check oauth fast path intentionally omits version detection.
|
||||
// AI agents should NOT semver-compare against current_version/latest_version when
|
||||
// either equals "skipped"; the full path provides those fields when needed.
|
||||
return {
|
||||
schema_version: SCHEMA_VERSION,
|
||||
timestamp: new Date().toISOString(),
|
||||
ready_to_upgrade: fail_count === 0,
|
||||
current_version: opts.mockVersion || "skipped",
|
||||
latest_version: opts.mockLatest || "skipped",
|
||||
from_version_supported: true,
|
||||
fail_count,
|
||||
warn_count: 0,
|
||||
checks,
|
||||
next_action
|
||||
};
|
||||
}
|
||||
|
||||
// CLI entrypoint — use fileURLToPath + realpath to handle symlinked install paths
|
||||
// (e.g. /tmp/ → /private/tmp/ on macOS would otherwise miss the guard).
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { realpathSync } from "node:fs";
|
||||
function _isMain() {
|
||||
if (!process.argv[1]) return false;
|
||||
try {
|
||||
return realpathSync(fileURLToPath(import.meta.url)) === realpathSync(process.argv[1]);
|
||||
} catch { return false; }
|
||||
}
|
||||
if (_isMain()) {
|
||||
const wantJson = process.argv.includes("--json");
|
||||
const checkIdx = process.argv.indexOf("--check");
|
||||
const checkOnly = checkIdx !== -1 ? process.argv[checkIdx + 1] : undefined;
|
||||
const result = await runDoctor({ checkOnly });
|
||||
if (wantJson) {
|
||||
console.log(JSON.stringify(result, null, 2));
|
||||
} else {
|
||||
console.log(`OCP doctor — ${result.current_version} → ${result.latest_version}`);
|
||||
for (const c of result.checks) console.log(` [${c.level}] ${c.id}: ${c.message}`);
|
||||
console.log(`\nSummary: ${result.fail_count} FAIL, ${result.warn_count} WARN`);
|
||||
console.log(`Next action: ${result.next_action.kind}`);
|
||||
}
|
||||
process.exit(result.fail_count === 0 ? 0 : 1);
|
||||
}
|
||||
Executable
+132
@@ -0,0 +1,132 @@
|
||||
#!/bin/bash
|
||||
# One-shot field-evidence gatherer for OCP v3.12.0 SSE heartbeat.
|
||||
# Scheduled by ~/Library/LaunchAgents/dev.ocp.heartbeat-check.plist to fire
|
||||
# once at 2026-05-02 09:00 Australia/Brisbane. Gathers evidence from local
|
||||
# OCP logs + GitHub issue #47 + repo issue search, posts a summary comment
|
||||
# on #47, and exits. Does NOT open PRs or change code — the maintainer
|
||||
# decides after reading the summary.
|
||||
#
|
||||
# Dry-run: ./heartbeat-field-check.sh --dry-run (prints summary, skips post)
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO="dtzp555-max/ocp"
|
||||
SHIP_DATE="2026-04-25"
|
||||
# Baseline captured at script-install time so internal testing entries from
|
||||
# Phase 3 verification (~5 entries from 2026-04-25T00:00–00:48Z) don't get
|
||||
# counted as field evidence. Any heartbeat_active log entry with ts >= this
|
||||
# timestamp is treated as a real opt-in.
|
||||
BASELINE_TS="2026-04-25T01:00:00Z"
|
||||
PROXY_LOG="$HOME/ocp/logs/proxy.log"
|
||||
OUT_DIR="$HOME/ocp/logs"
|
||||
SELF_LOG="$OUT_DIR/heartbeat-field-check-$(date +%Y-%m-%d).log"
|
||||
DRY_RUN=0
|
||||
[ "${1:-}" = "--dry-run" ] && DRY_RUN=1
|
||||
|
||||
mkdir -p "$OUT_DIR"
|
||||
|
||||
exec > >(tee -a "$SELF_LOG") 2>&1
|
||||
|
||||
echo "=== heartbeat field-evidence check: $(date -u +%Y-%m-%dT%H:%M:%SZ) (dry_run=$DRY_RUN) ==="
|
||||
|
||||
# ── signal 1: local proxy log ─────────────────────────────────────────────
|
||||
if [ -r "$PROXY_LOG" ]; then
|
||||
# Only count entries with ts >= BASELINE_TS (string sort works on RFC3339)
|
||||
HEARTBEAT_COUNT=$(grep '"event":"heartbeat_active"' "$PROXY_LOG" 2>/dev/null \
|
||||
| awk -v base="$BASELINE_TS" '
|
||||
match($0, /"ts":"[^"]+"/) {
|
||||
ts = substr($0, RSTART+6, RLENGTH-7);
|
||||
if (ts >= base) c++
|
||||
}
|
||||
END { print c+0 }')
|
||||
else
|
||||
HEARTBEAT_COUNT=0
|
||||
fi
|
||||
echo "signal 1 — heartbeat_active log entries since $BASELINE_TS: $HEARTBEAT_COUNT"
|
||||
|
||||
# ── signal 2: comments on #47 since ship ──────────────────────────────────
|
||||
NEW_47_JSON="/tmp/ocp-47-new-comments-$$.json"
|
||||
gh issue view 47 --repo "$REPO" --json comments \
|
||||
--jq '[.comments[] | select(.createdAt >= "'"$SHIP_DATE"'T00:00:00Z")]' \
|
||||
> "$NEW_47_JSON" 2>/dev/null || echo "[]" > "$NEW_47_JSON"
|
||||
NEW_COMMENTS=$(jq 'length' "$NEW_47_JSON")
|
||||
echo "signal 2 — new comments on #47 since $SHIP_DATE: $NEW_COMMENTS"
|
||||
|
||||
# Build a compact, human-readable excerpt for the summary body
|
||||
NEW_47_EXCERPT=""
|
||||
if [ "$NEW_COMMENTS" -gt 0 ]; then
|
||||
NEW_47_EXCERPT=$(jq -r '.[] | "- **@\(.author.login)** (\(.createdAt)): " + (.body | gsub("\r"; "") | split("\n")[0])[:180]' "$NEW_47_JSON")
|
||||
fi
|
||||
|
||||
# ── signal 3: other heartbeat-related issues since ship ──────────────────
|
||||
OTHER_ISSUES_JSON="/tmp/ocp-heartbeat-issues-$$.json"
|
||||
gh search issues "repo:$REPO heartbeat" --json number,title,state,createdAt --limit 30 \
|
||||
--jq '[.[] | select(.createdAt >= "'"$SHIP_DATE"'T00:00:00Z" and .number != 47 and .number != 48)]' \
|
||||
> "$OTHER_ISSUES_JSON" 2>/dev/null || echo "[]" > "$OTHER_ISSUES_JSON"
|
||||
OTHER_ISSUES=$(jq 'length' "$OTHER_ISSUES_JSON")
|
||||
echo "signal 3 — other heartbeat-related issues since ship: $OTHER_ISSUES"
|
||||
|
||||
OTHER_ISSUES_EXCERPT=""
|
||||
if [ "$OTHER_ISSUES" -gt 0 ]; then
|
||||
OTHER_ISSUES_EXCERPT=$(jq -r '.[] | "- #\(.number) [\(.state)] \(.title)"' "$OTHER_ISSUES_JSON")
|
||||
fi
|
||||
|
||||
# ── compose summary ──────────────────────────────────────────────────────
|
||||
BODY_FILE="/tmp/ocp-47-summary-$$.md"
|
||||
{
|
||||
echo "### Automated 7-day field-evidence check (v3.12.0)"
|
||||
echo
|
||||
echo "_Triggered by a local launchd scheduled task on the maintainer's rig at $(date -u +%Y-%m-%dT%H:%M:%SZ)._"
|
||||
echo
|
||||
echo "| Signal | Count |"
|
||||
echo "|---|---|"
|
||||
echo "| \`heartbeat_active\` log entries on prod rig (since baseline $BASELINE_TS) | $HEARTBEAT_COUNT |"
|
||||
echo "| New comments on #47 since $SHIP_DATE | $NEW_COMMENTS |"
|
||||
echo "| Other heartbeat-related issues filed since $SHIP_DATE | $OTHER_ISSUES |"
|
||||
echo
|
||||
if [ -n "$NEW_47_EXCERPT" ]; then
|
||||
echo "**New #47 comments (first line each):**"
|
||||
echo
|
||||
echo "$NEW_47_EXCERPT"
|
||||
echo
|
||||
fi
|
||||
if [ -n "$OTHER_ISSUES_EXCERPT" ]; then
|
||||
echo "**Other heartbeat-related issues:**"
|
||||
echo
|
||||
echo "$OTHER_ISSUES_EXCERPT"
|
||||
echo
|
||||
fi
|
||||
echo "**Decision guidance for maintainer (manual):**"
|
||||
echo
|
||||
echo "- If any of the above indicate a **crash report** on \`: keepalive\` comment frames → leave default at \`0\` and file a \`CLAUDE_HEARTBEAT_FORMAT=empty-delta\` follow-up issue (spec \`§D2\` fallback plan)."
|
||||
echo "- If there is at least one **opt-in confirmation** (a user reports \`CLAUDE_HEARTBEAT_INTERVAL\` fixed their timeout issue) and no crash reports → consider opening a PR for v3.13.0 flipping the default to \`30000\`, following the same ALIGNMENT + independent-reviewer + release-kit discipline as PR #49."
|
||||
echo "- If all three signals are zero → extend the soak window or close this follow-up as \"no field evidence.\""
|
||||
echo
|
||||
echo "This bot does not open PRs or change code. The maintainer reviews and acts."
|
||||
} > "$BODY_FILE"
|
||||
|
||||
echo "--- summary preview ---"
|
||||
cat "$BODY_FILE"
|
||||
echo "--- end preview ---"
|
||||
|
||||
# ── post (unless dry-run) ────────────────────────────────────────────────
|
||||
if [ "$DRY_RUN" -eq 1 ]; then
|
||||
echo "DRY RUN — skipping gh issue comment"
|
||||
else
|
||||
gh issue comment 47 --repo "$REPO" --body-file "$BODY_FILE" && echo "comment posted on #47"
|
||||
fi
|
||||
|
||||
# ── cleanup + self-disable so the plist doesn't linger loaded forever ────
|
||||
rm -f "$NEW_47_JSON" "$OTHER_ISSUES_JSON" "$BODY_FILE"
|
||||
|
||||
if [ "$DRY_RUN" -eq 0 ]; then
|
||||
# Unload + remove the plist so this never fires again
|
||||
PLIST="$HOME/Library/LaunchAgents/dev.ocp.heartbeat-check.plist"
|
||||
if [ -f "$PLIST" ]; then
|
||||
launchctl bootout "gui/$(id -u)" "$PLIST" 2>/dev/null || launchctl unload "$PLIST" 2>/dev/null || true
|
||||
rm -f "$PLIST"
|
||||
echo "self-disabled: removed $PLIST"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "=== done ==="
|
||||
@@ -0,0 +1,101 @@
|
||||
// scripts/lib/plist-merge.mjs
|
||||
//
|
||||
// Preserves user-customised env vars when setup.mjs rewrites the unit file.
|
||||
//
|
||||
// Rule:
|
||||
// - keys present in NEW template → template value wins (template is source of truth)
|
||||
// - keys ONLY in EXISTING (not in template) → preserved verbatim
|
||||
//
|
||||
// No new dependencies — regex-based, plist <key>X</key><string>Y</string> shape
|
||||
// is stable enough for our hand-written templates in setup.mjs.
|
||||
//
|
||||
// SECURITY DENYLIST (A4): keys that must NEVER be carried into a service unit, even when a
|
||||
// prior unit already contained them. OCP's key store honors OCP_DIR_OVERRIDE only when
|
||||
// NODE_ENV === "test" (keys.mjs). If BOTH somehow reached a daemon's environment, the server
|
||||
// would open a scratch/empty key store instead of ~/.ocp/ocp.db — in AUTH_MODE=multi a silent
|
||||
// total auth outage. The preservation rule below ("keys only in EXISTING are kept verbatim")
|
||||
// is exactly a vector for that: a unit that once carried these test-only vars would otherwise
|
||||
// survive every setup re-run. So we strip them from the preserved set unconditionally. This is
|
||||
// defense-in-depth: setup.mjs's own template never injects them, so the only way they enter is
|
||||
// preservation, and this closes it. (The residual path — a hand-rolled `node server.mjs` with
|
||||
// both vars exported — is out of any launcher's reach; keys.mjs's loud "NOT the default" log is
|
||||
// the backstop there.)
|
||||
export const NEVER_PRESERVE = new Set(["NODE_ENV", "OCP_DIR_OVERRIDE"]);
|
||||
|
||||
// Note: setup.mjs XML-escapes all injected values before writing (via xmlEscape()),
|
||||
// so raw `<` / `>` / `&` never appear in plist <string> bodies — the [^<]* regex below is safe.
|
||||
const PLIST_KV_RE = /<key>([^<]+)<\/key>\s*<string>([^<]*)<\/string>/g;
|
||||
|
||||
export function parsePlistEnv(plistContent) {
|
||||
if (!plistContent) return {};
|
||||
if (Buffer.isBuffer(plistContent)) plistContent = plistContent.toString("utf8");
|
||||
// Restrict to the EnvironmentVariables dict to avoid catching Label, etc.
|
||||
const envBlock = plistContent.match(/<key>EnvironmentVariables<\/key>\s*<dict>([\s\S]*?)<\/dict>/);
|
||||
if (!envBlock) return {};
|
||||
const out = {};
|
||||
let m;
|
||||
PLIST_KV_RE.lastIndex = 0;
|
||||
while ((m = PLIST_KV_RE.exec(envBlock[1])) !== null) {
|
||||
out[m[1]] = m[2];
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function mergePlistEnv(existing, template) {
|
||||
if (!existing) return template;
|
||||
const existingEnv = parsePlistEnv(existing);
|
||||
const templateEnv = parsePlistEnv(template);
|
||||
const KNOWN = new Set(Object.keys(templateEnv));
|
||||
|
||||
const preserved = {};
|
||||
for (const [k, v] of Object.entries(existingEnv)) {
|
||||
if (!KNOWN.has(k) && !NEVER_PRESERVE.has(k)) preserved[k] = v;
|
||||
}
|
||||
if (Object.keys(preserved).length === 0) return template;
|
||||
|
||||
const lines = Object.entries(preserved)
|
||||
.map(([k, v]) => ` <key>${k}</key>\n <string>${v}</string>`)
|
||||
.join("\n");
|
||||
|
||||
// Inject before the closing </dict> of EnvironmentVariables
|
||||
return template.replace(
|
||||
/(<key>EnvironmentVariables<\/key>\s*<dict>[\s\S]*?)(\n\s*<\/dict>)/,
|
||||
`$1\n${lines}$2`
|
||||
);
|
||||
}
|
||||
|
||||
const SYSTEMD_KV_RE = /^Environment=([^=]+)=(.*)$/gm;
|
||||
|
||||
export function parseSystemdEnv(serviceContent) {
|
||||
if (!serviceContent) return {};
|
||||
if (Buffer.isBuffer(serviceContent)) serviceContent = serviceContent.toString("utf8");
|
||||
const out = {};
|
||||
let m;
|
||||
SYSTEMD_KV_RE.lastIndex = 0;
|
||||
while ((m = SYSTEMD_KV_RE.exec(serviceContent)) !== null) {
|
||||
out[m[1]] = m[2];
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function mergeSystemdEnv(existing, template) {
|
||||
if (!existing) return template;
|
||||
const existingEnv = parseSystemdEnv(existing);
|
||||
const templateEnv = parseSystemdEnv(template);
|
||||
const KNOWN = new Set(Object.keys(templateEnv));
|
||||
|
||||
const preservedLines = Object.entries(existingEnv)
|
||||
.filter(([k]) => !KNOWN.has(k) && !NEVER_PRESERVE.has(k))
|
||||
.map(([k, v]) => `Environment=${k}=${v}`);
|
||||
if (preservedLines.length === 0) return template;
|
||||
|
||||
// Guard: if template has no Environment= anchor, cannot inject — return template as-is.
|
||||
// (In practice the OCP systemd template always has Environment= lines.)
|
||||
if (!/^Environment=/m.test(template)) return template;
|
||||
|
||||
// Inject after the last existing Environment= line in the template
|
||||
return template.replace(
|
||||
/(^Environment=[^\n]+\n)((?!Environment=).*$)/ms,
|
||||
`$1${preservedLines.join("\n")}\n$2`
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,130 @@
|
||||
import { mkdirSync, writeFileSync, readFileSync, copyFileSync, existsSync, readdirSync, statSync, rmSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
|
||||
export function writeSnapshot({ homeDir, fromCommit, fromVersion, toVersion, extraFiles = [] }) {
|
||||
const ts = formatSnapshotTimestamp(new Date());
|
||||
const root = join(homeDir, ".ocp", `upgrade-snapshot-${ts}`);
|
||||
mkdirSync(root, { recursive: true });
|
||||
|
||||
// Standard manifest files
|
||||
writeFileSync(join(root, "from-commit.txt"), fromCommit + "\n");
|
||||
writeFileSync(join(root, "from-version.txt"), fromVersion + "\n");
|
||||
writeFileSync(join(root, "to-version.txt"), toVersion + "\n");
|
||||
|
||||
// Optional captures (best-effort, never fatal)
|
||||
const tryCopy = (src, dst) => {
|
||||
try {
|
||||
if (existsSync(src)) copyFileSync(src, dst);
|
||||
} catch (err) {
|
||||
console.error(`[snapshot] warn: could not copy ${src} (${err.code || err.message})`);
|
||||
}
|
||||
};
|
||||
tryCopy(join(homeDir, "Library", "LaunchAgents", "dev.ocp.proxy.plist"), join(root, "plist"));
|
||||
tryCopy(join(homeDir, ".config", "systemd", "user", "ocp-proxy.service"), join(root, "service"));
|
||||
tryCopy(join(homeDir, ".ocp", "ocp.db"), join(root, "db.bak"));
|
||||
tryCopy(join(homeDir, ".ocp", "admin-key"), join(root, "admin-key"));
|
||||
tryCopy(join(homeDir, ".openclaw", "openclaw.json"), join(root, "openclaw.json"));
|
||||
|
||||
for (const { src, name } of extraFiles) tryCopy(src, join(root, name));
|
||||
|
||||
return root;
|
||||
}
|
||||
|
||||
export function readSnapshot(snapshotPath) {
|
||||
const read = (n) => {
|
||||
try { return readFileSync(join(snapshotPath, n), "utf8").trim(); } catch { return null; }
|
||||
};
|
||||
return {
|
||||
path: snapshotPath,
|
||||
fromCommit: read("from-commit.txt"),
|
||||
fromVersion: read("from-version.txt"),
|
||||
toVersion: read("to-version.txt")
|
||||
};
|
||||
}
|
||||
|
||||
export function listSnapshots(homeDir) {
|
||||
const root = join(homeDir, ".ocp");
|
||||
if (!existsSync(root)) return [];
|
||||
return readdirSync(root)
|
||||
.filter(name => name.startsWith("upgrade-snapshot-"))
|
||||
.map(name => ({ name, path: join(root, name), mtime: statSync(join(root, name)).mtimeMs }))
|
||||
.sort((a, b) => {
|
||||
const chronological = parseSnapshotTimestamp(a.name) - parseSnapshotTimestamp(b.name);
|
||||
return chronological || a.name.localeCompare(b.name);
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Garbage-collect old upgrade snapshots.
|
||||
*
|
||||
* Retention rule (a snapshot is KEPT if any of these is true):
|
||||
* - It is among the last `keepCount` snapshots (sorted oldest→newest)
|
||||
* - Its timestamp is within `keepDays` of `now`
|
||||
* - It is the single most-recent snapshot (always-keep safety net)
|
||||
*
|
||||
* @param {string} homeDir - Root containing ~/.ocp/
|
||||
* @param {object} opts
|
||||
* @param {number} [opts.keepCount=5] - Minimum count to keep
|
||||
* @param {number} [opts.keepDays=30] - Keep snapshots newer than N days
|
||||
* @param {boolean} [opts.dryRun=false] - If true, report plan but don't delete
|
||||
* @param {Date} [opts.now=new Date()] - Override clock for testing
|
||||
* @returns {{kept: Array, removed: Array, dryRun: boolean}}
|
||||
*/
|
||||
export function gcSnapshots(homeDir, opts = {}) {
|
||||
const keepCount = opts.keepCount ?? 5;
|
||||
const keepDays = opts.keepDays ?? 30;
|
||||
const dryRun = !!opts.dryRun;
|
||||
const now = opts.now || new Date();
|
||||
|
||||
const all = listSnapshots(homeDir); // sorted oldest→newest
|
||||
if (all.length === 0) return { kept: [], removed: [], dryRun };
|
||||
if (all.length === 1) return { kept: all, removed: [], dryRun }; // always keep most recent
|
||||
|
||||
const cutoffMs = now.getTime() - keepDays * 24 * 60 * 60 * 1000;
|
||||
const lastN = new Set(all.slice(-keepCount).map(s => s.path));
|
||||
|
||||
const kept = [], removed = [];
|
||||
for (let i = 0; i < all.length; i++) {
|
||||
const s = all[i];
|
||||
const isMostRecent = i === all.length - 1;
|
||||
const isInLastN = lastN.has(s.path);
|
||||
const isWithinDays = parseSnapshotTimestamp(s.name) >= cutoffMs;
|
||||
if (isMostRecent || isInLastN || isWithinDays) {
|
||||
kept.push(s);
|
||||
} else {
|
||||
removed.push(s);
|
||||
}
|
||||
}
|
||||
|
||||
if (!dryRun) {
|
||||
for (const s of removed) {
|
||||
try {
|
||||
rmSync(s.path, { recursive: true, force: true });
|
||||
} catch (err) {
|
||||
console.error(`[snapshot] warn: could not remove ${s.path} (${err.code || err.message})`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return { kept, removed, dryRun };
|
||||
}
|
||||
|
||||
function parseSnapshotTimestamp(name) {
|
||||
// Both legacy ISO names and Windows-safe names are supported.
|
||||
// upgrade-snapshot-2026-05-11T08:30:00Z → epoch ms
|
||||
// upgrade-snapshot-2026-05-11T08-30-00Z → epoch ms
|
||||
const m = name.match(/upgrade-snapshot-(.+)$/);
|
||||
if (!m) return 0;
|
||||
const raw = m[1];
|
||||
const t = Date.parse(raw);
|
||||
if (Number.isFinite(t)) return t;
|
||||
const iso = raw.replace(/(T\d{2})-(\d{2})-(\d{2})Z$/, "$1:$2:$3Z");
|
||||
const portable = Date.parse(iso);
|
||||
return Number.isFinite(portable) ? portable : 0;
|
||||
}
|
||||
|
||||
function formatSnapshotTimestamp(date) {
|
||||
// Windows forbids ':' in directory names. Replacing only the time separators
|
||||
// preserves chronological lexical order and keeps the timestamp readable.
|
||||
return date.toISOString().replace(/\.\d+Z$/, "Z").replace(/:/g, "-");
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
#!/usr/bin/env node
|
||||
// Idempotently sync OCP's claude-local provider models into OpenClaw's registry.
|
||||
// Only touches:
|
||||
// - config.models.providers["claude-local"].models
|
||||
// - config.agents.defaults.models["claude-local/*"] keys
|
||||
// All other fields and providers are preserved.
|
||||
|
||||
import { readFileSync, writeFileSync, existsSync, copyFileSync } from "node:fs";
|
||||
import { join, dirname } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { homedir } from "node:os";
|
||||
import { DEFAULT_PORT, LOCAL_HOST, OPENAI_API_BASE } from "../lib/constants.mjs";
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const REPO_ROOT = join(__dirname, "..");
|
||||
const OPENCLAW_CONFIG = join(homedir(), ".openclaw", "openclaw.json");
|
||||
const PROVIDER_NAME = "claude-local";
|
||||
const QUIET = process.argv.includes("--quiet");
|
||||
|
||||
function log(msg) { if (!QUIET) console.log(` ✓ ${msg}`); }
|
||||
function warn(msg) { console.warn(` ⚠ ${msg}`); }
|
||||
|
||||
if (!existsSync(OPENCLAW_CONFIG)) {
|
||||
log(`OpenClaw not installed at ${OPENCLAW_CONFIG} — skipping (this is fine for non-OpenClaw users)`);
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const modelsConfig = JSON.parse(readFileSync(join(REPO_ROOT, "models.json"), "utf-8"));
|
||||
const desiredModels = modelsConfig.models.map(m => ({
|
||||
id: m.id,
|
||||
name: m.openclawName,
|
||||
reasoning: m.reasoning,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: m.contextWindow,
|
||||
maxTokens: m.maxTokens,
|
||||
}));
|
||||
const desiredAliases = Object.fromEntries(
|
||||
modelsConfig.models.map(m => [`${PROVIDER_NAME}/${m.id}`, { alias: m.displayName }])
|
||||
);
|
||||
|
||||
const config = JSON.parse(readFileSync(OPENCLAW_CONFIG, "utf-8"));
|
||||
|
||||
// Compute diff before writing
|
||||
const existingModels = config?.models?.providers?.[PROVIDER_NAME]?.models ?? [];
|
||||
const existingIds = new Set(existingModels.map(m => m.id));
|
||||
const desiredIds = new Set(desiredModels.map(m => m.id));
|
||||
const added = [...desiredIds].filter(id => !existingIds.has(id));
|
||||
const removed = [...existingIds].filter(id => !desiredIds.has(id));
|
||||
|
||||
if (added.length === 0 && removed.length === 0 && existingModels.length === desiredModels.length) {
|
||||
// Check deep equality too in case names/maxTokens changed
|
||||
const changed = desiredModels.some((d) => {
|
||||
const e = existingModels.find(x => x.id === d.id);
|
||||
return !e || e.name !== d.name || e.maxTokens !== d.maxTokens;
|
||||
});
|
||||
if (!changed) {
|
||||
log("OpenClaw registry already in sync");
|
||||
process.exit(0);
|
||||
}
|
||||
}
|
||||
|
||||
// Backup
|
||||
const backupPath = `${OPENCLAW_CONFIG}.bak.${Date.now()}`;
|
||||
copyFileSync(OPENCLAW_CONFIG, backupPath);
|
||||
log(`Backed up to ${backupPath}`);
|
||||
|
||||
// Surgical patch: only touch claude-local provider and claude-local/* aliases
|
||||
if (!config.models) config.models = {};
|
||||
if (!config.models.providers) config.models.providers = {};
|
||||
if (!config.models.providers[PROVIDER_NAME]) {
|
||||
// First-time registration
|
||||
config.models.providers[PROVIDER_NAME] = {
|
||||
baseUrl: `http://${LOCAL_HOST}:${DEFAULT_PORT}${OPENAI_API_BASE}`,
|
||||
api: "openai-completions",
|
||||
authHeader: false,
|
||||
models: desiredModels,
|
||||
};
|
||||
} else {
|
||||
// Update only the models array; leave baseUrl/api/authHeader untouched (user may have customized port)
|
||||
config.models.providers[PROVIDER_NAME].models = desiredModels;
|
||||
}
|
||||
|
||||
if (!config.agents) config.agents = {};
|
||||
if (!config.agents.defaults) config.agents.defaults = {};
|
||||
if (!config.agents.defaults.models) config.agents.defaults.models = {};
|
||||
|
||||
// Remove stale claude-local/* aliases, then add desired ones
|
||||
for (const key of Object.keys(config.agents.defaults.models)) {
|
||||
if (key.startsWith(`${PROVIDER_NAME}/`)) delete config.agents.defaults.models[key];
|
||||
}
|
||||
Object.assign(config.agents.defaults.models, desiredAliases);
|
||||
|
||||
writeFileSync(OPENCLAW_CONFIG, JSON.stringify(config, null, 2) + "\n");
|
||||
|
||||
if (added.length > 0) log(`Added: ${added.join(", ")}`);
|
||||
if (removed.length > 0) log(`Removed (no longer in models.json): ${removed.join(", ")}`);
|
||||
log(`OpenClaw registry synced: ${desiredModels.length} models registered`);
|
||||
@@ -0,0 +1,365 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* scripts/upgrade.mjs — OCP unified upgrade dispatcher.
|
||||
*
|
||||
* Paths:
|
||||
* noop current == latest, exit 0
|
||||
* light same major.minor, patch bump only (existing fast path; delegated to bash)
|
||||
* full cross-minor (snapshot + setup.mjs + post-flight)
|
||||
* fresh_install from-version < v3.4.0 (--yes required for non-interactive)
|
||||
* rollback restore from snapshot
|
||||
*/
|
||||
import { runDoctor } from "./doctor.mjs";
|
||||
import { execSync } from "node:child_process";
|
||||
import { homedir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { existsSync, copyFileSync } from "node:fs";
|
||||
import { writeSnapshot, listSnapshots, readSnapshot, gcSnapshots } from "./lib/snapshot.mjs";
|
||||
import { DEFAULT_PORT } from "../lib/constants.mjs";
|
||||
|
||||
// Post-flight acceptance predicate (issue #173). A health probe passes ONLY when the server
|
||||
// is authed AND actually serving the TARGET version. auth.ok alone is not enough: a stale
|
||||
// process holding the port answers auth.ok=true while still running the OLD code — exactly
|
||||
// what a nohup-fallback orphan did on 2026-07-17 (upgrade "succeeded", /health kept serving
|
||||
// 3.21.1). Comparing /health.version to the checkout target catches orphan-holds-port,
|
||||
// restart-didn't-take, and wrong-unit-restarted alike. `target` tolerates a leading "v"
|
||||
// (doctor reports "v3.22.1"; /health reports "3.22.1"); an empty/unknown target degrades to
|
||||
// the old auth-only check rather than blocking an otherwise-good upgrade.
|
||||
export function postFlightOk(body, target) {
|
||||
if (body?.auth?.ok !== true) return false;
|
||||
const want = String(target || "").replace(/^v/, "");
|
||||
return !want || body?.version === want;
|
||||
}
|
||||
|
||||
export async function runUpgrade(opts = {}) {
|
||||
const dryRun = !!opts.dryRun;
|
||||
const yes = !!opts.yes;
|
||||
// yes is reserved for Bundle 3 (fresh-install / rollback interactive gate); not used in upgrade-path here.
|
||||
const plan = [];
|
||||
|
||||
// --- rollback path (no doctor needed; snapshot is authoritative) ---
|
||||
if (opts.rollback) {
|
||||
return await runRollback(opts);
|
||||
}
|
||||
|
||||
// --- doctor pre-flight ---
|
||||
const doctor = opts.mockDoctor || await runDoctor();
|
||||
if (!doctor.ready_to_upgrade && doctor.next_action.kind !== "fresh_install") {
|
||||
throw new Error(`doctor FAIL: ${doctor.next_action.kind} (run "ocp doctor" for details)`);
|
||||
}
|
||||
|
||||
const kind = doctor.next_action.kind;
|
||||
plan.push(`[doctor] from=${doctor.current_version} to=${doctor.latest_version} kind=${kind}`);
|
||||
|
||||
// --- noop ---
|
||||
if (kind === "noop") {
|
||||
plan.push(`[noop] already at latest (${doctor.latest_version})`);
|
||||
return { path: "noop", executed: true, changed: false, plan };
|
||||
}
|
||||
|
||||
// --- dry-run early exit ---
|
||||
if (dryRun) {
|
||||
plan.push(`[plan] would proceed with ${kind} path`);
|
||||
if (kind === "upgrade") {
|
||||
plan.push(`[plan] phase 1: snapshot to ~/.ocp/upgrade-snapshot-<ts>/`);
|
||||
plan.push(`[plan] phase 2: git checkout ${doctor.latest_version} && npm install`);
|
||||
plan.push(`[plan] phase 3: node setup.mjs`);
|
||||
plan.push(`[plan] phase 4: launchctl bootout/bootstrap`);
|
||||
plan.push(`[plan] phase 5: post-flight /health + /v1/models`);
|
||||
} else if (kind === "update") {
|
||||
plan.push(`[plan] light path: git pull + npm install + restart`);
|
||||
} else if (kind === "fresh_install") {
|
||||
plan.push(`[plan] fresh-install ai_executable[]:`);
|
||||
for (const cmd of doctor.next_action.ai_executable) plan.push(` - ${cmd}`);
|
||||
}
|
||||
return { path: kind, executed: false, plan };
|
||||
}
|
||||
|
||||
// --- non-dry-run paths ---
|
||||
if (kind === "update") {
|
||||
return { path: "update", executed: true, changed: true, plan: [...plan, "[light] delegated to bash cmd_update existing logic"] };
|
||||
}
|
||||
|
||||
if (kind === "upgrade") {
|
||||
return await runFullUpgrade({ doctor, opts });
|
||||
}
|
||||
|
||||
if (kind === "fresh_install") {
|
||||
return await runFreshInstall({ doctor, opts });
|
||||
}
|
||||
|
||||
throw new Error(`path ${kind} not yet implemented`);
|
||||
}
|
||||
|
||||
async function runFullUpgrade({ doctor, opts }) {
|
||||
const phases = [];
|
||||
let snapshotPath = null;
|
||||
const exec = (cmd, label) => {
|
||||
if (opts.mockExec) {
|
||||
phases.push({ name: label, cmd, status: "skipped-mock" });
|
||||
return "";
|
||||
}
|
||||
try {
|
||||
const out = execSync(cmd, { stdio: ["pipe", "pipe", "pipe"] }).toString();
|
||||
phases.push({ name: label, cmd, status: "ok" });
|
||||
return out;
|
||||
} catch (err) {
|
||||
const detail = err.stderr?.toString().trim();
|
||||
phases.push({ name: label, cmd, status: "fail", stderr: detail });
|
||||
throw Object.assign(
|
||||
new Error(`phase ${label} failed: ${detail || err.message}`),
|
||||
{ phases, cmd }
|
||||
);
|
||||
}
|
||||
};
|
||||
const ocpDir = opts.ocpDir || join(homedir(), "ocp");
|
||||
|
||||
try {
|
||||
// phase 1: pre-flight (doctor already passed; just record)
|
||||
phases.push({ name: "pre-flight", status: "ok", note: `kind=upgrade from=${doctor.current_version} to=${doctor.latest_version}` });
|
||||
|
||||
// phase 2: snapshot
|
||||
const fromCommit = opts.mockExec
|
||||
? "mock-commit"
|
||||
: execSync(`git -C ${ocpDir} rev-parse HEAD`).toString().trim();
|
||||
snapshotPath = opts.mockExec
|
||||
? "/tmp/mock-snapshot"
|
||||
: writeSnapshot({ homeDir: homedir(), fromCommit, fromVersion: doctor.current_version, toVersion: doctor.latest_version });
|
||||
phases.push({ name: "snapshot", path: snapshotPath, status: "ok" });
|
||||
|
||||
// phase 3: fetch + install
|
||||
exec(`git -C ${ocpDir} fetch --tags --quiet`, "fetch+install");
|
||||
exec(`git -C ${ocpDir} checkout ${doctor.latest_version}`, "fetch+install");
|
||||
exec(`npm --prefix ${ocpDir} install --no-audit --no-fund`, "fetch+install");
|
||||
|
||||
// phase 4: reconfigure
|
||||
exec(`node ${ocpDir}/setup.mjs`, "reconfigure");
|
||||
|
||||
// phase 5: restart (heads-up note printed before invoking)
|
||||
if (!opts.mockExec) {
|
||||
console.error(`[heads-up] restarting OCP service in 3s — expect ~5–10s blip on requests in flight.`);
|
||||
await new Promise(r => setTimeout(r, 3000));
|
||||
}
|
||||
if (process.platform === "darwin") {
|
||||
exec(`launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null || true`, "restart");
|
||||
exec(`launchctl bootstrap gui/$(id -u) ${join(homedir(), "Library", "LaunchAgents", "dev.ocp.proxy.plist")}`, "restart");
|
||||
} else {
|
||||
exec(`systemctl --user restart ocp-proxy.service`, "restart");
|
||||
}
|
||||
|
||||
// phase 6: post-flight (10s budget; skipped under mockExec)
|
||||
if (!opts.mockExec) {
|
||||
const port = process.env.CLAUDE_PROXY_PORT || String(DEFAULT_PORT);
|
||||
let ok = false;
|
||||
let lastSeen = null;
|
||||
for (let i = 0; i < 10; i++) {
|
||||
try {
|
||||
const out = execSync(`curl -sf --max-time 2 http://127.0.0.1:${port}/health`).toString();
|
||||
const body = JSON.parse(out);
|
||||
lastSeen = body.version;
|
||||
if (postFlightOk(body, doctor.latest_version)) { ok = true; break; }
|
||||
} catch { /* retry */ }
|
||||
await new Promise(r => setTimeout(r, 1000));
|
||||
}
|
||||
if (!ok) {
|
||||
phases.push({
|
||||
name: "post-flight", status: "fail",
|
||||
message: `health did not return auth.ok=true AND version=${doctor.latest_version} within 10s`
|
||||
+ (lastSeen ? ` (last saw version=${lastSeen} — a stale process may still hold the port; check \`ss -ltnp\` / \`lsof -i\`)` : ""),
|
||||
});
|
||||
throw new Error("post-flight failed");
|
||||
}
|
||||
execSync(`curl -sf --max-time 3 http://127.0.0.1:${port}/v1/models > /dev/null`);
|
||||
phases.push({ name: "post-flight", status: "ok" });
|
||||
} else {
|
||||
phases.push({ name: "post-flight", status: "skipped-mock" });
|
||||
}
|
||||
|
||||
// Auto-GC old snapshots after successful upgrade (best-effort, never throws).
|
||||
try {
|
||||
const gc = gcSnapshots(homedir(), { keepCount: 5, keepDays: 30 });
|
||||
if (gc.removed.length > 0) {
|
||||
console.error(`[gc] removed ${gc.removed.length} old snapshots; kept ${gc.kept.length}`);
|
||||
}
|
||||
} catch (e) {
|
||||
console.error(`[gc] warn: snapshot GC failed: ${e.message}`);
|
||||
}
|
||||
|
||||
return { path: "upgrade", executed: true, changed: true, snapshotPath, phases };
|
||||
} catch (err) {
|
||||
if (snapshotPath && !err.snapshotPath) {
|
||||
Object.assign(err, {
|
||||
snapshotPath,
|
||||
phases,
|
||||
hint: "Working tree may be at new version. Run `ocp update --rollback` to restore from snapshot."
|
||||
});
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
async function runFreshInstall({ doctor, opts }) {
|
||||
if (!opts.yes) {
|
||||
throw new Error("fresh_install requires --yes for non-interactive execution (or run interactively and answer y)");
|
||||
}
|
||||
const steps = [];
|
||||
for (const cmd of doctor.next_action.ai_executable) {
|
||||
if (opts.mockExec) {
|
||||
steps.push({ cmd, status: "skipped-mock" });
|
||||
} else {
|
||||
try {
|
||||
execSync(cmd, { stdio: "inherit" });
|
||||
steps.push({ cmd, status: "ok" });
|
||||
} catch (e) {
|
||||
const detail = e.stderr?.toString().trim() || e.message;
|
||||
steps.push({ cmd, status: "fail", error: String(detail) });
|
||||
throw Object.assign(new Error(`fresh_install step failed: ${cmd} — ${detail}`), { steps });
|
||||
}
|
||||
}
|
||||
}
|
||||
return { path: "fresh_install", executed: true, changed: true, steps };
|
||||
}
|
||||
|
||||
async function runRollback(opts) {
|
||||
const homeDir = opts.homeDir || homedir();
|
||||
const snapshots = opts.mockSnapshots ?? listSnapshots(homeDir);
|
||||
|
||||
if (opts.gc) {
|
||||
const result = gcSnapshots(homeDir, { dryRun: opts.dryRun });
|
||||
return { path: opts.dryRun ? "rollback-gc-dry-run" : "rollback-gc", ...result };
|
||||
}
|
||||
|
||||
if (opts.list) {
|
||||
return { path: "rollback-list", snapshots };
|
||||
}
|
||||
if (snapshots.length === 0) {
|
||||
throw new Error("no upgrade snapshots found in ~/.ocp/upgrade-snapshot-*");
|
||||
}
|
||||
|
||||
const target = opts.snapshotPath
|
||||
? snapshots.find(s => s.path === opts.snapshotPath)
|
||||
: snapshots[snapshots.length - 1];
|
||||
if (!target) throw new Error(`snapshot not found: ${opts.snapshotPath} (must be inside ~/.ocp/upgrade-snapshot-*)`);
|
||||
|
||||
const meta = opts.mockSnapshotMeta ?? readSnapshot(target.path);
|
||||
if (!meta.fromCommit) throw new Error(`snapshot ${target.path} has no from-commit.txt`);
|
||||
|
||||
const phases = [];
|
||||
if (opts.dryRun) {
|
||||
return {
|
||||
path: "rollback-dry-run",
|
||||
executed: false,
|
||||
target: target.path,
|
||||
plan: [
|
||||
`git checkout ${meta.fromCommit}`,
|
||||
`cp ${target.path}/plist ~/Library/LaunchAgents/dev.ocp.proxy.plist`,
|
||||
`cp ${target.path}/db.bak ~/.ocp/ocp.db`,
|
||||
`launchctl bootout/bootstrap`,
|
||||
`ocp doctor`
|
||||
]
|
||||
};
|
||||
}
|
||||
|
||||
if (!opts.yes) throw new Error("rollback requires --yes for non-interactive execution");
|
||||
|
||||
const exec = (cmd, label) => {
|
||||
if (opts.mockExec) {
|
||||
phases.push({ name: label, cmd, status: "skipped-mock" });
|
||||
return "";
|
||||
}
|
||||
try {
|
||||
execSync(cmd, { stdio: ["pipe", "pipe", "pipe"] });
|
||||
phases.push({ name: label, cmd, status: "ok" });
|
||||
} catch (err) {
|
||||
const detail = err.stderr?.toString().trim();
|
||||
phases.push({ name: label, cmd, status: "fail", stderr: detail });
|
||||
throw Object.assign(
|
||||
new Error(`rollback phase ${label} failed: ${detail || err.message}`),
|
||||
{ phases, target: target.path }
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
const ocpDir = opts.ocpDir || join(homedir(), "ocp");
|
||||
exec(`git -C ${ocpDir} checkout ${meta.fromCommit}`, "git-checkout");
|
||||
|
||||
if (!opts.mockExec) {
|
||||
const tryCopy = (src, dst) => {
|
||||
try {
|
||||
if (existsSync(src)) copyFileSync(src, dst);
|
||||
} catch (err) {
|
||||
console.error(`[rollback] warn: could not restore ${src} → ${dst} (${err.code || err.message})`);
|
||||
}
|
||||
};
|
||||
tryCopy(join(target.path, "plist"), join(homeDir, "Library", "LaunchAgents", "dev.ocp.proxy.plist"));
|
||||
tryCopy(join(target.path, "service"), join(homeDir, ".config", "systemd", "user", "ocp-proxy.service"));
|
||||
tryCopy(join(target.path, "db.bak"), join(homeDir, ".ocp", "ocp.db"));
|
||||
tryCopy(join(target.path, "admin-key"), join(homeDir, ".ocp", "admin-key"));
|
||||
phases.push({ name: "restore-files", status: "ok" });
|
||||
} else {
|
||||
phases.push({ name: "restore-files", status: "skipped-mock" });
|
||||
}
|
||||
|
||||
exec(`npm --prefix ${ocpDir} install --no-audit --no-fund`, "npm-install");
|
||||
|
||||
if (!opts.mockExec) {
|
||||
console.error(`[heads-up] restarting OCP service in 3s — expect ~5–10s blip on requests in flight.`);
|
||||
await new Promise(r => setTimeout(r, 3000));
|
||||
}
|
||||
if (process.platform === "darwin") {
|
||||
exec(`launchctl bootout gui/$(id -u)/dev.ocp.proxy 2>/dev/null || true`, "restart");
|
||||
exec(`launchctl bootstrap gui/$(id -u) ${join(homedir(), "Library", "LaunchAgents", "dev.ocp.proxy.plist")}`, "restart");
|
||||
} else {
|
||||
exec(`systemctl --user restart ocp-proxy.service`, "restart");
|
||||
}
|
||||
|
||||
return { path: "rollback", executed: true, changed: true, target: target.path, phases };
|
||||
}
|
||||
|
||||
// CLI entrypoint — use fileURLToPath + realpath to handle symlinked install paths.
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { realpathSync } from "node:fs";
|
||||
function _isMain() {
|
||||
if (!process.argv[1]) return false;
|
||||
try {
|
||||
return realpathSync(fileURLToPath(import.meta.url)) === realpathSync(process.argv[1]);
|
||||
} catch { return false; }
|
||||
}
|
||||
if (_isMain()) {
|
||||
const args = process.argv.slice(2);
|
||||
const dryRun = args.includes("--dry-run");
|
||||
const yes = args.includes("--yes");
|
||||
const rollback = args.includes("--rollback");
|
||||
const list = args.includes("--list");
|
||||
const gc = args.includes("--gc");
|
||||
const targetIdx = args.indexOf("--target");
|
||||
const target = targetIdx !== -1 ? args[targetIdx + 1] : undefined;
|
||||
// First non-flag positional after --rollback is the snapshot path
|
||||
let snapshotPath;
|
||||
if (rollback) {
|
||||
const rb = args.indexOf("--rollback");
|
||||
const cand = args[rb + 1];
|
||||
if (cand && !cand.startsWith("--")) snapshotPath = cand;
|
||||
}
|
||||
try {
|
||||
const result = await runUpgrade({ dryRun, yes, rollback, list, gc, snapshotPath, target });
|
||||
if (result.plan) for (const line of result.plan) console.log(line);
|
||||
if (result.phases) for (const p of result.phases) console.log(`[${p.name}] ${p.status}${p.cmd ? `: ${p.cmd}` : ""}`);
|
||||
if (result.steps) for (const s of result.steps) console.log(` ${s.status === "ok" ? "✓" : s.status === "skipped-mock" ? "·" : "✗"} ${s.cmd}`);
|
||||
if (result.snapshots) {
|
||||
console.log(`Found ${result.snapshots.length} snapshots:`);
|
||||
for (const s of result.snapshots) console.log(` ${s.name}`);
|
||||
}
|
||||
if (result.removed && result.kept) {
|
||||
console.log(`Snapshots: kept ${result.kept.length}, ${result.dryRun ? "would remove" : "removed"} ${result.removed.length}`);
|
||||
for (const s of result.removed) console.log(` - ${s.name}`);
|
||||
}
|
||||
process.exit(0);
|
||||
} catch (e) {
|
||||
console.error(`✗ ${e.message}`);
|
||||
if (e.snapshotPath) console.error(` snapshot: ${e.snapshotPath}`);
|
||||
if (e.target) console.error(` target: ${e.target}`);
|
||||
if (e.hint) console.error(` hint: ${e.hint}`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
+3548
-352
File diff suppressed because it is too large
Load Diff
@@ -1,9 +1,10 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* openclaw-claude-proxy setup
|
||||
* OCP (Open Claude Proxy) setup
|
||||
*
|
||||
* Automatically configures OpenClaw to use Claude CLI as a model provider.
|
||||
* Run: node setup.mjs [--port 3456] [--default-model opus|sonnet|haiku] [--dry-run]
|
||||
* Run: node setup.mjs [--port N] [--default-model opus|sonnet|haiku] [--dry-run]
|
||||
* (default port = DEFAULT_PORT from lib/constants.mjs)
|
||||
*
|
||||
* What it does:
|
||||
* 1. Verifies claude CLI is installed and authenticated
|
||||
@@ -12,11 +13,13 @@
|
||||
* 4. Creates start.sh for easy launch
|
||||
* 5. Optionally starts the proxy
|
||||
*/
|
||||
import { readFileSync, writeFileSync, existsSync, mkdirSync } from "node:fs";
|
||||
import { readFileSync, writeFileSync, existsSync, mkdirSync, unlinkSync, readdirSync, chmodSync } from "node:fs";
|
||||
import { mergePlistEnv, mergeSystemdEnv } from "./scripts/lib/plist-merge.mjs";
|
||||
import { execSync } from "node:child_process";
|
||||
import { join, dirname } from "node:path";
|
||||
import { homedir } from "node:os";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { DEFAULT_PORT } from "./lib/constants.mjs";
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const HOME = homedir();
|
||||
@@ -31,55 +34,78 @@ const opt = (name, fallback) => {
|
||||
return i >= 0 && args[i + 1] ? args[i + 1] : fallback;
|
||||
};
|
||||
|
||||
const PORT = parseInt(opt("port", "3456"), 10);
|
||||
const PORT = parseInt(opt("port", String(DEFAULT_PORT)), 10);
|
||||
const DEFAULT_MODEL = opt("default-model", "opus"); // opus | sonnet | haiku
|
||||
const DRY_RUN = flag("dry-run");
|
||||
const SKIP_START = flag("no-start");
|
||||
const PROVIDER_NAME = opt("provider-name", "claude-local");
|
||||
const BIND_ADDRESS = opt("bind", "127.0.0.1");
|
||||
const AUTH_MODE_CONFIG = opt("auth-mode", "none");
|
||||
|
||||
const MODEL_ID_MAP = {
|
||||
opus: "claude-opus-4-6",
|
||||
sonnet: "claude-sonnet-4-6",
|
||||
haiku: "claude-haiku-4",
|
||||
};
|
||||
// ── Service-env injection: CLAUDE_BIN, OCP_ADMIN_KEY, PROXY_ANONYMOUS_KEY ──
|
||||
// These are read from the user's shell env at install time and written into
|
||||
// the service unit (plist / systemd) so the daemon picks them up on boot.
|
||||
|
||||
// CLAUDE_BIN — detect at install time; omit if not found (server.mjs fallback)
|
||||
let CLAUDE_BIN_INJECT = null;
|
||||
if (process.env.CLAUDE_BIN) {
|
||||
CLAUDE_BIN_INJECT = process.env.CLAUDE_BIN;
|
||||
} else {
|
||||
try {
|
||||
const detected = execSync("which claude 2>/dev/null", { encoding: "utf-8" }).trim();
|
||||
if (detected && existsSync(detected)) {
|
||||
CLAUDE_BIN_INJECT = detected;
|
||||
}
|
||||
} catch { /* which not available or claude not on PATH — omit */ }
|
||||
}
|
||||
|
||||
// OCP_ADMIN_KEY — omit entirely when empty/unset; don't write empty string
|
||||
const OCP_ADMIN_KEY_INJECT = process.env.OCP_ADMIN_KEY || null;
|
||||
|
||||
// PROXY_ANONYMOUS_KEY — same pattern
|
||||
const PROXY_ANON_KEY_INJECT = process.env.PROXY_ANONYMOUS_KEY || null;
|
||||
|
||||
// ── Inject-value helpers ─────────────────────────────────────────────────
|
||||
// Escape a value for safe inclusion in a plist <string>…</string> body.
|
||||
function xmlEscape(v) {
|
||||
return String(v).replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/"/g, """).replace(/'/g, "'");
|
||||
}
|
||||
// Validate an injected service value: no control chars (a newline would inject a
|
||||
// rogue systemd Environment= directive; other control chars corrupt the unit/plist).
|
||||
// Spaces are allowed — filesystem paths (CLAUDE_BIN) may legitimately contain them.
|
||||
function assertSafeInjectValue(name, v) {
|
||||
if (v == null) return v;
|
||||
if (/[\x00-\x1f]/.test(String(v))) {
|
||||
console.error(`FATAL: ${name} contains a newline or control character — refusing to write it into the service unit.`);
|
||||
process.exit(1);
|
||||
}
|
||||
return v;
|
||||
}
|
||||
|
||||
// Validate all three INJECT values before they are written into any service unit.
|
||||
assertSafeInjectValue("CLAUDE_BIN", CLAUDE_BIN_INJECT);
|
||||
assertSafeInjectValue("OCP_ADMIN_KEY", OCP_ADMIN_KEY_INJECT);
|
||||
assertSafeInjectValue("PROXY_ANONYMOUS_KEY", PROXY_ANON_KEY_INJECT);
|
||||
|
||||
// ── Models: derived from models.json (single source of truth) ──────────
|
||||
const modelsConfig = JSON.parse(readFileSync(join(__dirname, "models.json"), "utf-8"));
|
||||
|
||||
const MODEL_ID_MAP = modelsConfig.aliases;
|
||||
const DEFAULT_MODEL_ID = MODEL_ID_MAP[DEFAULT_MODEL] || MODEL_ID_MAP.opus;
|
||||
|
||||
// ── Models to register ──────────────────────────────────────────────────
|
||||
const MODELS = [
|
||||
{
|
||||
id: "claude-opus-4-6",
|
||||
name: "Claude Opus 4.6 (via CLI)",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
maxTokens: 16384,
|
||||
},
|
||||
{
|
||||
id: "claude-sonnet-4-6",
|
||||
name: "Claude Sonnet 4.6 (via CLI)",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
maxTokens: 16384,
|
||||
},
|
||||
{
|
||||
id: "claude-haiku-4",
|
||||
name: "Claude Haiku 4 (via CLI)",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200000,
|
||||
maxTokens: 8192,
|
||||
},
|
||||
];
|
||||
const MODELS = modelsConfig.models.map(m => ({
|
||||
id: m.id,
|
||||
name: m.openclawName,
|
||||
reasoning: m.reasoning,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: m.contextWindow,
|
||||
maxTokens: m.maxTokens,
|
||||
}));
|
||||
|
||||
const MODEL_ALIASES = {
|
||||
[`${PROVIDER_NAME}/claude-opus-4-6`]: { alias: "Claude Opus 4.6" },
|
||||
[`${PROVIDER_NAME}/claude-sonnet-4-6`]: { alias: "Claude Sonnet 4.6" },
|
||||
[`${PROVIDER_NAME}/claude-haiku-4`]: { alias: "Claude Haiku 4" },
|
||||
};
|
||||
const MODEL_ALIASES = Object.fromEntries(
|
||||
modelsConfig.models.map(m => [`${PROVIDER_NAME}/${m.id}`, { alias: m.displayName }])
|
||||
);
|
||||
|
||||
// ── Helpers ─────────────────────────────────────────────────────────────
|
||||
function log(msg) { console.log(` ✓ ${msg}`); }
|
||||
@@ -115,107 +141,134 @@ try {
|
||||
}
|
||||
|
||||
// Check claude auth (quick test)
|
||||
try {
|
||||
const out = execSync('claude -p --output-format text --no-session-persistence -- "ping"', {
|
||||
encoding: "utf-8",
|
||||
timeout: 30000,
|
||||
env: { ...process.env, CLAUDECODE: undefined, ANTHROPIC_API_KEY: undefined, ANTHROPIC_BASE_URL: undefined, ANTHROPIC_AUTH_TOKEN: undefined },
|
||||
}).trim();
|
||||
if (out.length > 0) {
|
||||
log(`Claude CLI authenticated (test response: "${out.slice(0, 40)}...")`);
|
||||
// NOTE: This probe uses `claude -p` (sdk-cli spawn). After the 2026-06-15 Anthropic billing
|
||||
// split, every `claude -p` call draws from the Agent SDK credit pool rather than the
|
||||
// Pro/Max subscription. Re-running setup after 6/15 will consume one metered credit.
|
||||
// Set OCP_SKIP_AUTH_TEST=1 to skip this probe (auth is still validated at first real request).
|
||||
if (process.env.OCP_SKIP_AUTH_TEST === "1") {
|
||||
warn("OCP_SKIP_AUTH_TEST=1 — skipping claude auth probe (will be validated at first request).");
|
||||
} else {
|
||||
try {
|
||||
const out = execSync('claude -p --output-format text --no-session-persistence -- "ping"', {
|
||||
encoding: "utf-8",
|
||||
timeout: 30000,
|
||||
env: { ...process.env, CLAUDECODE: undefined, ANTHROPIC_API_KEY: undefined, ANTHROPIC_BASE_URL: undefined, ANTHROPIC_AUTH_TOKEN: undefined },
|
||||
}).trim();
|
||||
if (out.length > 0) {
|
||||
log(`Claude CLI authenticated (test response: "${out.slice(0, 40)}...")`);
|
||||
}
|
||||
} catch (e) {
|
||||
warn(`Claude CLI auth test failed: ${e.message.slice(0, 100)}`);
|
||||
warn("Make sure you're logged in: claude login");
|
||||
}
|
||||
} catch (e) {
|
||||
warn(`Claude CLI auth test failed: ${e.message.slice(0, 100)}`);
|
||||
warn("Make sure you're logged in: claude login");
|
||||
}
|
||||
|
||||
// Check openclaw config
|
||||
if (!existsSync(CONFIG_PATH)) fail(`OpenClaw config not found at ${CONFIG_PATH}`);
|
||||
log(`OpenClaw config: ${CONFIG_PATH}`);
|
||||
// Check openclaw config (optional — OCP runs standalone without OpenClaw)
|
||||
const OPENCLAW_PRESENT = existsSync(CONFIG_PATH);
|
||||
if (OPENCLAW_PRESENT) {
|
||||
log(`OpenClaw config: ${CONFIG_PATH}`);
|
||||
} else {
|
||||
warn(`OpenClaw not detected at ${CONFIG_PATH} — skipping OpenClaw integration.`);
|
||||
warn(`To register OCP with OpenClaw later, install OpenClaw and re-run \`node setup.mjs\`,`);
|
||||
warn(`or run \`ocp update\` if OpenClaw is installed afterward.`);
|
||||
}
|
||||
|
||||
// ── Step 2: Patch openclaw.json ─────────────────────────────────────────
|
||||
console.log("\n📝 Configuring OpenClaw...\n");
|
||||
if (OPENCLAW_PRESENT) {
|
||||
console.log("\n📝 Configuring OpenClaw...\n");
|
||||
|
||||
const config = readJSON(CONFIG_PATH);
|
||||
const config = readJSON(CONFIG_PATH);
|
||||
|
||||
// Ensure models.providers exists
|
||||
if (!config.models) config.models = {};
|
||||
if (!config.models.providers) config.models.providers = {};
|
||||
// Ensure models.providers exists
|
||||
if (!config.models) config.models = {};
|
||||
if (!config.models.providers) config.models.providers = {};
|
||||
|
||||
// Add/update claude-local provider
|
||||
config.models.providers[PROVIDER_NAME] = {
|
||||
baseUrl: `http://127.0.0.1:${PORT}/v1`,
|
||||
api: "openai-completions",
|
||||
authHeader: false,
|
||||
models: MODELS,
|
||||
};
|
||||
log(`Provider "${PROVIDER_NAME}" → http://127.0.0.1:${PORT}/v1`);
|
||||
// Add/update claude-local provider
|
||||
config.models.providers[PROVIDER_NAME] = {
|
||||
baseUrl: `http://127.0.0.1:${PORT}/v1`,
|
||||
api: "openai-completions",
|
||||
authHeader: false,
|
||||
models: MODELS,
|
||||
};
|
||||
log(`Provider "${PROVIDER_NAME}" → http://127.0.0.1:${PORT}/v1`);
|
||||
|
||||
// Ensure auth profile in config
|
||||
if (!config.auth) config.auth = {};
|
||||
if (!config.auth.profiles) config.auth.profiles = {};
|
||||
config.auth.profiles[`${PROVIDER_NAME}:default`] = {
|
||||
provider: PROVIDER_NAME,
|
||||
mode: "api_key",
|
||||
};
|
||||
log(`Auth profile "${PROVIDER_NAME}:default" registered`);
|
||||
// Ensure auth profile in config
|
||||
if (!config.auth) config.auth = {};
|
||||
if (!config.auth.profiles) config.auth.profiles = {};
|
||||
config.auth.profiles[`${PROVIDER_NAME}:default`] = {
|
||||
provider: PROVIDER_NAME,
|
||||
mode: "api_key",
|
||||
};
|
||||
log(`Auth profile "${PROVIDER_NAME}:default" registered`);
|
||||
|
||||
// Add models to agents.defaults.models
|
||||
if (!config.agents) config.agents = {};
|
||||
if (!config.agents.defaults) config.agents.defaults = {};
|
||||
if (!config.agents.defaults.models) config.agents.defaults.models = {};
|
||||
for (const [key, val] of Object.entries(MODEL_ALIASES)) {
|
||||
config.agents.defaults.models[key] = val;
|
||||
}
|
||||
log(`Model aliases added to agents.defaults.models`);
|
||||
|
||||
writeJSON(CONFIG_PATH, config);
|
||||
log(`Config saved`);
|
||||
|
||||
// ── Step 3: Patch auth-profiles.json ────────────────────────────────────
|
||||
console.log("\n🔑 Configuring auth profiles...\n");
|
||||
|
||||
// Find all agent auth-profiles.json files
|
||||
const agentsDir = join(OPENCLAW_DIR, "agents");
|
||||
const agentDirs = existsSync(agentsDir)
|
||||
? readdirSync(agentsDir).filter((d) => {
|
||||
const ap = join(agentsDir, d, "agent", "auth-profiles.json");
|
||||
return existsSync(ap);
|
||||
})
|
||||
: [];
|
||||
|
||||
import { readdirSync } from "node:fs";
|
||||
|
||||
for (const agentId of agentDirs) {
|
||||
const apPath = join(agentsDir, agentId, "agent", "auth-profiles.json");
|
||||
try {
|
||||
const ap = readJSON(apPath);
|
||||
if (!ap.profiles) ap.profiles = {};
|
||||
|
||||
// Add claude-local profile if missing
|
||||
if (!ap.profiles[`${PROVIDER_NAME}:default`]) {
|
||||
ap.profiles[`${PROVIDER_NAME}:default`] = {
|
||||
type: "api_key",
|
||||
provider: PROVIDER_NAME,
|
||||
key: "local-proxy-no-auth",
|
||||
};
|
||||
}
|
||||
|
||||
// Add to lastGood if missing
|
||||
if (!ap.lastGood) ap.lastGood = {};
|
||||
if (!ap.lastGood[PROVIDER_NAME]) {
|
||||
ap.lastGood[PROVIDER_NAME] = `${PROVIDER_NAME}:default`;
|
||||
}
|
||||
|
||||
writeJSON(apPath, ap);
|
||||
log(`Agent "${agentId}" auth profile updated`);
|
||||
} catch (e) {
|
||||
warn(`Skipped agent "${agentId}": ${e.message}`);
|
||||
// Add models to agents.defaults.models
|
||||
if (!config.agents) config.agents = {};
|
||||
if (!config.agents.defaults) config.agents.defaults = {};
|
||||
if (!config.agents.defaults.models) config.agents.defaults.models = {};
|
||||
for (const [key, val] of Object.entries(MODEL_ALIASES)) {
|
||||
config.agents.defaults.models[key] = val;
|
||||
}
|
||||
}
|
||||
log(`Model aliases added to agents.defaults.models`);
|
||||
|
||||
if (agentDirs.length === 0) {
|
||||
warn("No agent auth-profiles.json found — you may need to restart the gateway first");
|
||||
// Set idleTimeoutSeconds to 0 — critical for Claude tool-use.
|
||||
// When Claude calls tools (Bash, Read, etc.), the token stream pauses for 30-120s.
|
||||
// OpenClaw's default idleTimeoutSeconds (60s) kills the connection mid-tool-call,
|
||||
// causing exit 143 (SIGTERM) and stuck sessions. Setting to 0 disables the idle timer.
|
||||
if (!config.agents.defaults.llm) config.agents.defaults.llm = {};
|
||||
if (config.agents.defaults.llm.idleTimeoutSeconds === undefined ||
|
||||
config.agents.defaults.llm.idleTimeoutSeconds > 0) {
|
||||
config.agents.defaults.llm.idleTimeoutSeconds = 0;
|
||||
log(`Set agents.defaults.llm.idleTimeoutSeconds = 0 (prevents tool-call timeouts)`);
|
||||
} else {
|
||||
log(`idleTimeoutSeconds already configured: ${config.agents.defaults.llm.idleTimeoutSeconds}`);
|
||||
}
|
||||
|
||||
writeJSON(CONFIG_PATH, config);
|
||||
log(`Config saved`);
|
||||
|
||||
// ── Step 3: Patch auth-profiles.json ────────────────────────────────────
|
||||
console.log("\n🔑 Configuring auth profiles...\n");
|
||||
|
||||
// Find all agent auth-profiles.json files
|
||||
const agentsDir = join(OPENCLAW_DIR, "agents");
|
||||
const agentDirs = existsSync(agentsDir)
|
||||
? readdirSync(agentsDir).filter((d) => {
|
||||
const ap = join(agentsDir, d, "agent", "auth-profiles.json");
|
||||
return existsSync(ap);
|
||||
})
|
||||
: [];
|
||||
|
||||
for (const agentId of agentDirs) {
|
||||
const apPath = join(agentsDir, agentId, "agent", "auth-profiles.json");
|
||||
try {
|
||||
const ap = readJSON(apPath);
|
||||
if (!ap.profiles) ap.profiles = {};
|
||||
|
||||
// Add claude-local profile if missing
|
||||
if (!ap.profiles[`${PROVIDER_NAME}:default`]) {
|
||||
ap.profiles[`${PROVIDER_NAME}:default`] = {
|
||||
type: "api_key",
|
||||
provider: PROVIDER_NAME,
|
||||
key: "local-proxy-no-auth",
|
||||
};
|
||||
}
|
||||
|
||||
// Add to lastGood if missing
|
||||
if (!ap.lastGood) ap.lastGood = {};
|
||||
if (!ap.lastGood[PROVIDER_NAME]) {
|
||||
ap.lastGood[PROVIDER_NAME] = `${PROVIDER_NAME}:default`;
|
||||
}
|
||||
|
||||
writeJSON(apPath, ap);
|
||||
log(`Agent "${agentId}" auth profile updated`);
|
||||
} catch (e) {
|
||||
warn(`Skipped agent "${agentId}": ${e.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (agentDirs.length === 0) {
|
||||
warn("No agent auth-profiles.json found — you may need to restart the gateway first");
|
||||
}
|
||||
}
|
||||
|
||||
// ── Step 4: Create start.sh ─────────────────────────────────────────────
|
||||
@@ -226,7 +279,7 @@ const logDir = join(OPENCLAW_DIR, "logs");
|
||||
if (!existsSync(logDir)) mkdirSync(logDir, { recursive: true });
|
||||
|
||||
const startSh = `#!/bin/bash
|
||||
# Start openclaw-claude-proxy if not already running
|
||||
# Start OCP (Open Claude Proxy) if not already running
|
||||
PORT=\${CLAUDE_PROXY_PORT:-${PORT}}
|
||||
if ! lsof -i :\$PORT -sTCP:LISTEN &>/dev/null; then
|
||||
unset CLAUDECODE
|
||||
@@ -247,64 +300,133 @@ if (!DRY_RUN) {
|
||||
log(`Launcher: ${startPath}`);
|
||||
|
||||
// ── Step 5: Summary ─────────────────────────────────────────────────────
|
||||
console.log(`
|
||||
╔══════════════════════════════════════════════════════════════╗
|
||||
║ Setup complete! ║
|
||||
╠══════════════════════════════════════════════════════════════╣
|
||||
║ ║
|
||||
║ Provider: ${PROVIDER_NAME.padEnd(44)}║
|
||||
║ Port: ${String(PORT).padEnd(44)}║
|
||||
║ Models: claude-opus-4-6, claude-sonnet-4-6, claude-haiku-4║
|
||||
║ Default: ${DEFAULT_MODEL_ID.padEnd(44)}║
|
||||
║ ║
|
||||
║ Start proxy: ║
|
||||
║ bash ${startPath.replace(HOME, "~").padEnd(50)}║
|
||||
║ ║
|
||||
║ Or directly: ║
|
||||
║ node ${serverPath.replace(HOME, "~").padEnd(49)}║
|
||||
║ ║
|
||||
║ Set as default model in openclaw.json: ║
|
||||
║ agents.defaults.model.primary = ║
|
||||
║ "${PROVIDER_NAME}/${DEFAULT_MODEL_ID}"${" ".repeat(Math.max(0, 30 - PROVIDER_NAME.length - DEFAULT_MODEL_ID.length))}║
|
||||
║ ║
|
||||
║ Then restart gateway: ║
|
||||
║ openclaw gateway restart ║
|
||||
║ ║
|
||||
╚══════════════════════════════════════════════════════════════╝
|
||||
`);
|
||||
|
||||
// ── Step 6: Optionally start ────────────────────────────────────────────
|
||||
if (!SKIP_START && !DRY_RUN) {
|
||||
try {
|
||||
execSync(`bash "${startPath}"`, { stdio: "inherit" });
|
||||
} catch { /* ignore */ }
|
||||
const banner = [
|
||||
`╔══════════════════════════════════════════════════════════════╗`,
|
||||
`║ Setup complete! ║`,
|
||||
`╠══════════════════════════════════════════════════════════════╣`,
|
||||
`║ ║`,
|
||||
`║ Provider: ${PROVIDER_NAME.padEnd(44)}║`,
|
||||
`║ Port: ${String(PORT).padEnd(44)}║`,
|
||||
`║ Models: ${`see models.json (${MODELS.length} available)`.padEnd(44)}║`,
|
||||
`║ Default: ${DEFAULT_MODEL_ID.padEnd(44)}║`,
|
||||
`║ ║`,
|
||||
`║ Start proxy: ║`,
|
||||
`║ bash ${startPath.replace(HOME, "~").padEnd(50)}║`,
|
||||
`║ ║`,
|
||||
`║ Or directly: ║`,
|
||||
`║ node ${serverPath.replace(HOME, "~").padEnd(49)}║`,
|
||||
`║ ║`,
|
||||
];
|
||||
if (OPENCLAW_PRESENT) {
|
||||
banner.push(
|
||||
`║ Set as default model in openclaw.json: ║`,
|
||||
`║ agents.defaults.model.primary = ║`,
|
||||
`║ "${PROVIDER_NAME}/${DEFAULT_MODEL_ID}"${" ".repeat(Math.max(0, 30 - PROVIDER_NAME.length - DEFAULT_MODEL_ID.length))}║`,
|
||||
`║ ║`,
|
||||
`║ Then restart gateway: ║`,
|
||||
`║ openclaw gateway restart ║`,
|
||||
`║ ║`,
|
||||
);
|
||||
} else {
|
||||
banner.push(
|
||||
`║ OpenClaw not detected — running in standalone mode. ║`,
|
||||
`║ Point your IDE (Cline / Cursor / Continue / OpenCode / ║`,
|
||||
`║ Aider / OpenClaw) at: ║`,
|
||||
`║ http://${BIND_ADDRESS}:${String(PORT)}/v1${" ".repeat(Math.max(0, 47 - BIND_ADDRESS.length - String(PORT).length))}║`,
|
||||
`║ ║`,
|
||||
`║ See docs/lan-mode.md for per-IDE client setup. ║`,
|
||||
`║ ║`,
|
||||
);
|
||||
}
|
||||
banner.push(`╚══════════════════════════════════════════════════════════════╝`);
|
||||
console.log("\n" + banner.join("\n") + "\n");
|
||||
|
||||
// ── Step 7: Install auto-start on boot ──────────────────────────────────
|
||||
|
||||
// Log service-env injection plan (shown in both dry-run and live mode)
|
||||
console.log("\n🔧 Service unit env vars to inject:\n");
|
||||
if (CLAUDE_BIN_INJECT) {
|
||||
log(`CLAUDE_BIN: ${CLAUDE_BIN_INJECT}`);
|
||||
} else {
|
||||
log(`CLAUDE_BIN: (not found — server.mjs will auto-detect at runtime)`);
|
||||
}
|
||||
if (OCP_ADMIN_KEY_INJECT) {
|
||||
log(`OCP_ADMIN_KEY: injected (length: ${OCP_ADMIN_KEY_INJECT.length})`);
|
||||
} else {
|
||||
log(`OCP_ADMIN_KEY: (unset — admin endpoints disabled)`);
|
||||
}
|
||||
if (PROXY_ANON_KEY_INJECT) {
|
||||
log(`PROXY_ANONYMOUS_KEY: injected (set)`);
|
||||
} else {
|
||||
log(`PROXY_ANONYMOUS_KEY: (unset — anonymous access disabled)`);
|
||||
}
|
||||
|
||||
if (DRY_RUN) {
|
||||
console.log("\n [dry-run] would write service unit with above env vars\n");
|
||||
}
|
||||
|
||||
if (!DRY_RUN) {
|
||||
console.log("\n🔄 Installing auto-start on login...\n");
|
||||
|
||||
const platform = process.platform;
|
||||
const nodeBin = process.execPath;
|
||||
// Use stable symlink path instead of versioned Cellar path (e.g. /opt/homebrew/opt/node/bin/node
|
||||
// instead of /opt/homebrew/Cellar/node/25.8.0/bin/node) so the plist survives node upgrades.
|
||||
let nodeBin = process.execPath;
|
||||
if (platform === "darwin" && nodeBin.includes("/Cellar/")) {
|
||||
const stable = nodeBin.replace(/\/Cellar\/[^/]+\/[^/]+\//, "/opt/");
|
||||
if (existsSync(stable)) {
|
||||
nodeBin = stable;
|
||||
log(`Using stable node path: ${nodeBin}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Ensure logs dir exists
|
||||
const logsDir = join(OPENCLAW_DIR, "logs");
|
||||
if (!existsSync(logsDir)) mkdirSync(logsDir, { recursive: true });
|
||||
|
||||
// Use neutral service names to avoid OpenClaw gateway's extra-service detection.
|
||||
// OpenClaw scans LaunchAgent plists and systemd units for "openclaw" / "clawdbot"
|
||||
// markers and flags them as conflicting gateway-like services. Using "dev.ocp.*"
|
||||
// and "ocp-proxy" keeps the proxy invisible to that heuristic.
|
||||
const OCP_HOME = join(HOME, ".ocp");
|
||||
const ocpLogsDir = join(OCP_HOME, "logs");
|
||||
// mode 0700: with `recursive`, this call can create ~/.ocp ITSELF on a fresh install, and
|
||||
// without an explicit mode that parent lands at the umask default (world-listable 0755).
|
||||
if (!existsSync(ocpLogsDir)) mkdirSync(ocpLogsDir, { recursive: true, mode: 0o700 });
|
||||
|
||||
// Uninstall legacy service names if present (upgrade path)
|
||||
if (platform === "darwin") {
|
||||
const legacyPlist = join(HOME, "Library", "LaunchAgents", "ai.openclaw.proxy.plist");
|
||||
if (existsSync(legacyPlist)) {
|
||||
try { execSync(`launchctl bootout gui/$(id -u) "${legacyPlist}" 2>/dev/null`); } catch { /* ignore */ }
|
||||
try { unlinkSync(legacyPlist); } catch { /* ignore */ }
|
||||
log(`Removed legacy plist: ai.openclaw.proxy`);
|
||||
}
|
||||
} else if (platform === "linux") {
|
||||
const legacyService = join(HOME, ".config", "systemd", "user", "openclaw-proxy.service");
|
||||
if (existsSync(legacyService)) {
|
||||
try { execSync(`systemctl --user stop openclaw-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
try { execSync(`systemctl --user disable openclaw-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
try { unlinkSync(legacyService); } catch { /* ignore */ }
|
||||
try { execSync(`systemctl --user daemon-reload`); } catch { /* ignore */ }
|
||||
log(`Removed legacy systemd service: openclaw-proxy`);
|
||||
}
|
||||
}
|
||||
|
||||
if (platform === "darwin") {
|
||||
// macOS: launchd
|
||||
const plistDir = join(HOME, "Library", "LaunchAgents");
|
||||
if (!existsSync(plistDir)) mkdirSync(plistDir, { recursive: true });
|
||||
|
||||
const plistPath = join(plistDir, "ai.openclaw.proxy.plist");
|
||||
const logPath = join(logsDir, "proxy.log");
|
||||
const plistPath = join(plistDir, "dev.ocp.proxy.plist");
|
||||
const logPath = join(ocpLogsDir, "proxy.log");
|
||||
|
||||
const plistXml = `<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>ai.openclaw.proxy</string>
|
||||
<string>dev.ocp.proxy</string>
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>${nodeBin}</string>
|
||||
@@ -313,7 +435,17 @@ if (!DRY_RUN) {
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>CLAUDE_PROXY_PORT</key>
|
||||
<string>${PORT}</string>
|
||||
<string>${xmlEscape(PORT)}</string>
|
||||
<key>CLAUDE_BIND</key>
|
||||
<string>${xmlEscape(BIND_ADDRESS)}</string>
|
||||
<key>CLAUDE_AUTH_MODE</key>
|
||||
<string>${xmlEscape(AUTH_MODE_CONFIG)}</string>${CLAUDE_BIN_INJECT ? `
|
||||
<key>CLAUDE_BIN</key>
|
||||
<string>${xmlEscape(CLAUDE_BIN_INJECT)}</string>` : ""}${OCP_ADMIN_KEY_INJECT ? `
|
||||
<key>OCP_ADMIN_KEY</key>
|
||||
<string>${xmlEscape(OCP_ADMIN_KEY_INJECT)}</string>` : ""}${PROXY_ANON_KEY_INJECT ? `
|
||||
<key>PROXY_ANONYMOUS_KEY</key>
|
||||
<string>${xmlEscape(PROXY_ANON_KEY_INJECT)}</string>` : ""}
|
||||
</dict>
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
@@ -327,30 +459,40 @@ if (!DRY_RUN) {
|
||||
</plist>
|
||||
`;
|
||||
|
||||
writeFileSync(plistPath, plistXml);
|
||||
log(`Plist written: ${plistPath}`);
|
||||
const existingPlist = existsSync(plistPath) ? readFileSync(plistPath, "utf8") : null;
|
||||
const finalPlistXml = mergePlistEnv(existingPlist, plistXml);
|
||||
writeFileSync(plistPath, finalPlistXml);
|
||||
chmodSync(plistPath, 0o600);
|
||||
if (existingPlist && finalPlistXml !== plistXml) {
|
||||
log(`Plist written: ${plistPath} (mode 600, preserved user env vars)`);
|
||||
} else {
|
||||
log(`Plist written: ${plistPath} (mode 600)`);
|
||||
}
|
||||
|
||||
// Unload first (in case it was already loaded) then load
|
||||
try { execSync(`launchctl unload "${plistPath}" 2>/dev/null`); } catch { /* ignore */ }
|
||||
execSync(`launchctl load "${plistPath}"`);
|
||||
log(`launchctl loaded ai.openclaw.proxy`);
|
||||
// Bootout first (in case it was already loaded) then bootstrap
|
||||
try { execSync(`launchctl bootout gui/$(id -u) "${plistPath}" 2>/dev/null`); } catch { /* ignore */ }
|
||||
execSync(`launchctl bootstrap gui/$(id -u) "${plistPath}"`);
|
||||
log(`launchctl loaded dev.ocp.proxy`);
|
||||
|
||||
} else if (platform === "linux") {
|
||||
// Linux: systemd user service
|
||||
const systemdDir = join(HOME, ".config", "systemd", "user");
|
||||
if (!existsSync(systemdDir)) mkdirSync(systemdDir, { recursive: true });
|
||||
|
||||
const servicePath = join(systemdDir, "openclaw-proxy.service");
|
||||
const logPath = join(logsDir, "proxy.log");
|
||||
const servicePath = join(systemdDir, "ocp-proxy.service");
|
||||
const logPath = join(ocpLogsDir, "proxy.log");
|
||||
|
||||
const serviceUnit = `[Unit]
|
||||
Description=OpenClaw Claude Proxy
|
||||
Description=OCP — Open Claude Proxy
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
ExecStart=${nodeBin} ${serverPath}
|
||||
Environment=CLAUDE_PROXY_PORT=${PORT}
|
||||
Environment=CLAUDE_BIND=${BIND_ADDRESS}
|
||||
Environment=CLAUDE_AUTH_MODE=${AUTH_MODE_CONFIG}${CLAUDE_BIN_INJECT ? `\nEnvironment=CLAUDE_BIN=${CLAUDE_BIN_INJECT}` : ""}${OCP_ADMIN_KEY_INJECT ? `\nEnvironment=OCP_ADMIN_KEY=${OCP_ADMIN_KEY_INJECT}` : ""}${PROXY_ANON_KEY_INJECT ? `\nEnvironment=PROXY_ANONYMOUS_KEY=${PROXY_ANON_KEY_INJECT}` : ""}
|
||||
Restart=always
|
||||
RestartSec=5
|
||||
StandardOutput=append:${logPath}
|
||||
StandardError=append:${logPath}
|
||||
|
||||
@@ -358,12 +500,19 @@ StandardError=append:${logPath}
|
||||
WantedBy=default.target
|
||||
`;
|
||||
|
||||
writeFileSync(servicePath, serviceUnit);
|
||||
log(`Service file written: ${servicePath}`);
|
||||
const existingService = existsSync(servicePath) ? readFileSync(servicePath, "utf8") : null;
|
||||
const finalServiceUnit = mergeSystemdEnv(existingService, serviceUnit);
|
||||
writeFileSync(servicePath, finalServiceUnit);
|
||||
chmodSync(servicePath, 0o600);
|
||||
if (existingService && finalServiceUnit !== serviceUnit) {
|
||||
log(`Service file written: ${servicePath} (mode 600, preserved user env vars)`);
|
||||
} else {
|
||||
log(`Service file written: ${servicePath} (mode 600)`);
|
||||
}
|
||||
|
||||
execSync(`systemctl --user daemon-reload`);
|
||||
execSync(`systemctl --user enable openclaw-proxy`);
|
||||
execSync(`systemctl --user start openclaw-proxy`);
|
||||
execSync(`systemctl --user enable ocp-proxy`);
|
||||
execSync(`systemctl --user start ocp-proxy`);
|
||||
log(`systemd user service enabled and started`);
|
||||
|
||||
} else {
|
||||
@@ -371,4 +520,52 @@ WantedBy=default.target
|
||||
}
|
||||
|
||||
console.log("\n✅ Auto-start installed — proxy will start automatically on login\n");
|
||||
|
||||
// ── Step 8: Post-install health verification ───────────────────────────
|
||||
if (!SKIP_START) {
|
||||
console.log("⏳ Waiting for server to bind...\n");
|
||||
await new Promise(r => setTimeout(r, 3000));
|
||||
|
||||
const healthUrl = `http://127.0.0.1:${PORT}/health`;
|
||||
let verified = false;
|
||||
try {
|
||||
const controller = new AbortController();
|
||||
const timer = setTimeout(() => controller.abort(), 5000);
|
||||
const res = await fetch(healthUrl, { signal: controller.signal });
|
||||
clearTimeout(timer);
|
||||
|
||||
if (res.ok) {
|
||||
const body = await res.json().catch(() => ({}));
|
||||
console.log(` ✓ Health check passed (${healthUrl})`);
|
||||
console.log(` version: ${body.version ?? "unknown"}`);
|
||||
console.log(` authMode: ${body.authMode ?? "unknown"}`);
|
||||
|
||||
// Verify bind socket
|
||||
try {
|
||||
const bindCheck = process.platform === "linux"
|
||||
? execSync(`ss -tlnp 2>/dev/null | grep ':${PORT}'`, { encoding: "utf-8" }).trim()
|
||||
: execSync(`lsof -nP -iTCP:${PORT} -sTCP:LISTEN 2>/dev/null`, { encoding: "utf-8" }).trim();
|
||||
if (bindCheck) {
|
||||
console.log(` bind: ${bindCheck.split("\n")[0]}`);
|
||||
}
|
||||
} catch { /* bind check is best-effort */ }
|
||||
|
||||
verified = true;
|
||||
} else {
|
||||
warn(`Health check returned HTTP ${res.status} — service may not have started cleanly`);
|
||||
}
|
||||
} catch (e) {
|
||||
const isTimeout = e.name === "AbortError" || (e.cause && e.cause.code === "UND_ERR_CONNECT_TIMEOUT");
|
||||
warn(`Health check failed: ${isTimeout ? "timeout (5s)" : e.message}`);
|
||||
}
|
||||
|
||||
if (!verified) {
|
||||
const logHint = process.platform === "linux"
|
||||
? "journalctl --user -u ocp-proxy -n 50"
|
||||
: `tail -n 100 ~/.ocp/logs/proxy.log`;
|
||||
console.error(`\n ✗ Server did not respond on port ${PORT} within 5 seconds.`);
|
||||
console.error(` Check service logs:\n ${logHint}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,12 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Start openclaw-claude-proxy if not already running
|
||||
PORT=${CLAUDE_PROXY_PORT:-3456}
|
||||
if ! lsof -i :$PORT -sTCP:LISTEN &>/dev/null; then
|
||||
unset CLAUDECODE
|
||||
nohup node "/Users/taodeng/.openclaw/projects/claude-proxy/openclaw-claude-proxy/server.mjs" \
|
||||
>> "/Users/taodeng/.openclaw/logs/claude-proxy.log" \
|
||||
2>> "/Users/taodeng/.openclaw/logs/claude-proxy.err.log" &
|
||||
echo "claude-proxy started on port $PORT (pid $!)"
|
||||
else
|
||||
echo "claude-proxy already running on port $PORT"
|
||||
fi
|
||||
@@ -0,0 +1,26 @@
|
||||
// Imported FIRST by test-features.mjs, before keys.mjs, so this runs before anything can open
|
||||
// the key store. ESM hoists imports and evaluates them in order, so a `process.env.X = ...`
|
||||
// statement in the test's own body would run too late — hence a separate module.
|
||||
//
|
||||
// Why this exists: `npm test` used to write real, UNREVOKED api_keys rows into the operator's
|
||||
// live ~/.ocp/ocp.db (the same database the running server reads) — two per run, unbounded.
|
||||
// It also made the suite racy: two concurrent runs (e.g. review worktrees) shared one file, so
|
||||
// `listKeys()` could miss "test-user-1" and the `in` check would throw on undefined.
|
||||
import { mkdtempSync, rmSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
|
||||
export const TEST_OCP_DIR = mkdtempSync(join(tmpdir(), "ocp-test-"));
|
||||
|
||||
// BOTH are required. keys.mjs honors OCP_DIR_OVERRIDE only when NODE_ENV === "test", so neither
|
||||
// var alone redirects anything — a stray OCP_DIR_OVERRIDE in a production env is inert without
|
||||
// NODE_ENV=test alongside it. (A daemon OCP launches never carries either: the service units and
|
||||
// the `ocp` restart fallback strip both — see plist-merge NEVER_PRESERVE / keys.mjs's comment.)
|
||||
process.env.NODE_ENV = "test";
|
||||
process.env.OCP_DIR_OVERRIDE = TEST_OCP_DIR;
|
||||
|
||||
// Remove the scratch store on exit. Without this the fix would trade unbounded growth in
|
||||
// ~/.ocp/ocp.db for unbounded growth in $TMPDIR — better, but still litter.
|
||||
process.on("exit", () => {
|
||||
try { rmSync(TEST_OCP_DIR, { recursive: true, force: true }); } catch { /* best effort */ }
|
||||
});
|
||||
+4978
File diff suppressed because it is too large
Load Diff
+36
-24
@@ -1,8 +1,11 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* openclaw-claude-proxy uninstaller
|
||||
* OCP (Open Claude Proxy) uninstaller
|
||||
*
|
||||
* Stops and removes the launchd (macOS) or systemd (Linux) auto-start entry.
|
||||
* Handles both legacy (ai.openclaw.proxy / openclaw-proxy) and current
|
||||
* (dev.ocp.proxy / ocp-proxy) service names.
|
||||
*
|
||||
* Run: node uninstall.mjs
|
||||
*/
|
||||
import { existsSync, unlinkSync } from "node:fs";
|
||||
@@ -15,43 +18,52 @@ const HOME = homedir();
|
||||
function log(msg) { console.log(` ✓ ${msg}`); }
|
||||
function warn(msg) { console.log(` ⚠ ${msg}`); }
|
||||
|
||||
console.log("\n🗑 Uninstalling openclaw-claude-proxy auto-start...\n");
|
||||
console.log("\n🗑 Uninstalling OCP auto-start...\n");
|
||||
|
||||
const platform = process.platform;
|
||||
|
||||
if (platform === "darwin") {
|
||||
const plistPath = join(HOME, "Library", "LaunchAgents", "ai.openclaw.proxy.plist");
|
||||
|
||||
// Remove current service
|
||||
const plistPath = join(HOME, "Library", "LaunchAgents", "dev.ocp.proxy.plist");
|
||||
if (existsSync(plistPath)) {
|
||||
try {
|
||||
execSync(`launchctl unload "${plistPath}" 2>/dev/null`);
|
||||
log("launchd service stopped and unloaded");
|
||||
} catch {
|
||||
warn("launchctl unload failed (service may not have been running)");
|
||||
}
|
||||
try { execSync(`launchctl bootout gui/$(id -u) "${plistPath}" 2>/dev/null`); } catch { /* ignore */ }
|
||||
unlinkSync(plistPath);
|
||||
log(`Plist removed: ${plistPath}`);
|
||||
} else {
|
||||
warn(`Plist not found: ${plistPath}`);
|
||||
log(`Removed: ${plistPath}`);
|
||||
}
|
||||
|
||||
// Remove legacy service
|
||||
const legacyPath = join(HOME, "Library", "LaunchAgents", "ai.openclaw.proxy.plist");
|
||||
if (existsSync(legacyPath)) {
|
||||
try { execSync(`launchctl bootout gui/$(id -u) "${legacyPath}" 2>/dev/null`); } catch { /* ignore */ }
|
||||
unlinkSync(legacyPath);
|
||||
log(`Removed legacy: ${legacyPath}`);
|
||||
}
|
||||
|
||||
if (!existsSync(plistPath) && !existsSync(legacyPath)) {
|
||||
warn("No plist found (service may not have been installed)");
|
||||
}
|
||||
|
||||
} else if (platform === "linux") {
|
||||
const servicePath = join(HOME, ".config", "systemd", "user", "openclaw-proxy.service");
|
||||
|
||||
try { execSync(`systemctl --user stop openclaw-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
log("systemd service stopped");
|
||||
|
||||
try { execSync(`systemctl --user disable openclaw-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
log("systemd service disabled");
|
||||
|
||||
// Remove current service
|
||||
const servicePath = join(HOME, ".config", "systemd", "user", "ocp-proxy.service");
|
||||
try { execSync(`systemctl --user stop ocp-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
try { execSync(`systemctl --user disable ocp-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
if (existsSync(servicePath)) {
|
||||
unlinkSync(servicePath);
|
||||
log(`Service file removed: ${servicePath}`);
|
||||
} else {
|
||||
warn(`Service file not found: ${servicePath}`);
|
||||
log(`Removed: ${servicePath}`);
|
||||
}
|
||||
|
||||
// Remove legacy service
|
||||
const legacyPath = join(HOME, ".config", "systemd", "user", "openclaw-proxy.service");
|
||||
try { execSync(`systemctl --user stop openclaw-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
try { execSync(`systemctl --user disable openclaw-proxy 2>/dev/null`); } catch { /* ignore */ }
|
||||
if (existsSync(legacyPath)) {
|
||||
unlinkSync(legacyPath);
|
||||
log(`Removed legacy: ${legacyPath}`);
|
||||
}
|
||||
|
||||
try { execSync(`systemctl --user daemon-reload`); } catch { /* ignore */ }
|
||||
log("systemd daemon reloaded");
|
||||
|
||||
} else {
|
||||
warn(`Auto-start not supported on ${platform} — nothing to remove`);
|
||||
|
||||
Reference in New Issue
Block a user