diff --git a/.codexclaw/goalplans/opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr/goalplan.json b/.codexclaw/goalplans/opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr/goalplan.json deleted file mode 100644 index 937dba65c..000000000 --- a/.codexclaw/goalplans/opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr/goalplan.json +++ /dev/null @@ -1,356 +0,0 @@ -{ - "objective": "opencodex 저장소 거버넌스 및 기여 정책 4건을 PABCD 다중 work-phase 루프로 처리한다. WP1 문서화 우선 사이클, WP2 dev2-go 기반 PR 및 포팅/리베이스 PR 허용 정책, WP3 대상 PR 두 개로 분할, WP4 Wibias 메인테이너 추가. 검증은 실제 파일과 명령 출력으로만. 서브에이전트는 gpt-5.6-terra medium 일반 티어 활용.", - "slug": "opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr", - "createdAt": "2026-07-27T00:41:29.352Z", - "updatedAt": "2026-07-27T04:58:54.547Z", - "activeWorkPhaseId": "wp8", - "workPhases": [ - { - "id": "wp1", - "title": "docs-only 로드맵 사이클: devlog/_plan/260727_governance_intake/ 000/010/020/030 작성", - "status": "done", - "tasks": [ - { - "id": "wp1-t1", - "title": "000 현황 조사: MAINTAINERS.md/CODEOWNERS/AGENTS.md/CONTRIBUTING 및 브랜치 실측", - "status": "done" - }, - { - "id": "wp1-t2", - "title": "010 dev2-go 기반 PR + 포팅/리베이스 PR 허용 정책 diff-level 안", - "status": "done" - }, - { - "id": "wp1-t3", - "title": "020 대상 PR 분할 안 (어떤 PR을 어떤 경계로 두 개로)", - "status": "done" - }, - { - "id": "wp1-t4", - "title": "030 Wibias 메인테이너 추가 diff-level 안", - "status": "done" - } - ], - "criteriaIds": [ - "c1" - ] - }, - { - "id": "wp2", - "title": "브랜치 정책 문서화 (문서 전용): dev2-go 통합선 + 포팅/리베이스 PR 환영", - "status": "done", - "tasks": [ - { - "id": "wp2-t1", - "title": "AGENTS.md Branch policy + Review guidelines", - "status": "done" - }, - { - "id": "wp2-t2", - "title": "CONTRIBUTING.md Branches 절", - "status": "done" - }, - { - "id": "wp2-t3", - "title": "MAINTAINERS.md 리뷰/머지 정책", - "status": "done" - } - ], - "criteriaIds": [ - "c2" - ] - }, - { - "id": "wp4", - "title": "Wibias 메인테이너 추가 (MAINTAINERS.md + CODEOWNERS)", - "status": "done", - "tasks": [ - { - "id": "wp4-t1", - "title": "MAINTAINERS.md Current maintainers 표에 @Wibias 추가", - "status": "done" - }, - { - "id": "wp4-t2", - "title": ".github/CODEOWNERS 기본 리뷰어 및 보안 경로 갱신", - "status": "done" - }, - { - "id": "wp4-t3", - "title": "Maintainer changes 절차 이행 기록", - "status": "done" - } - ], - "criteriaIds": [ - "c4" - ] - }, - { - "id": "wp3", - "title": "대상 PR 두 개로 분할 (WP4 이후 — 분할 대상 #518은 Wibias 소유이므로 메인테이너 지위가 선행)", - "status": "done", - "tasks": [ - { - "id": "wp3-t1", - "title": "020 문서가 지정한 분할 대상 및 경계 확정", - "status": "done" - }, - { - "id": "wp3-t2", - "title": "분할 브랜치 생성 및 커밋 분리", - "status": "done" - }, - { - "id": "wp3-t3", - "title": "typecheck + 관련 테스트 통과 확인", - "status": "done" - } - ], - "criteriaIds": [ - "c3" - ] - }, - { - "id": "wp5", - "title": "PR 타깃 게이트 재설계 (enforce-pr-target.yml) — 사용자 승인 완료, allow-list 방식 채택", - "status": "done", - "tasks": [ - { - "id": "wp5-t1", - "title": "040 문서에 승인 결과와 채택안(c: allow-list) 확정 기록", - "status": "done" - }, - { - "id": "wp5-t2", - "title": "enforce-pr-target.yml: EXPECTED_BASE -> ALLOWED_BASES allow-list", - "status": "done" - }, - { - "id": "wp5-t3", - "title": "tests/ci-workflows.test.ts: dev2-go 통과 / 그 외 차단 / 복원 시나리오", - "status": "done" - }, - { - "id": "wp5-t4", - "title": "AGENTS.md/CONTRIBUTING.md/MAINTAINERS.md 서술을 실제 동작에 동기화", - "status": "done" - }, - { - "id": "wp5-t5", - "title": "독립 감사 + 변이 회귀 재실행 (신규 allow-list 우회 변이 포함)", - "status": "pending" - }, - { - "id": "wp5-t6", - "title": "CI 트리거 갭 해소: ci.yml/service-lifecycle.yml pull_request에 dev2-go 추가 + 드리프트 테스트", - "status": "done" - } - ], - "criteriaIds": [ - "c5" - ] - }, - { - "id": "wp6", - "title": "enforce-pr-target.yml 특성화 테스트 — 현재 동작을 고정 (승인 불필요)", - "status": "done", - "tasks": [ - { - "id": "wp6-t1", - "title": "현재 워크플로 계약을 tests/ci-workflows.test.ts에 고정", - "status": "done" - }, - { - "id": "wp6-t2", - "title": "050에서 관측한 상태 마커 정합성 결함을 테스트로 표현", - "status": "done" - }, - { - "id": "wp6-t3", - "title": "bun test tests/ci-workflows.test.ts 통과 확인", - "status": "done" - } - ], - "criteriaIds": [ - "c6" - ] - }, - { - "id": "wp7", - "title": "특성화 테스트 하드닝 — 감사 2/3라운드가 찾은 우회 봉쇄 (wp6 조기 종료 후속)", - "status": "done", - "tasks": [ - { - "id": "wp7-t1", - "title": "감사 2라운드가 찾은 6가지 우회 봉쇄 (잡 인벤토리/잡 권한/스텝 수/블록 주석/복원 분기/pull_number 바인딩)", - "status": "done" - }, - { - "id": "wp7-t2", - "title": "알려진 변이 10종 재주입 검증, 워크플로 원복 확인", - "status": "done" - }, - { - "id": "wp7-t3", - "title": "감사 3라운드 (독립 서브에이전트, 새 각도)", - "status": "done" - }, - { - "id": "wp7-t4", - "title": "3라운드 지적 반영 후 커밋 및 dev 푸시", - "status": "done" - }, - { - "id": "wp7-t5", - "title": "감사 3라운드 FAIL(14건) → 부정 목록을 허용 목록으로 반전", - "status": "done" - }, - { - "id": "wp7-t6", - "title": "감사 4라운드 FAIL(12건, 전부 스크립트 내부) → 실행 하네스 도입", - "status": "done" - }, - { - "id": "wp7-t7", - "title": "감사 5라운드 FAIL(6건, 하네스 충실도) → 하네스를 실제 러너에 맞춤 + 시나리오 3개", - "status": "done" - }, - { - "id": "wp7-t8", - "title": "감사 6~10라운드 대응 (하네스 충실도 + 계약 어설션)", - "status": "done" - }, - { - "id": "wp7-t9", - "title": "감사 11~14라운드 대응 (트리거 모양, 도달 가능 상태, 버전 스큐, 스키마 truthiness 계약)", - "status": "done" - } - ], - "criteriaIds": [ - "c7" - ] - }, - { - "id": "wp8", - "title": "게이트를 main으로 승격 (fast-forward) — pull_request_target은 기본 브랜치 워크플로를 실행하므로 이것 없이는 발효되지 않음", - "status": "deferred", - "tasks": [ - { - "id": "wp8-t1", - "title": "dev HEAD b4a9fe5c 호스팅 CI success 확인 (Cross-platform CI + Service lifecycle)", - "status": "deferred" - }, - { - "id": "wp8-t2", - "title": "main..dev 98커밋 범위에 릴리스 파이프라인/보안 리뷰 대상 변경이 있는지 확인", - "status": "deferred" - }, - { - "id": "wp8-t3", - "title": "git push origin b4a9fe5c:main (fast-forward, 로컬 체크아웃은 dev 유지)", - "status": "deferred" - }, - { - "id": "wp8-t4", - "title": "승격 후 원격 main의 enforce-pr-target.yml에 ALLOWED_BASES 존재 확인 + c5 met 기록", - "status": "deferred" - } - ], - "criteriaIds": [], - "note": "사용자가 세션 범위를 dev로 좁힘(\"일단 dev에서만 처리해\"). 사전 감사도 DO NOT PROMOTE 판정: 승격 범위가 게이트 한정이 아니라 main..dev 98커밋/9172줄이고, main 푸시는 deploy-docs(실제 Pages 배포)를 트리거한다. 메인테이너의 명시적 수용이 선행 조건. 계획서: devlog/_plan/260727_governance_intake/070_main_promotion.md", - "blocker": "EXTERNAL: main 승격은 (a) 사용자가 이번 세션 범위에서 제외했고(\"일단 dev에서만 처리해\"), (b) 사전 감사가 DO NOT PROMOTE 판정을 냈다 — 승격 범위가 게이트 한정이 아니라 main..dev 98커밋/9172줄이며 main 푸시는 deploy-docs(실제 GitHub Pages 배포)를 트리거하므로 메인테이너의 명시적 수용이 선행 조건이다. 에이전트 권한으로 해소할 수 없다." - }, - { - "id": "wp9", - "title": "ocx account 인증 코드를 셸 인자에서 stdin으로 (WP8 사전 감사 High 지적)", - "status": "done", - "tasks": [ - { - "id": "wp9-t1", - "title": "stdin 읽기 헬퍼 + 주입 가능한 deps (runtime-api.ts)", - "status": "done" - }, - { - "id": "wp9-t2", - "title": "account code 위치 인자 / login --code를 선택으로, 기본 stdin, '-'는 명시적 stdin", - "status": "done" - }, - { - "id": "wp9-t3", - "title": "인자 사용 시 stderr 경고 (값은 절대 에코 금지)", - "status": "done" - }, - { - "id": "wp9-t4", - "title": "tests/cli-account.test.ts 회귀 + 독립 감사", - "status": "done" - } - ], - "criteriaIds": [] - } - ], - "criteria": [ - { - "id": "c1", - "scenario": "WP1 docs-only 사이클이 끝나면 devlog/_plan/260727_governance_intake/ 아래 000/010/020/030 문서가 존재하고 각각 변경 파일 경로와 정확한 문구를 담는다. 프로덕션 코드 변경은 0건이다.", - "expectedEvidence": "ls devlog/_plan/260727_governance_intake/ 출력 + git show --stat 으로 src/ 변경 0건 확인", - "capturedEvidence": "ls devlog/_plan/260727_governance_intake/ → 000_survey.md, 010_branch_policy.md, 011_audit_round1.md, 012_audit_round2.md, 013_audit_round3_and_scope_split.md, 020_pr_split.md, 030_maintainer_wibias.md, 040_pr_target_gate.md (8개). git diff --name-only edf3b2c6..HEAD | rg -v '^devlog/' | wc -l → 0 (프로덕션 코드 변경 0건). 커밋 f43ead8c..e0d600f3 (9개). 독립 감사 6회: 1~5차 FAIL, 6차 NEAR-PASS.", - "status": "met" - }, - { - "id": "c2", - "scenario": "AGENTS.md/CONTRIBUTING.md/MAINTAINERS.md 세 파일 모두에서 dev2-go가 정식 통합선으로 문서화되고, 포팅/리베이스 PR이 환영 대상으로 명시되며, 현재 자동화가 dev2-go PR에 [WRONG BRANCH]를 붙인다는 사실이 숨겨지지 않고 적힌다. .github/ 변경은 0건이다.", - "expectedEvidence": "rg -n 'dev2-go' 세 파일 + rg -n -i 'porting|rebase' + rg -n 'WRONG BRANCH' + git diff --name-only에 .github/ 없음", - "capturedEvidence": "rg -c dev2-go AGENTS.md/CONTRIBUTING.md/MAINTAINERS.md → 2/1/2. rg -c -i \"porting|rebase\" AGENTS.md CONTRIBUTING.md → 1/1. rg -c \"WRONG BRANCH\" → 2/1/1 (세 파일 모두 현 자동화 동작 명시). git diff --name-only | rg \"^\\.github/\" → 0건 (워크플로 미변경). docs-site 5개 로케일(en/ko/ja/ru/zh-cn) 각 dev2-go 1건. 커밋 a37cc55b, e4062e32. 독립 감사 2라운드 VERDICT: PASS.", - "status": "met" - }, - { - "id": "c3", - "scenario": "지정된 PR이 두 개의 독립적인 변경으로 분리되고, 각각 bun run typecheck 통과 및 관련 테스트 통과가 증명된다.", - "expectedEvidence": "git log --oneline 분리 브랜치 + bun run typecheck exit 0 + bun test 관련 파일 결과", - "capturedEvidence": "SRC=d93b46932b58b963ea5dfa27dee5c63f3a7b0f2a 고정. B(codex/catalog-written-signal, PR #526): typecheck exit 0, 22 pass 0 fail, 6파일. A(codex/app-server-restart, PR #527): typecheck exit 0, 22 pass 1 skip 0 fail, 15파일. 합집합 검증 diff pr518-manifest split-union → IDENTICAL (21파일). SRC 불변 재확인. 계획 경계 가정 3회 실측 정정: config-routes.ts, gui i18n syncStaleHint, codex-models-cache-invalidate.test.ts 전부 A 소속.", - "status": "met" - }, - { - "id": "c4", - "scenario": "MAINTAINERS.md의 Current maintainers 표에 @Wibias 행이 있고 .github/CODEOWNERS의 기본 리뷰어에 @Wibias가 포함되며, Maintainer changes 절차 이행이 기록된다.", - "expectedEvidence": "rg -n 'Wibias' MAINTAINERS.md .github/CODEOWNERS 출력", - "capturedEvidence": "rg -c Wibias MAINTAINERS.md → 2 (표 행 + change log). rg -c Wibias .github/CODEOWNERS → 1 (기본 리뷰어 * 규칙만). 보안 경로 미포함 검증 → 0건. MAINTAINERS.md 표 3행. 감사 확인: CODEOWNERS 마지막 매칭 우선 규칙으로 @Wibias는 /src/oauth/, /.github/, /scripts/release.ts, /MAINTAINERS.md, /SECURITY.md, /package.json, /bun.lock의 코드오너가 아님(GitHub API errors:[]). 커밋 01e831d0, 71e43d9c, 491373f3.", - "status": "met" - }, - { - "id": "c5", - "scenario": "enforce-pr-target.yml이 dev2-go PR을 정당하게 통과시키고, 그 외 타깃은 차단·복원 경로가 회귀 테스트로 덮이며, PR을 받는 브랜치는 CI 트리거도 함께 받는다. actor 검증 + head SHA 바인딩은 040 문서의 근거 셋에 따라 채택하지 않았다 (강제력 없는 자동 판정 대신 리뷰가 잡는다). main 승격으로 실제 발효된다.", - "expectedEvidence": "워크플로 diff + 신규 테스트 통과 출력 + main 승격 확인", - "capturedEvidence": "커밋: d761e880 (ALLOWED_BASES allow-list, 워크플로 5지점) / 10b1d2aa (문서 8종, 5개 로케일 포함) / 5229717b (ci.yml+service-lifecycle.yml PR 트리거에 dev2-go) / 76c25710 (하네스 node24 충실도 + 트리거 키집합 고정 + live base 코멘트 어설션) / 2b03e908 (ci.yml paths 양쪽 트리거 정확 집합 고정) / 99679376 (040 문서에 15~17라운드 기록). 검증: bun x tsc --noEmit exit 0; bun run test 4945 pass / 0 fail / 24352 expect(); bun test tests/ci-workflows.test.ts 52 pass / 0 fail / 548 expect(). 변이 회귀: mut16 12/12, mut17 5/5, mut18 5/5 (봉쇄 전 5종 전부 SURVIVED), mut20 8/8 (봉쇄 전 4종 SURVIVED) 전부 CAUGHT. 독립 감사 3라운드: 15=FAIL(5건) -> 76c25710, 16=FAIL(1건) -> 2b03e908, 17=PASS (12/12 CAUGHT, 무해 생존 2건 근거 기록). 미충족 잔여: main 승격 (승인 완료, 다음 work-phase). [세션 종료 시점] dev 쪽 조건은 전부 충족(워크플로 diff + ci-workflows 테스트 통과 + 변이 회귀 전건 CAUGHT). 남은 조건인 'main 승격으로 발효'만 미충족이며, 이는 사용자가 이번 세션 범위에서 제외했다. origin/main은 origin/dev의 조상이라 fast-forward 가능한 상태로 남아 있다.", - "status": "deferred", - "blocker": "EXTERNAL: 'main 승격으로 발효' 조건만 미충족. 승격 권한/수용은 메인테이너 결정이며 사용자가 범위에서 제외했다." - }, - { - "id": "c6", - "scenario": "tests/ci-workflows.test.ts에 enforce-pr-target.yml 테스트가 존재하고, 현재 동작(pull_request_target 트리거 집합, EXPECTED_BASE, 제목 prefix, draft 전환, checkout/PR코드실행 없음, 최소 권한)을 고정한다. WP5에서 게이트를 바꿀 때 이 테스트가 회귀 그물이 된다.", - "expectedEvidence": "bun test tests/ci-workflows.test.ts 통과 출력 + rg enforce-pr-target tests/ci-workflows.test.ts", - "capturedEvidence": "tests/ci-workflows.test.ts에 PR target enforcement 테스트 3개(rg -c → 3). bun test → 14 pass 0 fail 282 expect(). bun run typecheck exit 0. 4라운드 적대적 변이 감사: 1차 FAIL(문자열 매칭 구멍 3), 2차 FAIL(YAML 문법 우회 5), 3차 FAIL(함수 shadow), 4차 PASS. 최종적으로 Bun.YAML 파싱 + quote-aware 주석 제거 + 함수 선언 개수/mutation 결속. 통과하는 변이는 전부 고의적 무력화만 남음.", - "status": "met" - }, - { - "id": "c7", - "scenario": "enforce-pr-target.yml 특성화 테스트가 알려진 모든 우회 변이를 잡는다", - "expectedEvidence": "변이 주입 스크립트 출력에서 알려진 변이 전부 CAUGHT, survived=none. 기준선 bun test 통과, bun run typecheck 오류 0. 독립 감사(gpt-5.6-terra) verdict PASS 또는 NEAR-PASS.", - "capturedEvidence": "변이 회귀 12개 스크립트 131종 전수: 알려진 무해 6종(context-eventname, core-tostring, promise-identity, err-tostringtag, response-headers, comment-after-write)과 들여쓰기 no-op 2종 외 전부 CAUGHT, git status --short .github/ clean. 기준선 bun test tests/ci-workflows.test.ts -> 44 pass / 0 fail / 485 expect(). bun run test -> 4937 pass / 0 fail / 24289 expect() exit 0. bun x tsc --noEmit exit 0. 독립 감사(gpt-5.6-terra, agent 019fa18f) 14라운드 verdict NEAR-PASS, 잔여 지적 1건은 dbd558e1로 봉쇄 후 CAUGHT 확인.", - "status": "met" - }, - { - "id": "c9", - "scenario": "ocx account login/code가 인증 코드를 셸 히스토리와 프로세스 목록에 남기지 않는 경로를 기본으로 제공하고, 인자 경로를 쓰면 값을 노출하지 않는 경고가 나온다.", - "expectedEvidence": "테스트 통과 출력 + 인자 경고 어설션 + 값 미노출 어설션", - "capturedEvidence": "구현: d4dfa24e(stdin 기본 경로), e7f5bef5(공백 문법 redact + 중복 --code 거부), d3dc9d79(미지 플래그를 코드로 먹지 않음), f10c5060(비밀 옵션 뒤 토큰은 모양 무관 은닉, -- 구분자 건너뜀, account code 잔여 위치 인자 은닉, 종료된 stdin 즉시 실패), 1de43286(CR/LF 줄바꿈 고정). 검증: bun run test 4965 pass / 0 fail / 24429 expect(); bun x tsc --noEmit exit 0; bun test tests/cli-account.test.ts 62 pass / 0 fail / 276 expect(). 변이: mut23 10/10 CAUGHT, mut21 9/9 CAUGHT, mut22 4/4 CAUGHT. 독립 감사 2라운드: R2 FAIL(High 2 + Medium 1 + Low 1, 전부 프로브로 실측 재현) → 수정 → R3 VERDICT: PASS. 잔여: `--code --nope`는 --nope를 값으로 간주해 은닉(사용 오류 메시지는 --code를 여전히 지목) — 진단성보다 노출 차단 우선.", - "status": "met" - } - ], - "host": { - "armed": false, - "armedAt": null, - "source": "none" - } -} \ No newline at end of file diff --git a/.codexclaw/goalplans/opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr/ledger.jsonl b/.codexclaw/goalplans/opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr/ledger.jsonl deleted file mode 100644 index 4cabb8387..000000000 --- a/.codexclaw/goalplans/opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr/ledger.jsonl +++ /dev/null @@ -1,21 +0,0 @@ -{"ts":"2026-07-27T00:41:29.353Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"created","detail":"init objective=\"opencodex 저장소 거버넌스 및 기여 정책 4건을 PABCD 다중 work-phase 루프로 처리한다. WP1 문서화 우선 사이클, WP2 dev2-go 기반 PR 및 포팅/리베이스 PR 허용 정책, WP3 대상 PR 두 개로 분할, WP4 Wibias 메인테이너 추가. 검증은 실제 파일과 명령 출력으로만. 서브에이전트는 gpt-5.6-terra medium 일반 티어 활용.\" criteria=0"} -{"ts":"2026-07-27T01:18:20.834Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp1"} -{"ts":"2026-07-27T01:18:20.834Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp2"} -{"ts":"2026-07-27T01:26:02.696Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp2"} -{"ts":"2026-07-27T01:26:02.696Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp4"} -{"ts":"2026-07-27T01:32:20.967Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp4"} -{"ts":"2026-07-27T01:32:20.967Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp3"} -{"ts":"2026-07-27T01:45:55.340Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp3"} -{"ts":"2026-07-27T01:45:55.340Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp5"} -{"ts":"2026-07-27T02:09:28.671Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp6"} -{"ts":"2026-07-27T02:09:28.671Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp5"} -{"ts": "2026-07-27T02:12:28.095252Z", "slug": "opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr", "event": "workphase_started", "detail": "started wp7 (P-phase amendment: wp6 closed before the round-2 audit returned FAIL; hardening is the next unit, LOOP-UNIT-CHAIN-01)"} -{"ts":"2026-07-27T02:14:19.004Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp7"} -{"ts":"2026-07-27T02:14:19.004Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp5"} -{"ts":"2026-07-27T03:11:25Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"re-activated wp7: rounds 7-10 hardening was the unit actually worked; wp5 remains blocked on user approval and was parked back to pending"} -{"ts":"2026-07-27T03:39:57.830Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp7"} -{"ts":"2026-07-27T03:39:57.830Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp5"} -{"ts":"2026-07-27T03:40:13Z","event":"work-phase-complete","workPhaseId":"wp7","criterion":"c7","status":"met","evidence":"rounds 11-14 audited; 131 mutation variants CAUGHT except 6 proven-harmless; suite 4937 pass/0 fail; typecheck 0; commits 5f7afc21, dbd558e1"} -{"ts":"2026-07-27T04:17:39.364Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp5"} -{"ts":"2026-07-27T04:58:54.547Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_done","detail":"closed wp9"} -{"ts":"2026-07-27T04:58:54.547Z","slug":"opencodex-4-pabcd-work-phase-wp1-wp2-dev2-go-pr","event":"workphase_started","detail":"started wp8"} diff --git a/.codexclaw/goalplans/opencodex-live-unfinished-issues-and-prs-triage/goalplan.json b/.codexclaw/goalplans/opencodex-live-unfinished-issues-and-prs-triage/goalplan.json deleted file mode 100644 index bf89c77b1..000000000 --- a/.codexclaw/goalplans/opencodex-live-unfinished-issues-and-prs-triage/goalplan.json +++ /dev/null @@ -1,365 +0,0 @@ -{ - "objective": "OpenCodex live unfinished issues and PRs triage against current GitHub state. Scope: repository lidge-jun/opencodex, dev target only, worktree . First work-phase is docs-only manifest/devlog: fetch live open PRs and open issues via gh, record current number/title/state/base/head/mergeable/checks/reviews/labels, classify every item into merge now, takeover-fix, comment/request-changes, needs-author-rebase, needs-human/security, later/enhancement, upstream-tracking, or close, and produce a priority table. Later work-phases process exactly one PR or one issue per full PABCD cycle. Allowed: directly fix bugfix/simple safe items on dev, create PRs, wait for CI, squash merge, and close linked issues when live evidence proves safety. Out of scope: main/preview/release branches, automatic merge of security/auth/permission/data-migration/privilege-boundary work, and unapproved GUI/UX decisions except the approved OpenRouter Free separate-provider direction. Terminal outcomes: DONE when all safe live items are processed and manifest evidence is current; NOOP when an item needs no action after live check; NEEDS_HUMAN or UNSAFE for risk-bound/security/UX-decision items; BLOCKED for external author/rebase/CI or upstream dependency; BUDGET_EXHAUSTED only if explicit runtime bounds are hit. Verification: gh live snapshots, code/diff review for candidate PRs, CI/check URLs where actions occur, comments/merge/close URLs for external state changes, and devlog evidence committed locally before completion.", - "slug": "opencodex-live-unfinished-issues-and-prs-triage", - "createdAt": "2026-07-27T10:35:02.724Z", - "updatedAt": "2026-07-27T12:25:15.899Z", - "activeWorkPhaseId": "WP8", - "workPhases": [ - { - "id": "WP0", - "title": "Live GitHub triage manifest and priority table", - "status": "done", - "tasks": [ - { - "id": "WP0-T1", - "title": "Refresh dev branch/worktree state from origin/dev without touching main/preview/release branches", - "status": "done" - }, - { - "id": "WP0-T2", - "title": "Query all current open PRs and issues via gh with state/base/head/mergeable/checks/reviews/labels", - "status": "done" - }, - { - "id": "WP0-T3", - "title": "Write numbered docs-first devlog manifest and priority table", - "status": "done" - }, - { - "id": "WP0-T4", - "title": "Append follow-up one-item work-phases for safe merge/fix/close candidates discovered by the manifest", - "status": "done" - } - ], - "criteriaIds": [ - "C-WP0-LIVE-MANIFEST" - ] - }, - { - "id": "WP1", - "title": "PR #526 catalog write signal rebase and coverage decision", - "status": "done", - "tasks": [ - { - "id": "WP1-T1", - "title": "Re-check PR #526 live head, checks, diff, and independent review", - "status": "done" - }, - { - "id": "WP1-T2", - "title": "Take over rebase/tests or leave a documented blocker for PR #526 only", - "status": "done" - } - ], - "criteriaIds": [ - "C-WP1-PR526" - ] - }, - { - "id": "WP9", - "title": "Dev baseline checkout blocker from devlog gitlinks", - "status": "done", - "tasks": [ - { - "id": "WP9-T1", - "title": "Confirm current dev gitlink/.gitmodules mismatch and whether devlog chase checkouts are required tracked inputs", - "status": "done" - }, - { - "id": "WP9-T2", - "title": "Remove or repair only the accidental devlog gitlinks on a dev-target branch", - "status": "done" - }, - { - "id": "WP9-T3", - "title": "Verify checkout/submodule commands and open/merge the dev baseline CI fix before rerunning PR #526", - "status": "done" - } - ], - "criteriaIds": [ - "C-WP9-DEVLOG-GITLINK" - ] - }, - { - "id": "WP10", - "title": "Dev baseline Desktop 3P target-platform path fix", - "status": "done", - "tasks": [ - { - "id": "WP10-T1", - "title": "Confirm hosted Windows failure and local Desktop 3P path resolver cause", - "status": "done" - }, - { - "id": "WP10-T2", - "title": "Patch target-platform-specific path joining and regression tests on a dev-target branch", - "status": "done" - }, - { - "id": "WP10-T3", - "title": "Verify locally, push PR, wait for hosted checks, and squash merge if green", - "status": "done" - } - ], - "criteriaIds": [ - "C-WP10-DESKTOP3P-PATH" - ] - }, - { - "id": "WP2", - "title": "PR #528 image bridge credential-origin request changes", - "status": "pending", - "tasks": [ - { - "id": "WP2-T1", - "title": "Re-check PR #528 live head, checks, diff, and independent review", - "status": "pending" - }, - { - "id": "WP2-T2", - "title": "Post request-changes comment for credential-origin binding and stale checks", - "status": "pending" - } - ], - "criteriaIds": [ - "C-WP2-PR528" - ] - }, - { - "id": "WP11", - "title": "PR #526 final rerun and merge after dev baseline fixes", - "status": "done", - "tasks": [ - { - "id": "WP11-T1", - "title": "Rebase PR #526 branch onto current origin/dev after WP9/WP10", - "status": "done" - }, - { - "id": "WP11-T2", - "title": "Run local targeted verification and push the rebased PR head", - "status": "done" - }, - { - "id": "WP11-T3", - "title": "Wait for hosted checks and squash merge PR #526 if the latest head is clean", - "status": "done" - } - ], - "criteriaIds": [ - "C-WP11-PR526-FINAL" - ] - }, - { - "id": "WP3", - "title": "Issue #543 Claude queued_command support reply", - "status": "pending", - "tasks": [ - { - "id": "WP3-T1", - "title": "Verify existing debug capture switch and issue context", - "status": "pending" - }, - { - "id": "WP3-T2", - "title": "Post one maintainer comment requesting marker frames", - "status": "pending" - } - ], - "criteriaIds": [ - "C-WP3-ISSUE543" - ] - }, - { - "id": "WP4", - "title": "Issue #547 Claude Desktop custom-model visibility reply", - "status": "pending", - "tasks": [ - { - "id": "WP4-T1", - "title": "Verify issue #547 context and likely evidence gaps", - "status": "pending" - }, - { - "id": "WP4-T2", - "title": "Post one maintainer comment requesting generated config/log/profile evidence", - "status": "pending" - } - ], - "criteriaIds": [ - "C-WP4-ISSUE547" - ] - }, - { - "id": "WP5", - "title": "Issue #545 64-token classifier investigation", - "status": "pending", - "tasks": [ - { - "id": "WP5-T1", - "title": "Trace max_tokens/max_output_tokens path for Desktop 3P classifier requests", - "status": "pending" - }, - { - "id": "WP5-T2", - "title": "Fix if local bug is proven; otherwise comment with exact evidence needed", - "status": "pending" - } - ], - "criteriaIds": [ - "C-WP5-ISSUE545" - ] - }, - { - "id": "WP6", - "title": "PR #527 wrong-base handling", - "status": "done", - "tasks": [ - { - "id": "WP6-T1", - "title": "Re-check #527 after #526 decision", - "status": "pending" - }, - { - "id": "WP6-T2", - "title": "Retarget/request rebase or leave blocker; do not merge in same phase as #526", - "status": "pending" - } - ], - "criteriaIds": [ - "C-WP6-PR527" - ] - }, - { - "id": "WP7", - "title": "Issue #418 V2 custom delegation investigation", - "status": "done", - "tasks": [ - { - "id": "WP7-T1", - "title": "Read custom-parent/custom-child delegation code path and issue evidence", - "status": "done" - }, - { - "id": "WP7-T2", - "title": "Fix or comment with code pointers and required reproduction evidence", - "status": "done" - } - ], - "criteriaIds": [ - "C-WP7-ISSUE418" - ] - }, - { - "id": "WP8", - "title": "Issue #509 JS heap watchdog investigation", - "status": "in_progress", - "tasks": [ - { - "id": "WP8-T1", - "title": "Trace RSS/heap watchdog logic and issue evidence", - "status": "pending" - }, - { - "id": "WP8-T2", - "title": "Fix or comment with code pointers and blocker", - "status": "pending" - } - ], - "criteriaIds": [ - "C-WP8-ISSUE509" - ] - } - ], - "criteria": [ - { - "id": "C-WP0-LIVE-MANIFEST", - "scenario": "Live GitHub state has been refreshed and every open PR/open issue is classified into the requested triage buckets.", - "expectedEvidence": "devlog/_plan/260727_live_unfinished_triage/ numbered manifest files with gh snapshot timestamp, PR/issue lists, classification rationale, and priority table.", - "capturedEvidence": "WP0 closed by cxc D at 2026-07-27T10:46:14Z. Commit 58b247ac records devlog/_plan/260727_live_unfinished_triage. C check: open PR count 14 with numbers 355,424,429,447,461,491,493,495,498,512,526,527,528,533; open issue count 23 with numbers 42,92,95,177,178,201,241,294,386,401,414,415,417,418,425,462,476,509,521,540,543,545,547; manifest has unsafe merge-now entries 0.", - "status": "met" - }, - { - "id": "C-WP1-PR526", - "scenario": "PR #526 is either taken over with rebase/direct coverage or left open with a fresh blocker comment.", - "expectedEvidence": "live PR head/checks, independent review verdict, test evidence or comment URL.", - "capturedEvidence": "WP1 closed by cxc D at 2026-07-27T11:01:22Z. Same-repo branch codex/catalog-written-signal was rebased onto origin/dev@7fcaa9119253d010393cb457427a2868cd935718 and pushed at 43d0efff4569711ed192e09d4d87b62fc803153c. Local pre-push passed typecheck, lint:gui, bun test 5051 pass/0 fail/24881 assertions, privacy scan, and React Doctor. Hosted Issue quality tests/test failed before tests during checkout with fatal missing .gitmodules mapping for devlog/_chase/_cca; origin/dev has the same gitlink mismatch. Merge held pending WP9 baseline checkout repair.", - "status": "met" - }, - { - "id": "C-WP9-DEVLOG-GITLINK", - "scenario": "The dev baseline checkout blocker caused by tracked devlog gitlinks without .gitmodules mappings is fixed or proven to require human direction.", - "expectedEvidence": "git tree/submodule evidence, commit/PR URL, local checkout/submodule verification, and hosted CI rerun evidence if pushed.", - "capturedEvidence": "WP9 closed by cxc D at 2026-07-27T11:11:58Z. Branch codex/devlog-gitlink-ci-fix removed exactly three accidental 160000 gitlinks without .gitmodules mappings: devlog/_chase/_cca, devlog/_chase/_litellm, and devlog/_fin/opencode-cursor. Local verification: git ls-files -s found no 160000 entries; git submodule status --recursive exit 0; git diff --check origin/dev exit 0. PR #550 https://github.com/lidge-jun/opencodex/pull/550 passed hosted checks and was squash-merged into origin/dev at ff831858388179d3f76f4dd7c119d84470214fa6.", - "status": "met" - }, - { - "id": "C-WP10-DESKTOP3P-PATH", - "scenario": "The dev baseline Desktop 3P path resolver returns target-platform separators across hosted OSes and unblocks PR #526 CI.", - "expectedEvidence": "code pointers, targeted Bun test output, TypeScript output, diff-check output, PR URL, hosted check evidence, and merge SHA if green.", - "capturedEvidence": "WP10 branch codex/desktop3p-path-windows-ci commit f6d2881dd422830eece502e0ba8de493205fe9d1 patched src/claude/desktop-3p-paths.ts to use target-platform posix/win32 joins and updated tests/desktop-3p.test.ts plus tests/claude-desktop-config-path.test.ts. Local verification: bun test tests/desktop-3p.test.ts tests/claude-desktop-config-path.test.ts = 30 pass/0 fail; bun x tsc --noEmit exit 0; git diff --check origin/dev exit 0. Prepush full gate: 5047 pass/0 fail, privacy scan passed, GUI doctor skipped. Independent C review Aquinas PASS. PR #552 https://github.com/lidge-jun/opencodex/pull/552 passed hosted CodeRabbit, enforce-target, label, react-doctor, ubuntu-latest, macos-latest, windows-latest, and all npm-global matrix jobs; squash-merged at 2026-07-27T11:43:58Z as origin/dev 7c74e0a22ec96dd5849d3d7253758f0ab15d9737. Remote topic branch deleted.", - "status": "met" - }, - { - "id": "C-WP2-PR528", - "scenario": "PR #528 receives a request-changes comment for credential-origin binding and stale checks.", - "expectedEvidence": "live PR head/checks, independent review verdict, and comment/review URL.", - "capturedEvidence": "", - "status": "open" - }, - { - "id": "C-WP11-PR526-FINAL", - "scenario": "PR #526 is rebased after dev baseline blockers, verified on the latest head, and merged if clean.", - "expectedEvidence": "rebase head SHA, local test/typecheck/diff-check output, pushed head SHA, hosted check list, merge commit URL/SHA or blocker evidence.", - "capturedEvidence": "WP11 closed by cxc D at 2026-07-27T12:05:44Z merge evidence. Branch codex/catalog-written-signal was rebased on origin/dev@7c74e0a22ec96dd5849d3d7253758f0ab15d9737 and pushed at ce716cc117ab23e4420c8c9fe860959968f66cdc. Local targeted verification passed: bun test tests/codex-refresh.test.ts tests/codex-sync-api.test.ts tests/injection-model-api.test.ts = 24 pass/0 fail/112 assertions; bun x tsc --noEmit exit 0; git diff --check origin/dev exit 0. Pre-push full gate passed: bun run test = 5051 pass/0 fail/24881 assertions, privacy scan passed, GUI doctor skipped. Hosted checks on PR #526 head ce716cc117ab23e4420c8c9fe860959968f66cdc all succeeded: CodeRabbit, label, react-doctor, ubuntu-latest, macos-latest, windows-latest, and npm-global ubuntu/macos/windows. Pre-merge gate passed: remote dev stayed 7c74e0a22ec96dd5849d3d7253758f0ab15d9737, PR head matched ce716cc117ab23e4420c8c9fe860959968f66cdc, mergeable MERGEABLE/CLEAN and REST mergeable_state clean. PR #526 https://github.com/lidge-jun/opencodex/pull/526 was squash-merged into dev at 2026-07-27T12:05:44Z as 9dd3c42dae2e7feda3581c6d477cf5a0d6e646bf. Remote codex/catalog-written-signal branch was preserved at ce716cc117ab23e4420c8c9fe860959968f66cdc.", - "status": "met" - }, - { - "id": "C-WP3-ISSUE543", - "scenario": "Issue #543 receives a concrete support reply naming the existing debug capture switch and requested evidence.", - "expectedEvidence": "comment URL plus code/docs pointers for `ocx debug claude` or `OCX_CLAUDE_DEBUG=1`.", - "capturedEvidence": "", - "status": "open" - }, - { - "id": "C-WP4-ISSUE547", - "scenario": "Issue #547 receives a concrete evidence request for Claude Desktop custom-model visibility.", - "expectedEvidence": "comment URL plus requested config/log/profile evidence.", - "capturedEvidence": "", - "status": "open" - }, - { - "id": "C-WP5-ISSUE545", - "scenario": "Issue #545 is narrowed to a proven local fix or exact remaining evidence need.", - "expectedEvidence": "code pointers, test/command output if fixed, PR/comment URL.", - "capturedEvidence": "", - "status": "open" - }, - { - "id": "C-WP6-PR527", - "scenario": "PR #527 wrong-base state is resolved or documented after PR #526 decision.", - "expectedEvidence": "live base/check state and retarget/request-rebase/comment URL.", - "capturedEvidence": "WP6 closed by cxc D at 2026-07-27T12:16:35Z. Live PR #527 remained OPEN on base codex/catalog-written-signal@ce716cc117ab23e4420c8c9fe860959968f66cdc with head codex/app-server-restart@a64aa585630f664a83c25253497a62810133e832, mergeable CONFLICTING and mergeStateStatus DIRTY. PR #526 had merged to dev as 9dd3c42dae2e7feda3581c6d477cf5a0d6e646bf. Topology showed #527 still carried duplicate old #526 commit 1ba588eff663a5be846a8723b90a452dca8cd04c; merge-tree against current dev conflicted in tests/codex-refresh.test.ts and tests/injection-model-api.test.ts. Independent A review returned GO-WITH-FIXES and the blocker was folded into the comment plan. Posted maintainer comment https://github.com/lidge-jun/opencodex/pull/527#issuecomment-5091163284 requesting a clean rebuild from current dev, dropping 1ba588e, preserving the Grok diagnostic, and retargeting after rebuild. No retarget, push, merge, or branch deletion was performed.", - "status": "met" - }, - { - "id": "C-WP7-ISSUE418", - "scenario": "Issue #418 has a proven fix or a fresh investigation comment with code pointers.", - "expectedEvidence": "test/command output if fixed, otherwise comment URL plus code pointers.", - "capturedEvidence": "WP7 closed as NOOP/comment-request-changes at 2026-07-27. Live issue #418 remains OPEN with bug label. Existing comments already satisfy this phase's intended maintainer action: owner request https://github.com/lidge-jun/opencodex/issues/418#issuecomment-5069836945 asks for raw provider tool-call, Responses event, and child lifecycle trace; reporter acknowledgement https://github.com/lidge-jun/opencodex/issues/418#issuecomment-5070272410 says same-run failing spawn_agent trace is still unavailable until usage limit clears; collaborator cross-link https://github.com/lidge-jun/opencodex/issues/418#issuecomment-5085535548 keeps #418 separate from #92 pending the three-boundary capture. Code review found no proven local argument-drop path: parser copies tool parameters at src/responses/parser.ts:134-139; OpenAI-compatible adapter forwards parameters at src/adapters/openai-chat.ts:432-445; provider function.arguments fragments are accumulated at src/adapters/openai-chat.ts:749-768; bridge forwards deltas at src/bridge.ts:610-617 and materializes {} only when accumulated bytes are empty at src/bridge.ts:366-375. No GitHub comment, close, merge, or code change was performed.", - "status": "met" - }, - { - "id": "C-WP8-ISSUE509", - "scenario": "Issue #509 heap watchdog gap has a proven fix or a fresh investigation comment with code pointers.", - "expectedEvidence": "test/command output if fixed, otherwise comment URL plus code pointers.", - "capturedEvidence": "", - "status": "open" - } - ], - "host": { - "armed": true, - "armedAt": "2026-07-27T10:35:02.725Z", - "source": "freeze" - } -} diff --git a/.codexclaw/goalplans/opencodex-live-unfinished-issues-and-prs-triage/ledger.jsonl b/.codexclaw/goalplans/opencodex-live-unfinished-issues-and-prs-triage/ledger.jsonl deleted file mode 100644 index 07d4aeae8..000000000 --- a/.codexclaw/goalplans/opencodex-live-unfinished-issues-and-prs-triage/ledger.jsonl +++ /dev/null @@ -1,24 +0,0 @@ -{"ts":"2026-07-27T10:35:02.725Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"created","detail":"init objective=\"OpenCodex live unfinished issues and PRs triage against current GitHub state. Scope: repository lidge-jun/opencodex, dev target only, worktree . First work-phase is docs-only manifest/devlog: fetch live open PRs and open issues via gh, record current number/title/state/base/head/mergeable/checks/reviews/labels, classify every item into merge now, takeover-fix, comment/request-changes, needs-author-rebase, needs-human/security, later/enhancement, upstream-tracking, or close, and produce a priority table. Later work-phases process exactly one PR or one issue per full PABCD cycle. Allowed: directly fix bugfix/simple safe items on dev, create PRs, wait for CI, squash merge, and close linked issues when live evidence proves safety. Out of scope: main/preview/release branches, automatic merge of security/auth/permission/data-migration/privilege-boundary work, and unapproved GUI/UX decisions except the approved OpenRouter Free separate-provider direction. Terminal outcomes: DONE when all safe live items are processed and manifest evidence is current; NOOP when an item needs no action after live check; NEEDS_HUMAN or UNSAFE for risk-bound/security/UX-decision items; BLOCKED for external author/rebase/CI or upstream dependency; BUDGET_EXHAUSTED only if explicit runtime bounds are hit. Verification: gh live snapshots, code/diff review for candidate PRs, CI/check URLs where actions occur, comments/merge/close URLs for external state changes, and devlog evidence committed locally before completion.\" criteria=0"} -{"ts":"2026-07-27T10:46:14.884Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP0"} -{"ts":"2026-07-27T10:46:14.884Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP1"} -{"ts":"2026-07-27T11:01:22.413Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP1"} -{"ts":"2026-07-27T11:01:22.413Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP2"} -{"ts":"2026-07-27T11:11:58.894Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP9"} -{"ts":"2026-07-27T11:11:58.894Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP2"} -{"ts":"2026-07-27T11:46:27.214Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP10"} -{"ts":"2026-07-27T11:46:27.214Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP2"} -{"ts":"2026-07-27T11:47:00.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP11"} -{"ts":"2026-07-27T12:05:44.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"task_done","detail":"WP11-T1 rebased codex/catalog-written-signal onto origin/dev@7c74e0a22ec96dd5849d3d7253758f0ab15d9737"} -{"ts":"2026-07-27T12:05:44.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"task_done","detail":"WP11-T2 verified locally and pushed ce716cc117ab23e4420c8c9fe860959968f66cdc"} -{"ts":"2026-07-27T12:05:44.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"task_done","detail":"WP11-T3 hosted checks passed and PR #526 squash-merged as 9dd3c42dae2e7feda3581c6d477cf5a0d6e646bf"} -{"ts":"2026-07-27T12:05:44.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"criterion_met","detail":"C-WP11-PR526-FINAL evidence in devlog/_plan/260727_wp11-pr526-final-rerun-merge/011_phase1_evidence.md"} -{"ts":"2026-07-27T12:05:44.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP11"} -{"ts":"2026-07-27T12:05:44.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP2"} -{"ts":"2026-07-27T12:08:50.093Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP11"} -{"ts":"2026-07-27T12:08:50.093Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP3"} -{"ts":"2026-07-27T12:09:30.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"steering","detail":"corrected next active work-phase from WP3 to WP6 because PR #527 explicitly depends on the PR #526 decision; WP2 and WP3 remain pending"} -{"ts":"2026-07-27T12:09:30.000Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP6"} -{"ts":"2026-07-27T12:16:35.216Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP6"} -{"ts":"2026-07-27T12:16:35.216Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP7"} -{"ts":"2026-07-27T12:25:15.900Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_done","detail":"closed WP7"} -{"ts":"2026-07-27T12:25:15.900Z","slug":"opencodex-live-unfinished-issues-and-prs-triage","event":"workphase_started","detail":"started WP8"} diff --git a/.github/AGENTS.md b/.github/AGENTS.md new file mode 100644 index 000000000..fc363e66f --- /dev/null +++ b/.github/AGENTS.md @@ -0,0 +1,26 @@ +# GitHub automation instructions + +This file applies to `.github/` and inherits the repository-wide rules in `/AGENTS.md`. + +## Security boundary + +Every workflow, ownership, branch-enforcement, release, or repository-automation +change requires explicit security review under `MAINTAINERS.md`. + +## Workflow rules + +- Grant the minimum required `permissions`. +- Pin third-party actions to immutable full commit SHAs. +- Preserve the human-readable version comment beside each pinned action. +- Do not run untrusted pull-request code with secrets or write permissions. +- Treat `pull_request_target`, workflow dispatch, reusable workflows, artifacts, caches, and generated command input as trust boundaries. +- Do not broaden triggers, write permissions, token exposure, release eligibility, or publish capability without an explicit task requirement. +- Preserve cross-platform coverage where the workflow currently promises Linux, macOS, and Windows behavior. +- Keep branch-enforcement text synchronized with `AGENTS.md`, `MAINTAINERS.md`, and the public contributing guide. + +## Validation + +- Inspect the complete workflow diff, including event triggers, permissions, conditions, interpolation, and shell behavior. +- Run the local commands represented by changed workflow steps where possible. +- Run `bun run prepush` for CI, release, dependency, packaging, or cross-platform workflow changes. +- Do not claim the workflow itself passed until GitHub Actions reports success for the exact commit. diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 556398275..297343edf 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -14,5 +14,7 @@ /src/server/management-api.ts @lidge-jun @Ingwannu # Governance and security policy +/AGENTS.md @lidge-jun @Ingwannu +**/AGENTS.md @lidge-jun @Ingwannu /MAINTAINERS.md @lidge-jun @Ingwannu /SECURITY.md @lidge-jun @Ingwannu diff --git a/.github/ISSUE_TEMPLATE/config.yml b/.github/ISSUE_TEMPLATE/config.yml index d85f47ee5..07fd1d1cf 100644 --- a/.github/ISSUE_TEMPLATE/config.yml +++ b/.github/ISSUE_TEMPLATE/config.yml @@ -1,5 +1,8 @@ blank_issues_enabled: false contact_links: + - name: Report a security vulnerability (private) + url: https://github.com/lidge-jun/opencodex/security/advisories/new + about: Report undisclosed vulnerabilities privately to the maintainers. Do not open a public issue. - name: Security policy url: https://github.com/lidge-jun/opencodex/blob/main/SECURITY.md about: Read the supported-version and reporting guidance before sharing security-sensitive details. diff --git a/.github/scripts/enforce-pr-target.test.cjs b/.github/scripts/enforce-pr-target.test.cjs new file mode 100644 index 000000000..79c3dbd26 --- /dev/null +++ b/.github/scripts/enforce-pr-target.test.cjs @@ -0,0 +1,75 @@ +"use strict"; + +const fs = require("node:fs"); +const path = require("node:path"); +const { describe, it } = require("node:test"); +const assert = require("node:assert/strict"); + +describe("enforce-pr-target workflow", () => { + const workflowPath = path.join(__dirname, "../workflows/enforce-pr-target.yml"); + const workflow = fs.readFileSync(workflowPath, "utf8"); + + it("uses pull_request_target without checking out PR head code", () => { + assert.match(workflow, /pull_request_target:/); + assert.doesNotMatch( + workflow, + /ref:\s*\$\{\{\s*github\.event\.pull_request\.head/, + "enforcer must not check out untrusted PR head code", + ); + }); + + it("grants contents:write so draft GraphQL mutations work with GITHUB_TOKEN", () => { + // convertPullRequestToDraft / markPullRequestReadyForReview fail with + // "Resource not accessible by integration" when contents stays unset/read + // (seen on #626). Assert the real permissions block, not comment text + // that also mentions these scopes. + const permissionsBlock = workflow.match(/^permissions:\n((?:[ \t]+.+\n)+)/m); + assert.ok(permissionsBlock, "workflow must declare a top-level permissions block"); + const lines = permissionsBlock[1] + .split("\n") + .map((line) => line.trim()) + .filter(Boolean) + .sort(); + assert.deepEqual(lines, ["contents: write", "pull-requests: write"]); + }); + + it("fails the required check on a wrong base even if draft conversion fails", () => { + assert.match(workflow, /core\.setFailed\(/); + assert.match(workflow, /draftConversionFailed/); + assert.match(workflow, /Could not convert pull request to draft/); + }); + + it("soft-fails ready-for-review restoration the same way", () => { + assert.match(workflow, /readyConversionFailed/); + assert.match(workflow, /Could not mark pull request ready for review/); + }); + + it("listens for synchronize so rebase can clear ancestry failures", () => { + assert.match(workflow, /synchronize/); + }); + + it("checks out trusted default-branch scripts only (never PR head)", () => { + assert.match(workflow, /actions\/checkout@[0-9a-f]{40}/); + assert.match(workflow, /ref:\s*\$\{\{\s*github\.event\.repository\.default_branch\s*\}\}/); + assert.match(workflow, /sparse-checkout:\s*\.github\/scripts/); + assert.match(workflow, /persist-credentials:\s*false/); + assert.doesNotMatch(workflow, /ref:\s*\$\{\{\s*github\.event\.pull_request\.head/); + }); + + it("loads pr-quality via require from the checked-out scripts", () => { + assert.match(workflow, /pr-quality\.cjs/); + assert.match(workflow, /collectPrQualityFailures/); + }); + + it("strips stale WRONG BRANCH prefix on failure when base is corrected", () => { + const failureBlock = workflow.match( + /if \(failures\.length > 0\) \{([\s\S]*?)core\.setFailed\(/, + ); + assert.ok(failureBlock, "workflow must have a failure path"); + const failurePath = failureBlock[1]; + assert.match(failurePath, /shouldStripTitlePrefix/); + assert.match(failurePath, /!hasWrongBase/); + assert.match(failurePath, /titlePrefixedByBot = false/); + assert.match(failurePath, /pr\.title\.slice\(TITLE_PREFIX\.length\)/); + }); +}); diff --git a/.github/scripts/issue-quality.cjs b/.github/scripts/issue-quality.cjs index b2c778301..9dccb742f 100644 --- a/.github/scripts/issue-quality.cjs +++ b/.github/scripts/issue-quality.cjs @@ -25,19 +25,29 @@ function unwrapSingleEnclosingFence(text) { return match[2]; } -function isPlaceholderOnlyValue(raw) { - if (typeof raw !== "string") return false; +/** + * Shared strip/trim/unwrap used by placeholder and unusable-stand-in matchers. + * Returns null when the value is absent after normalisation. + */ +function normalizeRawSectionValue(raw) { + if (typeof raw !== "string") return null; let value = raw.replace(//g, "").trim(); - if (!value) return false; + if (!value) return null; - // A lone fenced block whose entire body is a placeholder is still placeholder - // text (e.g. ```text\nN/A\n```), not a real example. + // A lone fenced block whose entire body is a stand-in is still a stand-in + // (e.g. ```text\nN/A\n```), not a real example. const unwrapped = unwrapSingleEnclosingFence(value); if (unwrapped !== null) { value = unwrapped.trim(); - if (!value) return false; + if (!value) return null; } + return value; +} + +function isPlaceholderOnlyValue(raw) { + const value = normalizeRawSectionValue(raw); + if (value === null) return false; return PLACEHOLDER_ONLY_RE.test(value); } @@ -374,7 +384,10 @@ function detectIssueKind(issue) { // --------------------------------------------------------------------------- function isEmpty(text) { - return clean(text).length === 0; + const c = clean(text); + if (c.length === 0) return true; + // Stand-ins like "...", "…", "---" are not actionable report content. + return /^[\p{P}\p{S}\s]+$/u.test(c); } function allSameCanonical(sections) { @@ -395,6 +408,20 @@ function isPlaceholder(text) { return isPlaceholderOnlyValue(text); } +/** + * True when Version is an "I don't know" stand-in rather than an install id. + * Kept separate from PLACEHOLDER_ONLY_RE so legacy N/A / No response soft-pass + * behaviour is unchanged. + */ +const UNUSABLE_VERSION_RE = + /^[\s_*~`]*(?:unknown|unkown|uknown|don'?t\s+know|do\s+not\s+know|idk|dunno|not\s+sure|unsure|\?+|모름|잘\s*모름|모르겠(?:습니다|음)?|不明|わからない|分からない|不知道|不清楚|keine\s+ahnung|wei[sß]{1,2}\s+nicht)[\s_*~`]*[.!?]*$/i; + +function isUnusableVersion(raw) { + const value = normalizeRawSectionValue(raw); + if (value === null) return false; + return UNUSABLE_VERSION_RE.test(value); +} + const CJK_RE = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/gu; @@ -430,6 +457,16 @@ function isTooTerseFeatureSection(text) { return true; } +/** + * Bug Reproduction needs steps or concrete signals. A title-like phrase with + * no commands, paths, digits, or product keywords is not actionable. + */ +function isTooTerseBugReproduction(text) { + if (isEmpty(text) || isPlaceholder(text)) return false; + if (hasConcreteDetail(text)) return false; + return countWords(text) < 12; +} + /** * Check if raw section text is a placeholder-only variant without relying on * clean() first. Used to distinguish intentionally blank optional fields @@ -625,6 +662,8 @@ function validateIssue(issue) { const repro = extractSection(body, "Reproduction"); const version = extractSection(body, "Version"); const os = extractSection(body, "Operating system") ?? extractSection(body, "OS"); + // New Bug report template always includes Client or integration. + const isNewBugForm = extractSection(body, "Client or integration") !== null; if (isEmpty(summary) && isEmpty(repro)) { // Soft-pass substantial non-English / freeform structured reports once @@ -642,16 +681,65 @@ function validateIssue(issue) { reasons.push("Both Summary and Reproduction are empty."); guidance.push("Describe what happened and how to reproduce it."); } + } else { + // Each mapped field is required on its own — a filled Summary with an + // empty / ellipsis Reproduction (e.g. #598) must not pass. + if (isEmpty(summary)) { + reasons.push("Summary is empty."); + guidance.push("Describe what happened (the symptom or error)."); + } + if (isEmpty(repro)) { + reasons.push("Reproduction is empty."); + guidance.push("List the exact steps to reproduce the problem."); + } else if (!softPass && isTooTerseBugReproduction(repro)) { + reasons.push("Reproduction is too vague to act on."); + guidance.push("List exact steps, commands, and the observed failure — not only a short phrase."); + } } - // Required environment fields removed after submission. - // Only fire when the headings exist in the body (new form). Legacy bug - // reports never had Version or OS fields, so null means absent, not removed. - // Skip when the raw value is a "No response" placeholder -- the old form had - // both fields as optional, so legacy issues legitimately contain those headings - // with the GitHub placeholder. Only close when the field was actively cleared. - if (!softPass && version !== null && os !== null && isEmpty(version) && isEmpty(os) && - !isRawPlaceholder(version) && !isRawPlaceholder(os)) { + // Version "Unknown" / "모름" / "idk" is never actionable, on any form. + if (!softPass && version !== null && isUnusableVersion(version)) { + reasons.push("Version is missing or unknown."); + guidance.push("Report the installed `@bitkyc08/opencodex` version (for example `2.7.42`) or a commit SHA from `ocx --version`."); + } else if ( + !softPass && + isNewBugForm && + (version === null || isEmpty(version) || isRawPlaceholder(version)) + ) { + // New form requires Version (including when the heading was removed). + // Legacy N/A / No response soft-pass stays only for bodies without + // Client or integration. + reasons.push("Version is missing."); + guidance.push("Add your OpenCodex version so we can reproduce the environment."); + } + + if (!softPass && isNewBugForm && os !== null && isUnusableVersion(os)) { + reasons.push("Operating system is missing or unknown."); + guidance.push("Add your OS name and version (for example Windows 11 24H2)."); + } else if ( + !softPass && + isNewBugForm && + (os === null || isEmpty(os) || isRawPlaceholder(os)) + ) { + reasons.push("Operating system is missing."); + guidance.push("Add your OS name and version (for example Windows 11 24H2)."); + } + + // Required environment fields removed after submission on bodies that are + // not the new form (no Client or integration). Legacy reports never had + // Version or OS fields, so null means absent, not removed. Skip when the + // raw value is a "No response" placeholder — the old form had both fields + // as optional. Only close when the field was actively cleared. + if ( + !softPass && + !isNewBugForm && + version !== null && + os !== null && + isEmpty(version) && + isEmpty(os) && + !isRawPlaceholder(version) && + !isRawPlaceholder(os) + ) { reasons.push("Version and Operating system are both missing."); guidance.push("Add your OpenCodex version and OS so we can reproduce the environment."); } @@ -859,6 +947,7 @@ module.exports = { isPlaceholderOnlyValue, isPlaceholder, isRawPlaceholder, + isUnusableVersion, countWords, hasConcreteDetail, labelForKind, diff --git a/.github/scripts/issue-quality.test.cjs b/.github/scripts/issue-quality.test.cjs index c6a58996a..c629a3d78 100644 --- a/.github/scripts/issue-quality.test.cjs +++ b/.github/scripts/issue-quality.test.cjs @@ -16,6 +16,7 @@ const { isPlaceholderOnlyValue, isPlaceholder, isRawPlaceholder, + isUnusableVersion, countWords, hasConcreteDetail, rejectsWorkflowDispatchPullRequest, @@ -362,8 +363,8 @@ describe("validateIssue - feature", () => { assert.equal(result.kind, "feature"); assert.equal(result.valid, false, `Expected terse goal "${goal}" to be invalid`); assert.ok( - result.reasons.some((r) => r.includes("too vague")), - `Expected too vague reason for "${goal}", got: ${result.reasons.join(", ")}`, + result.reasons.some((r) => r.includes("too vague") || /missing or empty/i.test(r)), + `Expected too vague or empty reason for "${goal}", got: ${result.reasons.join(", ")}`, ); } }); @@ -536,6 +537,70 @@ describe("validateIssue - bug", () => { assert.equal(result.valid, false); }); + it("rejects a bug with Summary filled but Reproduction empty", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "The 'gpt-5.6-sol' model is not supported when using Codex with a ChatGPT account.", + "### Reproduction", + "No response", + "### Version", + "2.7.42", + "### Operating system", + "macOS", + ].join("\n"); + const result = validateIssue({ title: "Open Codex Error", body, labels: ["bug"] }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Reproduction is empty/i.test(r))); + assert.ok(!result.reasons.some((r) => /Summary is empty/i.test(r))); + }); + + it("rejects a bug whose Reproduction is only an ellipsis (#598)", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "{\"detail\":\"The 'gpt-5.6-sol' model is not supported when using Codex with a ChatGPT account.\"}", + "### Reproduction", + "...", + "### Version", + "2.7.42", + "### Operating system", + "mac os", + ].join("\n"); + const result = validateIssue({ title: "Open Codex Error", body, labels: ["bug"] }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Reproduction is empty/i.test(r))); + }); + + it("rejects a bug with Reproduction filled but Summary empty", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "", + "### Reproduction", + "1. Run ocx start\n2. Send a request", + "### Version", + "2.7.42", + "### Operating system", + "macOS", + ].join("\n"); + const result = validateIssue({ title: "Crash", body, labels: ["bug"] }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Summary is empty/i.test(r))); + }); + it("accepts a terse real crash report", () => { const body = [ "### Client or integration", @@ -629,6 +694,202 @@ describe("validateIssue - bug", () => { assert.equal(result.valid, false); assert.ok(result.reasons.some((r) => r.includes("Version"))); }); + + it("rejects unknown / don't-know Version values (#624)", () => { + const versions = [ + "Unknown", + "Uknown", + "unkown", + "Don't know", + "dont know", + "idk", + "모름", + "잘 모름", + "?", + "???", + ]; + for (const version of versions) { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "The OpenCodex proxy keeps dropping the Codex CLI connection mid-request.", + "### Reproduction", + "1. ocx start --port 10100", + "2. Send any Codex CLI request through the proxy", + "3. Observe the connection drop", + "### Version", + version, + "### Operating system", + "Windows 11", + ].join("\n"); + const result = validateIssue({ + title: "Unexpected interruption continues to occur", + body, + labels: ["bug"], + }); + assert.equal(result.kind, "bug"); + assert.equal( + result.valid, + false, + `Expected unusable Version "${version}" to be invalid, got: ${result.reasons.join("; ")}`, + ); + assert.ok( + result.reasons.some((r) => /Version/i.test(r) && /unknown|missing/i.test(r)), + `Expected Version unknown/missing reason for "${version}", got: ${result.reasons.join("; ")}`, + ); + } + }); + + it("rejects issue #624-style low-effort new-form bug", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "CLI로 확인해봤는데 오픈코덱스 프록시가 중간에 자꾸 연결이 끊어져서 그런거라고 합니다.", + "", + "수정 바랍니다.", + "### Reproduction", + "예기치않게중단됨", + "### Version", + "모름", + "### Operating system", + "윈11", + "### Provider and model", + "_No response_", + "### Logs or error output", + "```shell", + "", + "```", + ].join("\n"); + const result = validateIssue({ + title: "Unexpected interruption continues to occur", + body, + labels: ["bug"], + }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Version/i.test(r))); + assert.ok(result.reasons.some((r) => /Reproduction/i.test(r) && /vague|empty/i.test(r))); + }); + + it("rejects a new-form bug with a usable Version but placeholder OS", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "Proxy returns 502 when streaming is enabled on Windows.", + "### Reproduction", + "1. ocx start", + "2. Send a streaming /v1/responses request", + "### Version", + "2.7.42", + "### Operating system", + "No response", + ].join("\n"); + const result = validateIssue({ title: "Streaming 502", body, labels: ["bug"] }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Operating system/i.test(r))); + }); + + it("rejects a new-form bug when the Version heading was removed", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "Proxy returns 502 when streaming is enabled on Windows.", + "### Reproduction", + "1. ocx start", + "2. Send a streaming /v1/responses request", + "### Operating system", + "Windows 11", + ].join("\n"); + const result = validateIssue({ title: "Streaming 502", body, labels: ["bug"] }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Version/i.test(r) && /missing/i.test(r))); + }); + + it("rejects a new-form bug when the Operating system heading was removed", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "Proxy returns 502 when streaming is enabled on Windows.", + "### Reproduction", + "1. ocx start", + "2. Send a streaming /v1/responses request", + "### Version", + "2.7.42", + ].join("\n"); + const result = validateIssue({ title: "Streaming 502", body, labels: ["bug"] }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Operating system/i.test(r) && /missing/i.test(r))); + }); + + it("rejects a new-form bug whose Reproduction is only a vague phrase", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "The OpenCodex proxy keeps dropping the Codex CLI connection mid-request.", + "### Reproduction", + "Unexpected interruption", + "### Version", + "2.7.42", + "### Operating system", + "Windows 11", + ].join("\n"); + const result = validateIssue({ + title: "Unexpected interruption continues to occur", + body, + labels: ["bug"], + }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Reproduction/i.test(r) && /vague/i.test(r))); + }); + + it("rejects unknown Operating system stand-ins on the new bug form", () => { + const body = [ + "### Client or integration", + "Codex CLI", + "### Area", + "CLI", + "### Summary", + "The OpenCodex proxy keeps dropping the Codex CLI connection mid-request.", + "### Reproduction", + "1. ocx start --port 10100", + "2. Send any Codex CLI request through the proxy", + "3. Observe the connection drop", + "### Version", + "2.7.42", + "### Operating system", + "Unknown", + ].join("\n"); + const result = validateIssue({ + title: "Unexpected interruption continues to occur", + body, + labels: ["bug"], + }); + assert.equal(result.kind, "bug"); + assert.equal(result.valid, false); + assert.ok(result.reasons.some((r) => /Operating system/i.test(r))); + }); }); // --------------------------------------------------------------------------- @@ -775,6 +1036,16 @@ describe("normalisation", () => { assert.equal(clean("Not available!"), ""); }); + it("detects unusable Version stand-ins without treating them as generic placeholders", () => { + for (const value of ["Unknown", "Uknown", "모름", "idk", "don't know"]) { + assert.equal(isUnusableVersion(value), true, value); + assert.equal(isPlaceholderOnlyValue(value), false, value); + } + for (const value of ["2.7.42", "N/A", "No response", "main@abc1234"]) { + assert.equal(isUnusableVersion(value), false, value); + } + }); + it("does not treat sentences containing placeholder phrases as empty", () => { assert.equal(clean("This is N/A for voice mode today."), "This is N/A for voice mode today."); assert.equal(clean("Not applicable to Claude Code."), "Not applicable to Claude Code."); diff --git a/.github/scripts/pr-quality.cjs b/.github/scripts/pr-quality.cjs new file mode 100644 index 000000000..ff286acdc --- /dev/null +++ b/.github/scripts/pr-quality.cjs @@ -0,0 +1,172 @@ +"use strict"; + +const path = require("node:path"); +const { + clean, + isPlaceholderOnlyValue, + hasSubstantialStructuredContent, +} = require(path.join(__dirname, "issue-quality.cjs")); + +const ANCESTRY_BEHIND_THRESHOLD = 20; +/** Cap on ahead_by vs main so stale `dev` forks (many commits ahead of main) are not flagged. */ +const ANCESTRY_AHEAD_MAIN_MAX = 5; +const MIN_SECTION_LEN = 40; +const MIN_RICH_SECTIONS = 2; +const UNSTRUCTURED_MIN_LEN = 120; +const UNSTRUCTURED_MIN_BLOCKS = 2; + +/** + * Exact instruction / checklist lines from `.github/PULL_REQUEST_TEMPLATE.md`. + * Untouched templates must not count as substance. + */ +const PR_TEMPLATE_BOILERPLATE_LINES = new Set([ + "explain the user-visible or maintainer-facing change.", + "list the commands or checks you ran.", + "scope stays focused and avoids unrelated cleanup.", + "docs or release notes were updated when needed.", + "security-sensitive changes were reviewed for secrets, auth, and unsafe defaults.", +]); + +function isWrongAncestry({ + behindMain, + behindBase, + aheadMain = 0, + threshold = ANCESTRY_BEHIND_THRESHOLD, + aheadMainMax = ANCESTRY_AHEAD_MAIN_MAX, +}) { + return ( + behindMain === 0 && + behindBase >= threshold && + aheadMain <= aheadMainMax + ); +} + +function authorHasPushPermission(permission) { + return permission === "admin" || permission === "maintain" || permission === "write"; +} + +/** + * True when the body uses literal backslash-n as the dominant line break + * (agent bug seen on #644) rather than real newlines. + */ +function hasEscapedNewlines(text) { + const escaped = (text.match(/\\n/g) || []).length; + if (escaped < 2) return false; + const real = (text.match(/\n/g) || []).length; + return escaped > real; +} + +function countContentBlocks(text) { + const blocks = text + .split(/\n\s*\n/) + .map((b) => b.trim()) + .filter(Boolean); + if (blocks.length >= 2) return blocks.length; + const bullets = text + .split("\n") + .map((l) => l.trim()) + .filter((l) => /^[-*+]\s+\S/.test(l)); + return Math.max(blocks.length, bullets.length); +} + +function normalizeTemplateLine(line) { + return line + .replace(/^\s*[-*+]\s+/, "") + .replace(/^\s*\[[ xX]\]\s+/, "") + .replace(/^\s*#{1,6}\s+/, "") + .trim() + .toLowerCase(); +} + +/** Drop stock PR template headings, instructions, and checklist lines. */ +function stripPrTemplateBoilerplate(text) { + return text + .split("\n") + .filter((line) => { + const normalized = normalizeTemplateLine(line); + if (!normalized) return true; + if (PR_TEMPLATE_BOILERPLATE_LINES.has(normalized)) return false; + if (/^(summary|verification|checklist)$/.test(normalized)) return false; + return true; + }) + .join("\n"); +} + +function assessPrDescription(body) { + if (typeof body !== "string" || !body.trim()) { + return { ok: false, reason: "empty" }; + } + if (hasEscapedNewlines(body)) { + return { ok: false, reason: "escaped_newlines" }; + } + const withoutTemplate = stripPrTemplateBoilerplate(body); + const cleaned = clean(withoutTemplate); + if (!cleaned) { + const strippedComments = withoutTemplate.replace(//g, "").trim(); + if (!strippedComments) return { ok: false, reason: "empty" }; + if (isPlaceholderOnlyValue(strippedComments)) { + return { ok: false, reason: "placeholder" }; + } + return { ok: false, reason: "empty" }; + } + if (isPlaceholderOnlyValue(cleaned)) { + return { ok: false, reason: "placeholder" }; + } + if (hasSubstantialStructuredContent(cleaned, MIN_SECTION_LEN, MIN_RICH_SECTIONS)) { + return { ok: true }; + } + if ( + cleaned.length >= UNSTRUCTURED_MIN_LEN && + countContentBlocks(cleaned) >= UNSTRUCTURED_MIN_BLOCKS + ) { + return { ok: true }; + } + return { ok: false, reason: "thin" }; +} + +function collectPrQualityFailures({ + baseRef, + allowedBases, + body, + behindMain, + behindBase, + aheadMain = 0, + authorPermission, + permissionLookupFailed = false, + ancestryLookupFailed = false, +}) { + const failures = []; + const wrongBase = !allowedBases.includes(baseRef); + if (wrongBase) { + failures.push({ code: "wrong_base" }); + } else { + // Permission lookup fails closed (still evaluate ancestry). Compare API + // failures skip ancestry — zeros would falsely pass the #644 heuristic. + const skipAncestry = + ancestryLookupFailed || + (!permissionLookupFailed && authorHasPushPermission(authorPermission)); + if ( + !skipAncestry && + isWrongAncestry({ behindMain, behindBase, aheadMain }) + ) { + failures.push({ code: "wrong_ancestry" }); + } + } + + const desc = assessPrDescription(body); + if (!desc.ok) { + failures.push({ code: "bad_description", reason: desc.reason }); + } + return failures; +} + +module.exports = { + ANCESTRY_BEHIND_THRESHOLD, + ANCESTRY_AHEAD_MAIN_MAX, + isWrongAncestry, + authorHasPushPermission, + assessPrDescription, + collectPrQualityFailures, + hasEscapedNewlines, + stripPrTemplateBoilerplate, +}; diff --git a/.github/scripts/pr-quality.test.cjs b/.github/scripts/pr-quality.test.cjs new file mode 100644 index 000000000..487ef38b2 --- /dev/null +++ b/.github/scripts/pr-quality.test.cjs @@ -0,0 +1,245 @@ +"use strict"; + +const { describe, it } = require("node:test"); +const assert = require("node:assert/strict"); +const { + ANCESTRY_BEHIND_THRESHOLD, + isWrongAncestry, + authorHasPushPermission, + assessPrDescription, + collectPrQualityFailures, +} = require("./pr-quality.cjs"); + +describe("isWrongAncestry", () => { + it("flags #644-shaped compares (0 behind main, far behind base, few ahead of main)", () => { + assert.equal( + isWrongAncestry({ behindMain: 0, behindBase: 44, aheadMain: 1 }), + true, + ); + }); + + it("uses threshold 20 by default", () => { + assert.equal(ANCESTRY_BEHIND_THRESHOLD, 20); + assert.equal(isWrongAncestry({ behindMain: 0, behindBase: 20, aheadMain: 1 }), true); + assert.equal(isWrongAncestry({ behindMain: 0, behindBase: 19, aheadMain: 1 }), false); + }); + + it("passes when head is behind main (not sitting on main tip)", () => { + assert.equal(isWrongAncestry({ behindMain: 1, behindBase: 44, aheadMain: 1 }), false); + }); + + it("passes stale dev-based branches that are many commits ahead of main", () => { + assert.equal( + isWrongAncestry({ behindMain: 0, behindBase: 44, aheadMain: 50 }), + false, + ); + }); +}); + +describe("authorHasPushPermission", () => { + it("accepts write/maintain/admin only", () => { + assert.equal(authorHasPushPermission("admin"), true); + assert.equal(authorHasPushPermission("maintain"), true); + assert.equal(authorHasPushPermission("write"), true); + assert.equal(authorHasPushPermission("triage"), false); + assert.equal(authorHasPushPermission("read"), false); + assert.equal(authorHasPushPermission(null), false); + }); +}); + +describe("assessPrDescription", () => { + it("rejects empty and comment-only bodies", () => { + assert.equal(assessPrDescription("").ok, false); + assert.equal(assessPrDescription(" ").ok, false); + assert.equal( + assessPrDescription("\n\n").reason, + "empty", + ); + }); + + it("rejects placeholder-only bodies", () => { + assert.equal(assessPrDescription("N/A").reason, "placeholder"); + assert.equal(assessPrDescription("TODO").reason, "placeholder"); + }); + + it("rejects literal escaped newlines like #644", () => { + const body = + "## What changed\\n- make the Windows tray launcher resolve Codex home\\n\\n## Validation\\n- git diff --check"; + assert.equal(assessPrDescription(body).reason, "escaped_newlines"); + }); + + it("rejects thin real-newline bodies", () => { + assert.equal(assessPrDescription("fix stuff").reason, "thin"); + }); + + it("rejects an untouched GitHub PR template as empty/thin", () => { + const body = [ + "## Summary", + "", + "- Explain the user-visible or maintainer-facing change.", + "", + "## Verification", + "", + "- List the commands or checks you ran.", + "", + "## Checklist", + "", + "- [ ] Scope stays focused and avoids unrelated cleanup.", + "- [ ] Docs or release notes were updated when needed.", + "- [ ] Security-sensitive changes were reviewed for secrets, auth, and unsafe defaults.", + ].join("\n"); + const result = assessPrDescription(body); + assert.equal(result.ok, false); + assert.ok(result.reason === "empty" || result.reason === "thin"); + }); + + it("accepts two rich markdown sections", () => { + const body = [ + "## Summary", + "This change updates the Windows tray launcher so it resolves CODEX_HOME through the shared helper instead of a hardcoded path.", + "", + "## Test plan", + "- Launch the tray app after setting CODEX_HOME", + "- Confirm the listener and launcher use the same workspace root", + ].join("\n"); + assert.equal(assessPrDescription(body).ok, true); + }); + + it("accepts unstructured bodies that are long enough with multiple blocks", () => { + const p1 = + "Updates the Windows tray launcher to resolve the active Codex home through the shared helper so listener and launcher stay aligned."; + const p2 = + "Validated with git diff --check on the changed tray module; typecheck was not available in that session so CI must cover it."; + assert.equal(assessPrDescription(`${p1}\n\n${p2}`).ok, true); + }); +}); + +describe("collectPrQualityFailures", () => { + const allowed = ["dev"]; + + it("reports wrong_base without requiring ancestry inputs", () => { + const failures = collectPrQualityFailures({ + baseRef: "main", + allowedBases: allowed, + body: "## Summary\n" + "x".repeat(50) + "\n\n## Test plan\n" + "y".repeat(50), + behindMain: 0, + behindBase: 0, + authorPermission: "read", + }); + assert.ok(failures.some((f) => f.code === "wrong_base")); + assert.ok(!failures.some((f) => f.code === "wrong_ancestry")); + }); + + it("reports wrong_base and bad_description together for main + empty body", () => { + const failures = collectPrQualityFailures({ + baseRef: "main", + allowedBases: allowed, + body: "", + behindMain: 0, + behindBase: 0, + authorPermission: "read", + }); + assert.ok(failures.some((f) => f.code === "wrong_base")); + assert.ok(failures.some((f) => f.code === "bad_description")); + assert.ok(!failures.some((f) => f.code === "wrong_ancestry")); + }); + + it("reports wrong_ancestry for contributor on #644-shaped compare", () => { + const failures = collectPrQualityFailures({ + baseRef: "dev", + allowedBases: allowed, + body: [ + "## Summary", + "This change updates the Windows tray launcher so it resolves CODEX_HOME through the shared helper instead of a hardcoded path.", + "", + "## Test plan", + "- Launch the tray app after setting CODEX_HOME", + "- Confirm the listener and launcher use the same workspace root", + ].join("\n"), + behindMain: 0, + behindBase: 44, + aheadMain: 1, + authorPermission: "read", + }); + assert.deepEqual( + failures.map((f) => f.code), + ["wrong_ancestry"], + ); + }); + + it("skips ancestry for push permission but still flags bad description", () => { + const failures = collectPrQualityFailures({ + baseRef: "dev", + allowedBases: allowed, + body: "", + behindMain: 0, + behindBase: 44, + aheadMain: 1, + authorPermission: "write", + }); + assert.ok(!failures.some((f) => f.code === "wrong_ancestry")); + assert.ok(failures.some((f) => f.code === "bad_description")); + }); + + it("applies ancestry when permission lookup failed (fail closed)", () => { + const failures = collectPrQualityFailures({ + baseRef: "dev", + allowedBases: allowed, + body: [ + "## Summary", + "This change updates the Windows tray launcher so it resolves CODEX_HOME through the shared helper instead of a hardcoded path.", + "", + "## Test plan", + "- Launch the tray app after setting CODEX_HOME", + "- Confirm the listener and launcher use the same workspace root", + ].join("\n"), + behindMain: 0, + behindBase: 44, + aheadMain: 1, + authorPermission: null, + permissionLookupFailed: true, + }); + assert.ok(failures.some((f) => f.code === "wrong_ancestry")); + }); + + it("does not flag stale dev-based branches that are far ahead of main", () => { + const failures = collectPrQualityFailures({ + baseRef: "dev", + allowedBases: allowed, + body: [ + "## Summary", + "This change updates the Windows tray launcher so it resolves CODEX_HOME through the shared helper instead of a hardcoded path.", + "", + "## Test plan", + "- Launch the tray app after setting CODEX_HOME", + "- Confirm the listener and launcher use the same workspace root", + ].join("\n"), + behindMain: 0, + behindBase: 44, + aheadMain: 50, + authorPermission: "read", + }); + assert.ok(!failures.some((f) => f.code === "wrong_ancestry")); + }); + + it("skips ancestry when compare lookup failed (cannot evaluate)", () => { + const failures = collectPrQualityFailures({ + baseRef: "dev", + allowedBases: allowed, + body: [ + "## Summary", + "This change updates the Windows tray launcher so it resolves CODEX_HOME through the shared helper instead of a hardcoded path.", + "", + "## Test plan", + "- Launch the tray app after setting CODEX_HOME", + "- Confirm the listener and launcher use the same workspace root", + ].join("\n"), + behindMain: 0, + behindBase: 0, + aheadMain: 0, + authorPermission: "read", + ancestryLookupFailed: true, + }); + assert.ok(!failures.some((f) => f.code === "wrong_ancestry")); + }); +}); diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0208df13f..39930f894 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,7 +2,7 @@ name: Cross-platform CI on: pull_request: - branches: [main, dev, dev2-go] + branches: [main, dev] paths: - "src/**" - "bin/**" @@ -48,9 +48,17 @@ jobs: test: name: ${{ matrix.os }} runs-on: ${{ matrix.os }} - # The Windows full suite now completes near the old 8-minute ceiling; leave - # enough room for the privacy/build smokes instead of cancelling a green run. - timeout-minutes: 12 + # Windows dominates this matrix: on run 30459554635 the same suite took + # ubuntu 4.6min / macos 5.6min / windows 11.8min. Against the previous + # 12-minute ceiling that left ~12s of headroom, so runner variance decided + # the result rather than the code under review — #711's rerun finished at + # 11.8min and passed while #653's was killed at 12.0min (issue #717). + # A cancelled job renders as `fail` in `gh pr checks`, so that flakiness + # reads as a broken PR. 20 minutes keeps a green Windows run green with + # real margin; it is not a licence for the suite to grow into it. If + # Windows approaches this ceiling too, fix the 2.5x platform gap instead + # of raising the number again. + timeout-minutes: 20 strategy: fail-fast: false matrix: diff --git a/.github/workflows/enforce-pr-target.yml b/.github/workflows/enforce-pr-target.yml index da76509d9..3c9b61459 100644 --- a/.github/workflows/enforce-pr-target.yml +++ b/.github/workflows/enforce-pr-target.yml @@ -7,8 +7,15 @@ on: - reopened - edited - ready_for_review + - synchronize +# pull-requests:write covers title/comment updates. +# contents:write is required for convertPullRequestToDraft / +# markPullRequestReadyForReview GraphQL mutations with GITHUB_TOKEN +# (otherwise: "Resource not accessible by integration"). This workflow +# never checks out PR head code. permissions: + contents: write pull-requests: write concurrency: @@ -19,30 +26,39 @@ jobs: runs-on: ubuntu-latest steps: - - name: Require dev as PR target + - name: Checkout trusted PR-quality scripts + uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + ref: ${{ github.event.repository.default_branch }} + persist-credentials: false + sparse-checkout: .github/scripts + + - name: Enforce PR target, ancestry, and description uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9 with: script: | - const ALLOWED_BASES = ["dev", "dev2-go"]; + const path = require("path"); + const { collectPrQualityFailures } = require( + path.join(process.cwd(), ".github", "scripts", "pr-quality.cjs"), + ); + + const ALLOWED_BASES = ["dev"]; const DEFAULT_BASE = "dev"; const TITLE_PREFIX = "[WRONG BRANCH] "; - const COMMENT_MARKER = ""; + const COMMENT_MARKER = ""; + const LEGACY_COMMENT_MARKER = ""; const STATE_PATTERN = - //; + //; const { owner, repo } = context.repo; const pull_number = context.payload.pull_request.number; - // Fetch the latest PR state instead of relying on a possibly - // outdated event payload. const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number }); - // Find the workflow's existing comment. Its hidden state records - // whether the workflow changed the title or draft status. const comments = await github.paginate( github.rest.issues.listComments, { @@ -56,8 +72,10 @@ jobs: const botComment = comments.find( comment => comment.user?.login === "github-actions[bot]" && - comment.body?.includes(COMMENT_MARKER) + (comment.body?.includes(COMMENT_MARKER) || + comment.body?.includes(LEGACY_COMMENT_MARKER)) ); + let botCommentId = botComment?.id ?? null; function parseState(body) { const match = body?.match(STATE_PATTERN); @@ -79,7 +97,7 @@ jobs: function stateMarker(state) { return ( - "" ); @@ -90,23 +108,24 @@ jobs: } async function upsertComment(body) { - if (botComment) { + if (botCommentId) { await github.rest.issues.updateComment({ owner, repo, - comment_id: botComment.id, + comment_id: botCommentId, body }); return; } - await github.rest.issues.createComment({ + const created = await github.rest.issues.createComment({ owner, repo, issue_number: pull_number, body }); + botCommentId = created.data.id; } async function convertToDraft() { @@ -153,82 +172,303 @@ jobs: ); } + function descriptionFailureLines(reason) { + switch (reason) { + case "empty": + return [ + "The pull request body is empty after stripping HTML comments.", + "", + "Include a real description: a **Summary** of what changed and why, plus a **Test plan** (or equivalent substance)." + ]; + case "placeholder": + return [ + "The pull request body contains only placeholder text (for example `N/A`, `TODO`, or `No response`).", + "", + "Replace placeholders with a **Summary** and **Test plan**, or another description with at least two substantive sections or paragraphs." + ]; + case "escaped_newlines": + return [ + "The pull request body uses literal `\\n` escape sequences instead of real line breaks.", + "", + "Fix the formatting so the body uses normal markdown line breaks, then add a **Summary** and **Test plan**." + ]; + case "thin": + default: + return [ + "The pull request description is too thin to review.", + "", + "Add a **Summary** and **Test plan** (two sections with at least 40 characters each), or an unstructured body of at least 120 characters with two paragraphs or bullet groups." + ]; + } + } + + function buildFailureSections(failures) { + const sections = []; + + if (failures.some(failure => failure.code === "wrong_base")) { + sections.push( + "⚠️ **Wrong target branch**", + "", + `This pull request currently targets ${inlineCode(pr.base.ref)}, but pull requests must target one of ${ALLOWED_BASES.map(inlineCode).join(" or ")}.`, + "", + `@${pr.user.login} Please retarget this PR to ${inlineCode(DEFAULT_BASE)}. All contributions go to ${inlineCode(DEFAULT_BASE)}; \`main\` receives only release promotions. See our [Contributing guide](https://lidge-jun.github.io/opencodex/contributing/) for details. Thanks! 🙏` + ); + } + + if (failures.some(failure => failure.code === "wrong_ancestry")) { + sections.push( + "⚠️ **Wrong branch ancestry**", + "", + `This pull request targets ${inlineCode(pr.base.ref)}, but its head appears to sit on the current ${inlineCode("main")} tip while being far behind ${inlineCode(pr.base.ref)}.`, + "", + `@${pr.user.login} Rebase onto the current ${inlineCode(pr.base.ref)} branch instead of opening from ${inlineCode("main")}. That keeps already-released commits out of the integration branch.` + ); + } + + const badDescription = failures.find( + failure => failure.code === "bad_description" + ); + if (badDescription) { + sections.push( + "⚠️ **Pull request description**", + "", + ...descriptionFailureLines(badDescription.reason) + ); + } + + return sections; + } + + function failureSummary(failures) { + return failures + .map(failure => { + if (failure.code === "wrong_base") { + return `wrong base (${pr.base.ref})`; + } + if (failure.code === "wrong_ancestry") { + return "wrong ancestry"; + } + if (failure.code === "bad_description") { + return `bad description (${failure.reason})`; + } + return failure.code; + }) + .join("; "); + } + const storedState = parseState(botComment?.body); - const wrongBase = !ALLOWED_BASES.includes(pr.base.ref); - - // Wrong branch: - // - add the title prefix - // - convert ready PRs to draft - // - post or update the explanation - // - remember which changes were made by this workflow - if (wrongBase) { + + let authorPermission = null; + let permissionLookupFailed = false; + try { + const { data: permissionData } = + await github.rest.repos.getCollaboratorPermissionLevel({ + owner, + repo, + username: pr.user.login + }); + authorPermission = permissionData.permission; + } catch (error) { + permissionLookupFailed = true; + core.warning( + `Could not look up collaborator permission: ${error.message}` + ); + } + + let behindMain = 0; + let behindBase = 0; + let aheadMain = 0; + let ancestryLookupFailed = false; + const baseAllowed = ALLOWED_BASES.includes(pr.base.ref); + + if (baseAllowed) { + const headSha = pr.head.sha; + try { + const { data: mainCompare } = + await github.rest.repos.compareCommitsWithBasehead({ + owner, + repo, + basehead: `main...${headSha}` + }); + behindMain = mainCompare.behind_by; + aheadMain = mainCompare.ahead_by; + + const { data: baseCompare } = + await github.rest.repos.compareCommitsWithBasehead({ + owner, + repo, + basehead: `${pr.base.ref}...${headSha}` + }); + behindBase = baseCompare.behind_by; + } catch (error) { + ancestryLookupFailed = true; + core.warning( + `Could not compare commits for ancestry check: ${error.message}` + ); + } + } + + const failures = collectPrQualityFailures({ + baseRef: pr.base.ref, + allowedBases: ALLOWED_BASES, + body: pr.body, + behindMain, + behindBase, + aheadMain, + authorPermission, + permissionLookupFailed, + ancestryLookupFailed + }); + + if (failures.length > 0) { + const hasWrongBase = failures.some( + failure => failure.code === "wrong_base" + ); + const willPrefixTitle = + hasWrongBase && !pr.title.startsWith(TITLE_PREFIX); const state = storedState?.active ? { ...storedState } : { version: 1, active: true, autoDraftedByBot: false, - titlePrefixedByBot: false + titlePrefixedByBot: false, + ancestryFailed: false, + descriptionFailed: false }; - if (!pr.draft) { - state.autoDraftedByBot = true; - } + state.ancestryFailed = failures.some( + failure => failure.code === "wrong_ancestry" + ); + state.descriptionFailed = failures.some( + failure => failure.code === "bad_description" + ); + + const shouldStripTitlePrefix = + !hasWrongBase && + state.titlePrefixedByBot && + pr.title.startsWith(TITLE_PREFIX); - if (!pr.title.startsWith(TITLE_PREFIX)) { + // Claim title ownership before adding a prefix. Do not clear + // ownership until strip succeeds — a failed update must retry. + if (willPrefixTitle) { state.titlePrefixedByBot = true; } - const draftExplanation = state.autoDraftedByBot - ? "This pull request is being kept as a draft automatically. Once the target branch is corrected, it will be marked ready for review again." - : "This pull request was already a draft. Its draft status will be preserved after the target branch is corrected."; + let draftConversionFailed = false; + const failureSections = buildFailureSections(failures); - // Store the restoration state before changing the PR, allowing - // a rerun to recover if an API request fails partway through. await upsertComment( [ COMMENT_MARKER, stateMarker(state), "", - "⚠️ **Wrong target branch**", - "", - `This pull request currently targets ${inlineCode(pr.base.ref)}, but pull requests must target one of ${ALLOWED_BASES.map(inlineCode).join(" or ")}.`, + ...failureSections, "", - `Its title has been prefixed with ${inlineCode(TITLE_PREFIX.trim())}.`, - "", - `@${pr.user.login} Please retarget this PR to ${inlineCode(DEFAULT_BASE)}. Most contributions go to ${inlineCode(DEFAULT_BASE)} first; use ${inlineCode("dev2-go")} only for scoped Go native-port work. \`main\` receives only release promotions. See our [Contributing guide](https://lidge-jun.github.io/opencodex/contributing/) for details. Thanks! 🙏`, - "", - draftExplanation + "Recording ownership state before applying title/draft changes…" ].join("\n") ); - if (!pr.title.startsWith(TITLE_PREFIX)) { + if (willPrefixTitle) { await github.rest.pulls.update({ owner, repo, pull_number, title: `${TITLE_PREFIX}${pr.title}` }); + } else if (shouldStripTitlePrefix) { + await github.rest.pulls.update({ + owner, + repo, + pull_number, + title: pr.title.slice(TITLE_PREFIX.length) + }); + state.titlePrefixedByBot = false; + await upsertComment( + [ + COMMENT_MARKER, + stateMarker(state), + "", + ...failureSections, + "", + "Stale title prefix removed; continuing…" + ].join("\n") + ); } if (!pr.draft) { - await convertToDraft(); + // Claim draft ownership before the mutation so a successful + // convert followed by a failed comment still restores later. + state.autoDraftedByBot = true; + await upsertComment( + [ + COMMENT_MARKER, + stateMarker(state), + "", + ...failureSections, + "", + "Draft conversion pending…" + ].join("\n") + ); + try { + await convertToDraft(); + await upsertComment( + [ + COMMENT_MARKER, + stateMarker(state), + "", + ...failureSections, + "", + "Draft conversion succeeded; finalising explanation…" + ].join("\n") + ); + } catch (error) { + draftConversionFailed = true; + state.autoDraftedByBot = false; + core.warning( + `Could not convert pull request to draft: ${error.message}` + ); + } + } + + const draftExplanation = draftConversionFailed + ? "Automatic draft conversion failed (token cannot change draft status). Please convert this pull request to a draft manually. The required `enforce-target` check will keep failing until every issue above is resolved." + : state.autoDraftedByBot + ? "This pull request is being kept as a draft automatically. Once every issue above is resolved, it will be marked ready for review again." + : "This pull request was already a draft. Its draft status will be preserved after every issue above is resolved."; + + const finalSections = [...failureSections]; + + if (hasWrongBase && state.titlePrefixedByBot) { + finalSections.push( + "", + `Its title has been prefixed with ${inlineCode(TITLE_PREFIX.trim())}.` + ); } + await upsertComment( + [ + COMMENT_MARKER, + stateMarker(state), + "", + ...finalSections, + "", + draftExplanation + ].join("\n") + ); + + core.setFailed(`PR quality gate failed: ${failureSummary(failures)}`); return; } - // The target is correct and the workflow has nothing to restore. if (!storedState?.active) { core.info( - "Target branch is correct and there is no active bot state." + "All PR quality gates passed and there is no active bot state." ); return; } - // Remove only the exact prefix added by this workflow. Other title - // edits made by the contributor are preserved. if ( storedState.titlePrefixedByBot && pr.title.startsWith(TITLE_PREFIX) @@ -241,38 +481,57 @@ jobs: }); } - // Mark the PR ready only if this workflow originally converted it - // to draft. Intentionally drafted PRs remain drafts. + let readyConversionFailed = false; if ( storedState.autoDraftedByBot && pr.draft ) { - await markReadyForReview(); + try { + await markReadyForReview(); + } catch (error) { + readyConversionFailed = true; + core.warning( + `Could not mark pull request ready for review: ${error.message}` + ); + } } - const completedState = { - version: 1, - active: false, - autoDraftedByBot: false, - titlePrefixedByBot: false - }; + const completedState = readyConversionFailed + ? { + version: 1, + active: true, + autoDraftedByBot: true, + titlePrefixedByBot: false, + ancestryFailed: false, + descriptionFailed: false + } + : { + version: 1, + active: false, + autoDraftedByBot: false, + titlePrefixedByBot: false, + ancestryFailed: false, + descriptionFailed: false + }; const titleResult = storedState.titlePrefixedByBot ? `The ${inlineCode(TITLE_PREFIX.trim())} title prefix has been removed.` : "The title was left unchanged."; - const draftResult = storedState.autoDraftedByBot - ? "The pull request has been marked ready for review again." - : "Its existing draft status has been preserved."; + const draftResult = readyConversionFailed + ? "Automatic ready-for-review conversion failed; please mark the pull request ready manually if it is still a draft." + : storedState.autoDraftedByBot + ? "The pull request has been marked ready for review again." + : "Its existing draft status has been preserved."; await upsertComment( [ COMMENT_MARKER, stateMarker(completedState), "", - "✅ **Target branch corrected**", + "✅ **PR quality gates passed**", "", - `This pull request now targets ${inlineCode(pr.base.ref)}.`, + `This pull request now targets ${inlineCode(pr.base.ref)} with acceptable ancestry and description.`, "", `${titleResult} ${draftResult}` ].join("\n") diff --git a/.github/workflows/issue-quality-tests.yml b/.github/workflows/issue-quality-tests.yml index 76d7e97ad..8c6d0da14 100644 --- a/.github/workflows/issue-quality-tests.yml +++ b/.github/workflows/issue-quality-tests.yml @@ -6,8 +6,11 @@ on: - ".github/ISSUE_TEMPLATE/**" - ".github/scripts/issue-quality.cjs" - ".github/scripts/issue-quality.test.cjs" + - ".github/scripts/pr-quality.cjs" + - ".github/scripts/pr-quality.test.cjs" - ".github/scripts/pr-labeler.cjs" - ".github/scripts/pr-labeler.test.cjs" + - ".github/scripts/enforce-pr-target.test.cjs" - ".github/scripts/issue-translation.cjs" - ".github/scripts/issue-translation.test.cjs" - ".github/scripts/issue-triage.cjs" @@ -15,6 +18,7 @@ on: - ".github/scripts/parse-issue-translation-response.cjs" - ".github/scripts/parse-issue-translation-response.test.cjs" - ".github/workflows/enforce-issue-quality.yml" + - ".github/workflows/enforce-pr-target.yml" - ".github/workflows/pr-labeler.yml" - ".github/workflows/issue-triage.yml" - ".github/workflows/issue-quality-tests.yml" @@ -23,8 +27,11 @@ on: - ".github/ISSUE_TEMPLATE/**" - ".github/scripts/issue-quality.cjs" - ".github/scripts/issue-quality.test.cjs" + - ".github/scripts/pr-quality.cjs" + - ".github/scripts/pr-quality.test.cjs" - ".github/scripts/pr-labeler.cjs" - ".github/scripts/pr-labeler.test.cjs" + - ".github/scripts/enforce-pr-target.test.cjs" - ".github/scripts/issue-translation.cjs" - ".github/scripts/issue-translation.test.cjs" - ".github/scripts/issue-triage.cjs" @@ -32,6 +39,7 @@ on: - ".github/scripts/parse-issue-translation-response.cjs" - ".github/scripts/parse-issue-translation-response.test.cjs" - ".github/workflows/enforce-issue-quality.yml" + - ".github/workflows/enforce-pr-target.yml" - ".github/workflows/pr-labeler.yml" - ".github/workflows/issue-triage.yml" - ".github/workflows/issue-quality-tests.yml" @@ -52,7 +60,9 @@ jobs: - name: Run validator tests run: | node --test .github/scripts/issue-quality.test.cjs + node --test .github/scripts/pr-quality.test.cjs node --test .github/scripts/pr-labeler.test.cjs + node --test .github/scripts/enforce-pr-target.test.cjs node --test .github/scripts/issue-translation.test.cjs node --test .github/scripts/issue-triage.test.cjs node --test .github/scripts/parse-issue-translation-response.test.cjs diff --git a/.github/workflows/react-doctor.yml b/.github/workflows/react-doctor.yml index f81b046e0..b9076bcb8 100644 --- a/.github/workflows/react-doctor.yml +++ b/.github/workflows/react-doctor.yml @@ -1,13 +1,13 @@ # React Doctor — finds security, performance, correctness, accessibility, # bundle-size, and architecture issues in React codebases. # -# Advisory-only and least-privilege: findings appear in the step log and the -# Actions run summary. All write-scoped outputs (sticky PR comments, inline -# review comments, commit statuses) are explicitly disabled so the workflow -# needs no write permissions. Do not re-add write scopes without revisiting -# tests/ci-workflows.test.ts, which pins this contract. +# Gating and least-privilege: findings fail the job (`blocking: warning`). +# Write-scoped outputs (sticky PR comments, inline review comments, commit +# statuses) stay disabled so the workflow needs no write permissions. Do not +# re-add write scopes without revisiting tests/ci-workflows.test.ts, which +# pins this contract. # -# Docs: https://www.react.doctor/ci +# Docs: https://www.react.doctor/docs/ci-and-prs/github-actions-setup # Source: https://github.com/millionco/react-doctor name: React Doctor @@ -24,7 +24,7 @@ permissions: contents: read # Needed so the action can list PR files for --changed-files-from. # Without this, listFiles fails, the changed-files file is never written, - # and the CLI exits 1 on ENOENT even with blocking: none (fork PRs). + # and the CLI exits 1 on ENOENT even for fork PRs. pull-requests: read # Cancels any in-flight scan for the same PR (or branch, on push) the moment a @@ -50,9 +50,9 @@ jobs: directory: gui # Pin the npm engine — the action wrapper would otherwise fetch # react-doctor@latest, silently skewing CI from the local pinned runs. - version: "0.9.1" - # Advisory contract: report to the step log only; never gate, never write. - blocking: none + version: "0.9.2" + # Fail the job on any finding (errors or warnings). + blocking: warning comment: false review-comments: false commit-status: false diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index edeaaba43..091058da3 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -154,20 +154,15 @@ jobs: echo "Cross-platform CI passed for ${GITHUB_SHA}: ${ci_url}" - # Compare within the release channel so preview and stable gates use matching history. - if [[ "$RELEASE_VERSION" == *-preview.* ]]; then - previous_tag="$( - git tag --merged HEAD --list 'v[0-9]*' --sort=-v:refname | - grep -- '-preview\.' | - head -n 1 || true - )" - else - previous_tag="$( - git tag --merged HEAD --list 'v[0-9]*' --sort=-v:refname | - grep -v -- '-preview\.' | - head -n 1 || true - )" - fi + # Notes / service baseline: + # - Preview: newest prior release of either channel (stable or preview). A + # preview→preview-only baseline skips a shipped stable and restates it. + # - Stable: newest prior stable only (matching preview carry adjusts the + # generate-notes range separately below). + previous_tag="$( + git tag --merged HEAD --list 'v[0-9]*' | + bun scripts/release-notes.ts previous-release-tag "$RELEASE_VERSION" + )" if [ -n "$previous_tag" ]; then changed_files="$(git diff --name-only "${previous_tag}..HEAD")" else @@ -299,22 +294,12 @@ jobs: exit 1 fi - # Channel previous tag (stable↔stable / preview↔preview) for the Full Changelog link. - if [[ "$RELEASE_VERSION" == *-preview.* ]]; then - previous_tag="$( - git tag --merged HEAD --list 'v[0-9]*' --sort=-v:refname | - grep -- '-preview\.' | - grep -vx "$release_tag" | - head -n 1 || true - )" - else - previous_tag="$( - git tag --merged HEAD --list 'v[0-9]*' --sort=-v:refname | - grep -v -- '-preview\.' | - grep -vx "$release_tag" | - head -n 1 || true - )" - fi + # Channel previous tag for Full Changelog + default notes baseline. + # Preview baselines any prior release; stable baselines prior stable only. + previous_tag="$( + git tag --merged HEAD --list 'v[0-9]*' | + bun scripts/release-notes.ts previous-release-tag "$RELEASE_VERSION" + )" npm_metadata="Published to npm as \`@bitkyc08/opencodex@${RELEASE_VERSION}\` with dist-tag \`${NPM_DIST_TAG}\`." # Preview builds must be marked prerelease so GitHub "latest" keeps pointing at the diff --git a/.github/workflows/service-lifecycle.yml b/.github/workflows/service-lifecycle.yml index 955afb713..0f9617268 100644 --- a/.github/workflows/service-lifecycle.yml +++ b/.github/workflows/service-lifecycle.yml @@ -2,7 +2,7 @@ name: Service lifecycle on: pull_request: - branches: [main, dev, dev2-go] + branches: [main, dev] paths: - "src/service.ts" # Keep in sync with the release.yml service-gate regex (release.yml "Require diff --git a/.gitignore b/.gitignore index 8ee0b5d9c..37712544b 100644 --- a/.gitignore +++ b/.gitignore @@ -3,13 +3,28 @@ dist/ .env *.log .DS_Store + +# Maintainer-only planning notes. `devlog` is a private submodule +# (lidge-jun/opencodex-internal) pinned by gitlink; its contents are never +# tracked here. Commit inside the submodule, then bump the pointer separately. devlog/ + +# Scratch space. Security working notes — unreleased findings, draft advisories, +# exploit reasoning, pre-disclosure patch plans — belong here or in a +# `mktemp -d` path, and nowhere else. Not devlog/, not a private repo. See the +# "Security working notes" section of AGENTS.md. .tmp/ .opencode/ # Local agent/session artifacts +# These are per-machine agent state (goalplans, ledgers, evidence scratch). +# They are never part of the product and must not be committed, not even with +# `git add -f` — see tests/repo-hygiene.test.ts, which fails if any path here +# becomes tracked again. .codexclaw/ +**/.codexclaw/ .omo/ +**/.omo/ # Test-generated artifacts tests/.tmp-*/ diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 000000000..9618fedc7 --- /dev/null +++ b/.gitmodules @@ -0,0 +1,6 @@ +[submodule "devlog"] + path = devlog + url = https://github.com/lidge-jun/opencodex-internal.git + ignore = dirty + update = none + shallow = true diff --git a/AGENTS.md b/AGENTS.md index b3e9989d2..fce9ba00c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -16,11 +16,69 @@ Bun-native TypeScript with no separate server compile step. `tests/helpers/`, broader scenarios in `tests/e2e-style/`. - `gui/` — React + Vite dashboard; packaged output is served from `gui/dist`. - `docs-site/` — public docs (Astro + Starlight), deployed to GitHub Pages. +- `go/` — retired Go native-runtime experiment; kept only where the TypeScript + runtime still references it. New work does not go here. - `structure/` — maintainer invariants and architecture notes; read before changing shared subsystems. - `scripts/` — release and maintenance tooling; `scripts/release.ts` is the release authority. -- `devlog/` — planning and investigation artifacts (mostly gitignored). +- `devlog/` — maintainer-only planning and investigation notes. This is a + **private submodule** (`lidge-jun/opencodex-internal`), not a directory of + this repository. See "The `devlog` submodule" below. + +Read the nearest nested `AGENTS.md` before changing files in a scoped +directory (`src/`, `gui/`, `docs-site/`, `scripts/`, `.github/`). + +## The `devlog` submodule + +Planning notes, triage matrices, and investigation artifacts live in the private +`lidge-jun/opencodex-internal` repository, wired in as the `devlog` submodule. +They quote live infrastructure state, provider behaviour, unfixed defects, and +internal triage reasoning, so a public clone should carry the runtime and its +docs and nothing else. + +The pointer is deliberately **loose**, so a missing or stale `devlog` can never +fail a check: + +- `.gitmodules` declares `ignore = dirty`, `update = none`, and `shallow = true`. + A dirty or moved submodule working tree does not show up in `git status` on + the parent, and `git submodule update` will not touch it unless asked + explicitly. +- No workflow checks it out. `actions/checkout` runs without `submodules:`, so + CI clones the public tree only and the private URL is never resolved. +- Nothing in the build, test, typecheck, or privacy-scan path reads from + `devlog/`. Contributors without access see an empty directory and every gate + still passes. +- `devlog/` stays listed in `.gitignore` for the working tree; the submodule + gitlink is tracked, its contents are not. + +Two rules keep it that way. Never commit anything under `devlog/` to *this* +repository — commit inside the submodule, then update the pointer here as a +separate commit. And never nest a git repository inside the submodule: a +`160000` gitlink in a tree that CI does not initialize breaks +`actions/checkout` for every contributor, which is exactly what happened before +this split. + +## Security working notes + +**Security work is done in scratch space, never in a tracked directory.** That +includes unreleased findings, severity assessments, draft advisories, exploit +or bypass reasoning, reproduction steps for an unfixed defect, and +pre-disclosure patch plans. + +Use `.tmp/` in the working tree (already gitignored) or a `mktemp -d` path. +`devlog/` is **not** an acceptable location, and neither is a private +repository: both get cloned across machines and CI, both outlive the embargo, +and neither history is practical to purge afterwards. + +Only the published outcome reaches a repository — the fix itself, its +regression test, the release note, the advisory once it is public. Draft the +advisory in scratch space and delete the scratch directory once the advisory is +live. + +This applies to `AGENTS.md`-following agents as much as to humans. If a task +asks you to write up a security finding, put the write-up in scratch space and +say where it is; do not add it to `devlog/`, `structure/`, or `docs-site/`. ## Commands @@ -38,20 +96,30 @@ non-trivial change. CI runs these on Linux, Windows, and macOS. ## Branch policy -- `dev` — integration branch and the default target. A pull request goes here - unless it belongs to a scoped line below. -- `dev2-go` — parallel integration line for the Go native port: `go/`, - `bin/native-runtime.mjs`, `src/lib/runtime-entry.ts`, and the Go - release-asset tooling. Open for pull requests: the target-branch check - accepts `dev` and `dev2-go` as integration targets. Keep it to scoped Go - native-port work — the check cannot tell an intentional target from a - mistaken one, so that boundary is a review decision. It converges back - through maintainer-controlled merges, and promotion to `main` still happens - only from `dev`. +- `dev` — the single integration branch and the target for every pull request. - `main` — release branch. It only moves by maintainer-controlled promotion from `dev` (releases, docs deploys). Do not open feature PRs against `main`. - `preview` — prerelease train (`x.y.z-preview.*` versions). +### The retired `dev2-go` line + +The project previously ran a parallel `dev2-go` integration line that was +rebuilding the runtime as a Go native port, and every merge into `dev` had to be +carried onto it. That dual-track policy is over: maintaining two integration +lines cost more than the port returned, and dogfooding the Go runtime kept +surfacing new defects. + +`dev2-go` has been deleted, along with the `codex/260728-go-port-*` and +`tmp/dev2-go-source-export` side branches. The full history lives in +[lidge-jun/opencodex-go-archive](https://github.com/lidge-jun/opencodex-go-archive) +and the final tip is tagged `archive/dev2-go` in this repository. There is no +carry or port obligation attached to a `dev` merge any more, and the +`needs-go-port` label is gone. + +Bun-native TypeScript is the only runtime line. If native code returns, the +expectation is an incremental module (for example Rust via N-API) landing on +`dev`, not a second full-runtime branch. + The Claude Desktop integration formerly carried on the `claudedesktop` branch is now fully merged into `dev`, and that branch has been retired. Desktop work continues as normal pull requests against `dev`. @@ -61,6 +129,13 @@ integration line to another, or rebasing a stale branch onto the current head, is ordinary maintenance rather than noise — open it as a normal pull request and name the source commits in the description. +The **`enforce-target`** CI check rejects pull requests whose head +ancestry sits on the **`main`** tip while far behind **`dev`**, and rejects +empty, thin, or malformed descriptions; authors with repository push permission +skip the ancestry heuristic only. As with approval requirements in +[`MAINTAINERS.md`](./MAINTAINERS.md), this is enforced by convention until +branch protection is configured. + [`MAINTAINERS.md`](./MAINTAINERS.md) is authoritative for review and merge policy (approvals, CI requirements, security review, promotion). This file summarizes; it never overrides it. @@ -74,12 +149,8 @@ reviewers (Codex, CodeRabbit). language. Be detailed and specific: name the file and line, describe the concrete failure mode, and suggest a fix. Avoid vague or purely stylistic commentary. -- **Branch targeting:** flag any pull request that targets neither `dev` nor - `dev2-go` (releases and maintainer promotions are the only exceptions). - `dev2-go` is accepted by the automation but scoped by review: if a pull - request targets it without touching `go/`, the native runtime entrypoint, or - the Go release-asset tooling, ask the author to retarget to `dev`. The - automation cannot make that judgement, which is why it is yours. +- **Branch targeting:** flag any pull request that does not target `dev` + (releases and maintainer promotions are the only exceptions). - **Security boundary (highest priority):** changes touching authentication, credential/token handling, OAuth flows, GitHub Actions workflows, release automation (`scripts/release.ts`, `.github/workflows/release.yml`), or diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index cbf673adf..eb3fdab11 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -10,18 +10,17 @@ Thanks for helping with opencodex. ## Branches -- `dev` — default integration target for pull requests. -- `dev2-go` — parallel integration line for the Go native port (`go/`, the - native runtime entrypoint, and the Go release-asset tooling). Open for pull - requests alongside `dev`. Send work here only when it belongs to the Go port; - anything else goes to `dev`. The automated check accepts both targets and - cannot tell them apart, so scope is settled in review. +- `dev` — the only integration target for pull requests. - `main` — releases only; moves by maintainer-controlled promotion from `dev`. - `preview` — prerelease train. -Porting and rebase pull requests are welcome: carrying a fix across integration -lines, or rebasing a stale branch onto the current head, is normal -contribution. Note the source commits in the description. +The `dev2-go` Go native-port line has been retired. Its history is archived at +[lidge-jun/opencodex-go-archive](https://github.com/lidge-jun/opencodex-go-archive), +and everything now goes to `dev`. See [`MAINTAINERS.md`](./MAINTAINERS.md) for +the reasoning. + +Rebase pull requests are welcome: bringing a stale branch onto the current head +is normal contribution. Note the source commits in the description. Agent-facing repository and review rules live in [`AGENTS.md`](./AGENTS.md). diff --git a/MAINTAINERS.md b/MAINTAINERS.md index bd5f3a487..867e7d6aa 100644 --- a/MAINTAINERS.md +++ b/MAINTAINERS.md @@ -14,26 +14,62 @@ review and merge policy. The table describes project responsibilities. Actual repository permissions remain controlled through GitHub repository settings. +`dev` is the only integration line. The former `dev2-go` carry duty is retired; +see [The retired `dev2-go` line](#the-retired-dev2-go-line). + ## Review and merge policy -- Pull requests target `dev` by default. `dev2-go` is a parallel integration - line reserved for Go native-port work; it converges back through - maintainer-controlled merges, and promotion to `main` still happens only from - `dev`. The target-branch check accepts both as integration targets. It cannot - distinguish scoped Go port work from anything else aimed at `dev2-go`, so - that boundary is enforced in review: redirect an out-of-scope pull request to - `dev` rather than treating the automation's silence as approval. +- Pull requests target `dev`. It is the only integration line, and promotion to + `main` happens only from `dev`. The target-branch check accepts `dev` alone. +- The **`enforce-target`** CI check rejects pull requests whose head + ancestry sits on the **`main`** tip while far behind **`dev`**, and rejects + empty, thin, or malformed descriptions; authors with repository push + permission skip the ancestry heuristic only. As with the approval requirement + above, this is enforced by convention until branch protection is configured + (see the note under the change log). - A pull request requires approval from at least one maintainer and successful required CI checks before merge. - Authors do not approve their own pull requests. - Authentication, credential handling, GitHub Actions, release automation, dependency installation, and other security-boundary changes require explicit security review. +- A new or promoted provider preset is a credential-destination change. Before merge it needs the + primary-source evidence listed under [Adding a provider to the + catalog](https://opencodex.me/contributing/#evidence-required-for-a-canonical-preset): documented + OpenAI-compatible endpoints (including authenticated `GET /v1/models` when the entry declares + `liveModels`), terms of service and operating legal entity, resale or routing authorization for + aggregators, a named maintenance owner, and a citable verification date. Contributor affiliation + with the service is disclosed, not disqualifying, and it does not lower the evidence bar. When the + evidence is incomplete, prefer an inert `src/providers/free-directory.ts` reference row over a + canonical registry entry. - Security-sensitive and release-related changes should be reviewed by both maintainers when practical. - Direct pushes are reserved for maintainer-owned integration work, urgent repairs, or incident recovery. The same CI and documentation requirements still apply. - Promotion from `dev` to `main` and npm releases is maintainer-controlled. +## The retired `dev2-go` line + +`dev2-go` was a parallel integration line that rebuilt the runtime as a Go +native port, and policy required every merge into `dev` to be rebased onto it +and ported under `go/`. That policy is withdrawn as of 2026-07-30. + +The dual-track cost outran its return: the carry backlog never cleared (17 +commits and 9 open `needs-go-port` issues at the time of the decision, against +594 commits of divergence), and dogfooding the Go runtime kept producing new +defects. Bun-native TypeScript on `dev` is the single runtime line again. + +- The branch has been deleted from this repository. Its full history is + published at + [lidge-jun/opencodex-go-archive](https://github.com/lidge-jun/opencodex-go-archive), + and its final tip stays reachable here as the `archive/dev2-go` tag. +- A merge into `dev` carries no port obligation. The nine open `needs-go-port` + issues (#661, #663, #666, #670, #674, #678, #680, #685, #703) were closed as + not planned, and the `needs-go-port` label no longer exists on the + repository. +- Future native work is expected to be an incremental module landing on `dev` + (Rust via N-API is the current candidate), not a second integration branch. + Reopening a parallel runtime line is an owner decision. + ## Maintainer changes Adding or removing a maintainer requires: @@ -46,16 +82,23 @@ Adding or removing a maintainer requires: - 2026-07-27 — [@Wibias](https://github.com/Wibias) added as a maintainer. Requirement 1 (agreement from the project owner) is met: the owner requested - the addition. **Requirement 2 (review by another current maintainer) is still - open** and is satisfied when this change is reviewed and merged; until then - this entry records an in-progress change, not a completed procedure. - Requirement 3 is met by this file and `.github/CODEOWNERS`. + the addition. **Requirement 2 (review by another current maintainer) was + never satisfied in the form this document describes.** The three commits that + carried the addition (`a2693c02`, `dc3a4ade`, `02bbd47a`) landed on `dev` as + direct owner pushes with no associated pull request, so no second maintainer + reviewed them. Requirement 3 is met by this file and `.github/CODEOWNERS`. + The addition is in effect regardless: @Wibias holds write access on the + repository and has been merging pull requests since 2026-07-26. This entry + records the gap rather than papering over it — a later maintainer change + should go through a reviewed pull request. Scope covers issue and pull-request triage, `dev` integration, and - provider/CI maintenance. Security-boundary ownership in `.github/CODEOWNERS` - is deliberately unchanged: authentication, credential handling, GitHub - Actions, and release automation keep the two owners already listed for those - paths, so this addition does not widen the review surface for them. + provider/CI maintenance. (This entry originally also described carrying + merged `dev` work onto `dev2-go`; that duty ended when the line was retired + on 2026-07-30.) Security-boundary ownership in `.github/CODEOWNERS` is + deliberately unchanged: authentication, credential handling, GitHub Actions, + and release automation keep the two owners already listed for those paths, so + this addition does not widen the review surface for them. CODEOWNERS requests reviews rather than enforcing them — no branch protection rule is configured on this repository, so code-owner approval is a convention diff --git a/README.md b/README.md index 4cbd50f62..86b30f082 100644 --- a/README.md +++ b/README.md @@ -240,11 +240,15 @@ next Codex session. opencodex keeps these behaviors: - **Existing sessions keep affinity.** A thread id is bound to the selected account and reused on later turns, so a long request or a mobile/SSH-attached session keeps using the same account. + Pausing an account clears its affinity map: in-flight requests keep captured credentials, but + subsequent turns are re-routed and cannot reuse the paused account. - **New sessions can auto-route.** When auto-switch is enabled, opencodex compares the hottest known quota window across 5h, weekly, and 30d usage, then picks a lower-usage eligible account for new sessions once the active account crosses the threshold. - **Quota lookup is built in.** The dashboard can refresh all account quotas in one click, and the - request log labels pool traffic with non-PII account ordinals. + request log labels pool traffic with non-PII account ordinals. **Pause exhausted** refreshes + eligible accounts that have credentials and pauses only those whose relevant quota window is + freshly confirmed at 100%; accounts without credentials and unknown or failed refreshes stay unchanged. - **Failures fail closed.** Token failures mark reauthentication instead of falling back to another credential silently; 429 quota responses put the account in cooldown and can fail over future work to another eligible pool account. @@ -505,6 +509,9 @@ The public docs — install, providers, routing, sidecars, Codex integration, Co Maintainer source-of-truth notes live under [`structure/`](./structure). Historical investigations remain under [`docs/`](./docs). Contributor setup lives in [`CONTRIBUTING.md`](./CONTRIBUTING.md), and security reporting guidance lives in [`SECURITY.md`](./SECURITY.md). +Report undisclosed vulnerabilities privately through +[GitHub private vulnerability reporting](https://github.com/lidge-jun/opencodex/security/advisories/new), +not a public issue. ## Development diff --git a/SECURITY.md b/SECURITY.md index a9802c920..b39af65d2 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -17,13 +17,19 @@ or the latest published package before triage continues. Please avoid posting undisclosed vulnerabilities as public GitHub issues. -- Prefer this repository's GitHub private vulnerability reporting or GitHub Security Advisory flow - when that option is available in the repository UI. -- If no private reporting option is available, do not include exploit details, secrets, or live - targets in a public issue. Open a minimal issue that asks maintainers for a safe coordination path. -- Include affected versions, reproduction steps, impact, and any required configuration details. +Report privately through GitHub private vulnerability reporting, which is enabled on this +repository: -The project does not publish a dedicated private security email in this repository. +**** + +The same form is reachable from the repository's **Security** tab under **Report a vulnerability**. +It is private between you and the maintainers, and it is the only channel this project offers for +undisclosed vulnerabilities — there is no dedicated private security email. + +Include affected versions, reproduction steps, impact, and any required configuration details. + +If the form is ever unreachable for you, open a minimal public issue that asks maintainers for a +safe coordination path. Do not include exploit details, secrets, or live targets in that issue. ## Response Expectations diff --git a/bin/ocx.mjs b/bin/ocx.mjs index 0218832de..eb9e734c4 100755 --- a/bin/ocx.mjs +++ b/bin/ocx.mjs @@ -14,6 +14,7 @@ import { existsSync, readFileSync, readdirSync, statSync } from "node:fs"; import { homedir } from "node:os"; import { dirname, join, resolve } from "node:path"; import { fileURLToPath } from "node:url"; +import { npmInvocation } from "../src/update/npm-invocation.mjs"; import { handoffWindowsTrayForUpdate, planWindowsTrayUpdate } from "../src/update/tray-update-plan.mjs"; const PKG = "@bitkyc08/opencodex"; @@ -29,10 +30,6 @@ function isBunGlobalInstall() { return /[\\/]\.bun[\\/]/.test(here); } -function npmBin() { - return process.platform === "win32" ? "npm.cmd" : "npm"; -} - function currentPackageVersion() { try { return JSON.parse(readFileSync(join(here, "..", "package.json"), "utf8")).version ?? "?"; @@ -115,15 +112,17 @@ function runTrayLifecycle(launcher, action) { function runNpmSelfUpdate() { const current = currentPackageVersion(); const tag = updateTag(current); - const npm = npmBin(); - // Node ≥18.20/20.12 refuses to spawn .cmd/.bat without a shell (CVE-2024-27980 - // hardening) — spawning "npm.cmd" shell-less throws EINVAL on Windows. - const winShell = process.platform === "win32"; - const latestResult = spawnSync(npm, ["view", `${PKG}@${tag}`, "version"], { + const latestInvocation = npmInvocation(["view", `${PKG}@${tag}`, "version"]); + const installInvocation = npmInvocation(["install", "-g", `${PKG}@${tag}`]); + if (!latestInvocation || !installInvocation) { + console.error("opencodex: could not resolve npm from a trusted absolute PATH entry; aborting before stopping the proxy."); + process.exit(1); + } + const latestResult = spawnSync(latestInvocation.file, latestInvocation.args, { encoding: "utf8", timeout: 12000, windowsHide: true, - shell: winShell, + ...latestInvocation.options, }); const latest = latestResult.status === 0 ? latestResult.stdout.trim() : ""; @@ -221,12 +220,12 @@ function runNpmSelfUpdate() { } } - console.log(`Updating${latest ? ` to v${latest}` : ""}...\n$ ${npm} install -g ${PKG}@${tag}`); - const res = spawnSync(npm, ["install", "-g", `${PKG}@${tag}`], { + console.log(`Updating${latest ? ` to v${latest}` : ""}...\n$ npm install -g ${PKG}@${tag}`); + const res = spawnSync(installInvocation.file, installInvocation.args, { stdio: "inherit", timeout: 180000, windowsHide: true, - shell: winShell, + ...installInvocation.options, }); if (res.status === 0) { console.log(`\nUpdated${latest ? ` to v${latest}` : ""}.`); @@ -278,7 +277,7 @@ function runNpmSelfUpdate() { process.exit(0); } if (trayBeforeUpdate.restoreOnFailure) runTrayLifecycle(launcher, "start"); - console.error(`\nUpdate failed (${npm} exit ${res.status ?? "?"}). Try manually: ${npm} install -g ${PKG}@${tag}`); + console.error(`\nUpdate failed (npm exit ${res.status ?? "?"}). Try manually: npm install -g ${PKG}@${tag}`); process.exit(1); } diff --git a/devlog b/devlog new file mode 160000 index 000000000..6c74ce6a2 --- /dev/null +++ b/devlog @@ -0,0 +1 @@ +Subproject commit 6c74ce6a2260f379ae16b7c4b1ff7ac90da0ab97 diff --git a/devlog/.DS_Store b/devlog/.DS_Store deleted file mode 100644 index 9ac38fcbe..000000000 Binary files a/devlog/.DS_Store and /dev/null differ diff --git a/devlog/_chase/00_parity-index.md b/devlog/_chase/00_parity-index.md deleted file mode 100644 index 996e31985..000000000 --- a/devlog/_chase/00_parity-index.md +++ /dev/null @@ -1,36 +0,0 @@ -# parity 분석 인덱스 (3 upstream) - -opencodex 프록시 레이어가 따라잡는 세 upstream의 parity 스냅샷. 각 문서는 -분석 시점의 upstream HEAD를 박아둔다. 재분석 시 HEAD를 갱신하고 delta만 본다. - -## 분석 baseline HEAD (2026-07-01 분석) - -| slug | repo | 분석 HEAD | 날짜 | 문서 | -|---|---|---|---|---| -| jawcode (gjc) | `lidge-jun/jawcode` | `27311f6` | 2026-07-01 04:35 | `10_jawcode.md` | -| cli-proxy-api (cca) | `router-for-me/CLIProxyAPI` | `00114be` | 2026-06-29 18:59 | `20_cli-proxy-api.md` | -| litellm | `BerriAI/litellm` | `be4d0d8` | 2026-06-30 12:25 | `30_litellm.md` | - -> jawcode 로컬 HEAD는 분석 직후 `a06f814`(docs-only, `docs(chase): close 10.062 → _fin`)로 -> 한 커밋 앞섰다. 코드 변경 없음 — 분석 baseline은 마지막 코드 커밋 `27311f6`이 정확하다. - -opencodex baseline (분석 시): registry provider **48개**, adapter 종류 6 -(`openai-chat`×37, `anthropic`×4, `google`×3, `openai-responses`×2, `kiro`×1, `azure-openai`×1). - -## 세 upstream의 성격이 다르다 (parity 축도 다름) - -| upstream | 성격 | opencodex와의 관계 | parity 축 | -|---|---|---|---| -| jawcode | TS 멀티-프로바이더 AI 패키지 (47 provider 모듈) | **직접 포팅 출처** (1차 SOT) | provider-by-provider wire/auth 1:1 대조 | -| cli-proxy-api | Go OAuth IDE 프록시 (antigravity/codex/claude/kimi/vertex/xai) | wire/auth/quirks **외부 교차검증** (2차) | executor/translator/signature 동작 대조 | -| litellm | Python 범용 SDK (130 provider 폴더, 모델가격 맵) | 모델 카탈로그/커버리지 폭 **참조** (3차) | provider·model 커버리지 폭, 컨텍스트 윈도우 | - -## 분석 방법 - -- jawcode: 로컬 `/Users/jun/Developer/new/700_projects/jawcode` (clone 불필요) -- cca: `devlog/_chase/_cca/` (shallow clone, gitignored) -- litellm: `devlog/_chase/_litellm/` (shallow clone, gitignored) - -클론 디렉터리는 `devlog/` 통째 gitignore로 자동 무시. 노트(`*.md`)만 `git add -f`. - -세부는 각 문서 참조. diff --git a/devlog/_chase/01_overview.md b/devlog/_chase/01_overview.md deleted file mode 100644 index 0baaed54e..000000000 --- a/devlog/_chase/01_overview.md +++ /dev/null @@ -1,63 +0,0 @@ -# chase — 개요 - -## 한 줄 - -**chase** = opencodex가 세 upstream 대비 **아직 안 따라온** 영역 + **무엇을 어떻게 따라갈지**의 참조 방안. -jawcode `struct_har/chase`와 동형 — 단, opencodex는 이미 jawcode 포팅을 끝냈고 별도 -`structure/`(patched SoT)가 있으므로 무거운 form-snapshot 트리는 두지 않고 **행동 레이어만** 둔다. - -## form vs action 역할 분담 (jawcode struct_har ↔ chase 대응) - -| form (스냅샷) | action (행동) | -|---|---| -| `10_jawcode.md` · `20_cli-proxy-api.md` · `30_litellm.md` | `01_overview` · `02_gap_inventory` · `03_follow_index` | -| upstream 규모·HEAD·wire 대조 정본 | 갭·다음에 볼 경로·완료 기준·우선순위 | -| 분석 시 수동 갱신 (HEAD 박기) | fetch·diff·상태 재평가 | - -opencodex에는 jawcode의 `gjc_origin/`·`jwc_patched/` 밴드 트리에 해당하는 게 `src/`(코드 정본)와 -`structure/`(아키텍처 SoT)다. 그래서 chase는 그 위에 **갭 추적**만 얹는다. - -## 갭 4종 (opencodex 맞춤) - -| 종류 | 설명 | 주 upstream | -|---|---|---| -| **G1 jawcode drift** | jawcode가 새 provider/모델을 추가해 opencodex가 뒤처짐 | jawcode | -| **G2 cca hardening** | CLIProxyAPI가 wire/auth/replay를 더 깊게 처리 | cli-proxy-api | -| **G3 catalog/longtail** | litellm 모델 맵·openai-호환 롱테일 provider 커버리지 | litellm | -| **G4 opencodex-only** | opencodex가 앞서거나 유일한 영역 (kiro·codex WS·구독 IDE 백엔드) | — | - -## 상태 어휘 - -`⬜` 미착수 · `🟡` 부분/설계 · `✅` opencodex 선행 또는 포팅 완료 · `—` 범위 밖 - -## 우선순위 (착수) - -1. **G1 선별** — jawcode HEAD 신규(`deepinfra`)·`models.json` delta 동기화 (저비용) -2. **G2 깊이** — antigravity replay 깊이(136→667 대조), vertex OAuth, xai WS -3. **G3 카탈로그** — litellm 컨텍스트 윈도우/가격 교차검증, openai-호환 롱테일 -4. **G1 고비용** — `cursor` agent 어댑터 (별도 work-phase) - -세부는 `02_gap_inventory.md`, 실행 순은 `03_follow_index.md`. - -## worktree에서 갭 검증 (스니펫) - -카드가 "⬜ gap"이라 쓸 때, 먼저 opencodex에 이미 있는지 grep으로 확인한다. - -```bash -# cursor 어댑터 (G1) -ls src/adapters/ | grep -i cursor -# deepinfra (G1) — jawcode HEAD 신규 -rg -i 'deepinfra' src/providers/registry.ts -# vertex OAuth (G2) — opencodex는 key/ADC만 -rg -il 'vertex.*oauth' src/oauth/ -# antigravity replay 깊이 (G2) -wc -l src/adapters/google-antigravity-replay.ts -``` - -## 읽기 순서 - -1. `00_parity-index.md` — 세 HEAD 기준선 -2. 이 문서 — 갭 분류·우선순위 -3. `02_gap_inventory.md` — 축별 요약 -4. `03_follow_index.md` — 실행 순서 -5. `10_/20_/30_` — upstream별 상세 diff --git a/devlog/_chase/02_gap_inventory.md b/devlog/_chase/02_gap_inventory.md deleted file mode 100644 index bfa24b9ab..000000000 --- a/devlog/_chase/02_gap_inventory.md +++ /dev/null @@ -1,41 +0,0 @@ -# chase — 갭 인벤토리 (횡단) - -> **reviewed through**: jawcode `27311f6` (code) · cli-proxy-api `00114be` · litellm `be4d0d8` (2026-07-01) -> 상태: `⬜` 미착수 · `🟡` 부분/설계 · `✅` opencodex 선행/포팅 완료 · `—` 범위 밖 -> 기록: `10_jawcode.md` · `20_cli-proxy-api.md` · `30_litellm.md` - -## 요약 (축별 앞섬/뒤처짐) - -| 축 | opencodex가 **앞서거나 유일** | opencodex가 **뒤처지거나 약함** | -|---|---|---| -| **jawcode** | kiro 9-모듈 분화, antigravity replay 분리, codex WS, 구독 IDE provider | `cursor`(미포팅), `deepinfra`(HEAD 신규), 직접 `amazon-bedrock` sigv4 | -| **cca** | kiro 풀세트, codex WS bridge, 단일 google adapter 흡수 | antigravity replay 깊이(136 vs 667), vertex OAuth, xai reasoning-replay/WS | -| **litellm** | OAuth/구독 IDE 백엔드, 코딩 에이전트 특화 라우팅 | chat 롱테일 provider 폭(cohere·databricks·ai21 등), 모델 가격/컨텍스트 맵 | - -## 항목별 (G-tag) - -| 갭 | 종류 | 상태 | upstream 근거 | opencodex 위치 | -|---|---|---|---|---| -| cursor 어댑터 | G1 | ⬜ | jawcode `cursor.ts` (~2.6k줄 agent 프로토콜) | `src/adapters/`에 없음 | -| deepinfra provider | G1 | ⬜ | jawcode HEAD `27311f6` 신규(10.062) | `registry.ts`에 없음 | -| 직접 amazon-bedrock | G1 | 🟡 | jawcode `amazon-bedrock.ts` + `aws-sigv4.ts` | kiro adapter로 Bedrock-on-Kiro만 (`eventstream-decoder.ts`는 kiro용) | -| google-gemini-cli OAuth | G1 | — | jawcode `google-gemini-cli.ts` | 레거시. opencodex migration 대상에서 제외 | -| models.json delta | G1 | 🟡 | jawcode 3758 모델 엔트리 | `codex-catalog.ts` + `*-models.ts` 주기 동기화 | -| antigravity replay 깊이 | G2 | 🟡 | cca `antigravity_reasoning_replay.go` (667줄) | `google-antigravity-replay.ts` (136줄) | -| vertex OAuth | G2 | ⬜ | cca `vertex_credentials.go` + vertex auth | `google` adapter는 key/ADC만, OAuth 없음 | -| xai reasoning-replay/WS | G2 | ⬜ | cca `xai_reasoning_replay.go` + `xai_websockets_executor.go` | xai 전용 어댑터 없음 (registry forward) | -| responses WS→SSE | G2 | ✅ | cca HEAD 머지 #4048 | `ws-bridge.ts` (양쪽 최신, delta 점검) | -| chat 롱테일 provider | G3 | ⬜ | litellm cohere·databricks·ai21·friendliai 등 | registry에 없음 (대부분 openai-호환, 한 줄 추가 가능) | -| 모델 가격/컨텍스트 맵 | G3 | 🟡 | litellm `model_prices_and_context_window_backup.json` (2910) | 카탈로그 컨텍스트 윈도우 교차검증 소스 | -| 멀티모달(embedding/tts/image/rerank) | G3 | — | litellm mode 분포 | opencodex 범위 밖 (chat/responses 프록시) | -| kiro 풀세트 | G4 | ✅ | (upstream에 없거나 약함) | `kiro*.ts` 9 모듈 | -| codex WebSocket | G4 | ✅ | cca codex WS와 동급 | `ws-bridge.ts` + `codex-websocket-registry.ts` | -| 구독 IDE provider | G4 | ✅ | — | umans·opencode-go·neuralwatt·zai 등 | - -## 다음에 볼 경로 - -- G1 deepinfra: `jawcode/packages/ai/src/providers/` (HEAD `27311f6` 신규 파일) → opencodex registry 한 줄 + 모델 시드 -- G2 replay: `devlog/_chase/_cca/internal/runtime/executor/antigravity_reasoning_replay.go` line 대조 -- G3 catalog: `devlog/_chase/_litellm/litellm/model_prices_and_context_window_backup.json` 컨텍스트 윈도우 값 - -실행 우선순위는 `03_follow_index.md`. diff --git a/devlog/_chase/03_follow_index.md b/devlog/_chase/03_follow_index.md deleted file mode 100644 index 25461a3e9..000000000 --- a/devlog/_chase/03_follow_index.md +++ /dev/null @@ -1,34 +0,0 @@ -# chase — 실행 인덱스 - -> current through: OpenCodex source and model chase re-audit on 2026-07-17 -> durable roadmap: [`devlog/_plan/260717_non_openai_provider_chase/000_plan.md`](../_plan/260717_non_openai_provider_chase/000_plan.md) - -오래된 7월 1일 분류의 “Cursor 어댑터 미포팅”, “xAI 전용 경로 없음”은 현재 소스와 맞지 않는다. 아래 표는 OpenAI와 xAI를 제외한 다음 실행 순서다. 한 행은 한 PABCD work-phase이며, 정확한 diff와 검증 기준은 연결된 decade 문서가 정본이다. - -## 실행 순서 - -| WP | 항목 | 결정 | 계획 | -|---|---|---|---| -| 1 | Sakana Fugu/Fugu Ultra direct | `ADAPT` | [`010`](../_plan/260717_non_openai_provider_chase/010_fugu_sakana_direct.md) | -| 2 | Cursor client-version owner | `ADAPT` after live probe | [`020`](../_plan/260717_non_openai_provider_chase/020_cursor_client_version_owner.md) | -| 3 | Antigravity indexed replay + picker/alias split | `RESEARCH` → `ADAPT/NOOP` | [`030`](../_plan/260717_non_openai_provider_chase/030_antigravity_replay_alias.md) | -| 4 | OpenCode Go Kimi effort | `RESEARCH` → `ADAPT/NOOP` | [`040`](../_plan/260717_non_openai_provider_chase/040_kimi_effort_matrix.md) | -| 5 | Z.AI weekly-limit classifier | fixture-gated | [`050`](../_plan/260717_non_openai_provider_chase/050_zai_weekly_limit.md) | -| 6 | Anthropic indexed stream/tool replay | fixture-gated | [`060`](../_plan/260717_non_openai_provider_chase/060_anthropic_stream_replay.md) | -| 7 | consumer-backed metadata | `ADAPT/NOOP` | [`070`](../_plan/260717_non_openai_provider_chase/070_metadata_consumers.md) | -| 8 | DeepInfra | `ADAPT` | [`080`](../_plan/260717_non_openai_provider_chase/080_deepinfra_provider.md) | -| 9 | Cohere compatibility API | `ADAPT` | [`090`](../_plan/260717_non_openai_provider_chase/090_cohere_provider.md) | -| 10 | AI21 Jamba | `ADAPT` | [`100`](../_plan/260717_non_openai_provider_chase/100_ai21_provider.md) | -| 11 | Databricks workspace serving | `ADAPT`, workspace-bound | [`110`](../_plan/260717_non_openai_provider_chase/110_databricks_provider.md) | -| 12 | Amazon Bedrock Mantle | `ADAPT`, OpenAI-compatible first | [`120`](../_plan/260717_non_openai_provider_chase/120_bedrock_mantle.md) | -| 13 | Vertex ADC/OAuth setup UX | existing auth productization | [`130`](../_plan/260717_non_openai_provider_chase/130_vertex_adc_oauth.md) | -| 14 | Native Bedrock Runtime | Mantle-gap only; optional SigV4 | [`140`](../_plan/260717_non_openai_provider_chase/140_bedrock_runtime_sigv4.md) | -| 15 | integration and chase closure | final gate | [`150`](../_plan/260717_non_openai_provider_chase/150_integration_closeout.md) | - -## opencodex 선행 (유지·회귀 방지) - -Kiro 풀세트, Codex WS, xAI OAuth/replay/live discovery는 따라잡을 대상이 아니라 현재 OpenCodex 소유 경로다. 이 로드맵은 해당 경로를 건드리지 않는다. - -## 갱신 규칙 - -각 work-phase의 P에서 현재 source/official contract를 재검증하고, D에서 decade 문서와 model chase 표를 함께 갱신한다. 전체 종료 시 `150`이 이 인덱스와 일반 문서를 최종 동기화한다. diff --git a/devlog/_chase/10_jawcode.md b/devlog/_chase/10_jawcode.md deleted file mode 100644 index 2115e69e9..000000000 --- a/devlog/_chase/10_jawcode.md +++ /dev/null @@ -1,84 +0,0 @@ -# 10 — jawcode parity - -- repo: `lidge-jun/jawcode` · 로컬: `/Users/jun/Developer/new/700_projects/jawcode` -- **분석 HEAD: `27311f6`** (2026-07-01 04:35, `fix(ai): type DeepInfra tokenizer test cases as TokenizerFamily (10.062)`) -- 역할: opencodex 어댑터의 **직접 포팅 출처** (1차 SOT). provider 레이어가 여기 산다. - -## chase MOC - -> 상태: 🟡 운영 중 · **의미**: jawcode 대비 opencodex **뒤처짐(G1)** -> 상태 어휘: `⬜` 미착수 · `🟡` 부분 · `✅` opencodex 선행/완료 · `—` 범위 밖 - -### Reviewed through - -| jawcode | opencodex | -|---|---| -| `27311f6` (code, 2026-07-01) | registry 48 provider / adapter 6종 (worktree) | - -### Recent jawcode deltas - -| 항목 | jawcode fact | opencodex 처리 | -|---|---|---| -| deepinfra | HEAD `27311f6` 신규 provider + tokenizer family 라우팅 (10.062) | ⬜ registry 미반영 | -| cursor | `cursor.ts` ~2.6k줄 agent 프로토콜 | ⬜ 어댑터 없음 | -| amazon-bedrock | `amazon-bedrock.ts` + `aws-sigv4.ts` 직접 경로 | 🟡 kiro adapter로 Bedrock-on-Kiro만 | -| google-gemini-cli | `google-gemini-cli.ts` OAuth | — 레거시. opencodex migration 대상에서 제외 | -| kiro | `kiro.ts` + 4 모듈 | ✅ 9 모듈로 분화 (opencodex 선행) | -| models.json | 3758 모델 엔트리 | 🟡 카탈로그 주기 동기화 필요 | - -## 규모 대조 (HEAD 기준) - -| 지표 | jawcode | opencodex | -|---|---|---| -| provider 모듈 (`packages/ai/src/providers/*.ts`, non-test) | 47 | adapter 24개 파일 | -| registry/카탈로그 provider | `models.json` **48** provider, **3758** 모델 엔트리 | registry **48** provider | -| 모델 카탈로그 소스 | `packages/ai/src/models.json` (81k줄) | `src/codex-catalog.ts` + `src/providers/*-models.ts` | - -jawcode는 멀티-API 클라이언트 패키지(google-generative-ai / openai-completions / -openai-responses / anthropic-messages / kiro 등 `api` 필드로 분기)이고, opencodex는 -그 API 패밀리를 **6개 adapter**로 압축했다: `openai-chat`×37, `anthropic`×4, -`google`×3, `openai-responses`×2, `kiro`×1, `azure-openai`×1. - -## provider 커버리지 diff (exact-id 기준, 39 공유) - -jawcode에만 있고 opencodex registry에 같은 id로 없는 것: -`alibaba-coding-plan`(opencodex는 `alibaba`로 보유), `amazon-bedrock`, -`cursor`, `deepinfra`, `google-gemini-cli`(legacy/out-of-scope), `minimax-code`, `minimax-code-cn`, -`openai-codex`, `opencode`. - -opencodex에만 있는 것: `kimi`, `lm-studio`, `neuralwatt`, `openai-apikey`, -`parallel`, `umans`, `vllm`, `ollama`(jawcode는 `ollama-cloud`), `alibaba`. - -### 의미 있는 gap (id 차이가 아닌 실제 미포팅) - -| jawcode provider | 상태 | 메모 | -|---|---|---| -| `cursor` | **미포팅** | jawcode `cursor.ts`는 ~2.6k줄 agent 프로토콜. opencodex에 adapter 없음. 가장 큰 단일 gap | -| `deepinfra` | **미포팅** | HEAD에서 막 추가된 신규(10.062). tokenizer family 라우팅 포함 | -| `amazon-bedrock` | 부분 | opencodex는 `kiro` adapter로 Bedrock-on-Kiro만. 직접 Bedrock(sigv4) 경로는 없음 | -| `google-gemini-cli` | — | 레거시 경로. opencodex는 ai-studio/vertex/cloud-code-assist만 유지하고 gemini-CLI OAuth는 migration하지 않음 | - -## API 변환 1:1 대조 포인트 - -| 영역 | jawcode | opencodex | -|---|---|---| -| Google generate | `google.ts` `streamGenerateContent?alt=sse`, `x-goog-api-key` | `src/adapters/google.ts` (동일 엔드포인트, vertex/cca 분기 추가) | -| Google 공통 | `google-shared.ts` `buildGoogleGenerateContentParams` | `google.ts` + `google-tool-schema.ts` + `google-truncation.ts` | -| Antigravity | `google.ts` cloud-code 분기 | `google-antigravity-wire.ts`(100), `google-antigravity-replay.ts`(136) | -| Kiro | `kiro.ts` + `kiro-{thinking,truncation,usage,tool-fallback}.ts` | `kiro.ts` + `kiro-{events,images,retry,thinking,tool-fallback,tools,truncation,wire,errors}.ts` (더 분화) | - -opencodex는 Kiro를 jawcode보다 더 잘게 쪼갰고(9 모듈 vs 5), Antigravity는 thought-signature -replay를 별도 모듈로 뺐다. 즉 **포팅 후 opencodex가 더 하드닝된 영역**(kiro/antigravity)과 -**아직 안 따라온 영역**(cursor/deepinfra/direct-bedrock)이 공존한다. - -## 따라잡을 우선순위 - -1. `deepinfra` — HEAD 신규, 가벼운 openai-chat 계열이라 포팅 저비용. -2. `cursor` — 고비용/고가치. agent 프로토콜이라 별도 work-phase 필요. -3. `models.json` delta — 3758 엔트리 중 opencodex 카탈로그에 없는 신규 모델·컨텍스트 윈도우 주기 동기화. - -## chase 로그 - -| 날짜 | jawcode HEAD | 분석 내용 | 결과 | -|---|---|---|---| -| 2026-07-01 | 27311f6 | provider 48/모델 3758 baseline, 커버리지 diff, cursor/deepinfra gap 식별 | 이 문서 | diff --git a/devlog/_chase/20_cli-proxy-api.md b/devlog/_chase/20_cli-proxy-api.md deleted file mode 100644 index fc6f20cc6..000000000 --- a/devlog/_chase/20_cli-proxy-api.md +++ /dev/null @@ -1,73 +0,0 @@ -# 20 — cli-proxy-api (CLIProxyAPI) parity - -- repo: `router-for-me/CLIProxyAPI` (Go, MIT) -- 로컬: `devlog/_chase/_cca/` (shallow clone, gitignored) -- **분석 HEAD: `00114be`** (2026-06-29 18:59, `Merge PR #4052 fix/responses-ws-to-sse-4048`) -- 역할: wire/auth/quirks **외부 교차검증 SOT** (2차). 코드 복사 아님, 동작 대조. - -## chase MOC - -> 상태: 🟡 운영 중 · **의미**: cca 대비 opencodex wire/auth 깊이 **부분 뒤처짐(G2)** -> 상태 어휘: `⬜` 미착수 · `🟡` 부분 · `✅` opencodex 선행/완료 · `—` 범위 밖 - -### Reviewed through - -| cli-proxy-api | opencodex | -|---|---| -| `00114be` (2026-06-29) | oauth 6 provider / `ws-bridge.ts` (worktree) | - -### Recent cca deltas - -| 항목 | cca fact | opencodex 처리 | -|---|---|---| -| responses WS→SSE | HEAD 머지 #4048 `codex_websockets_executor` | ✅ `ws-bridge.ts` (delta 점검 대상) | -| antigravity replay | `antigravity_reasoning_replay.go` 667줄 | 🟡 `google-antigravity-replay.ts` 136줄 | -| vertex OAuth | `vertex_credentials.go` + vertex auth | ⬜ key/ADC만, OAuth 없음 | -| xai replay/WS | `xai_reasoning_replay.go` + `xai_websockets_executor.go` | ⬜ 전용 어댑터 없음 | -| kiro | (없음) | ✅ 풀세트 (opencodex 선행) | - -## 규모 (HEAD 기준, non-test SLOC) - -| 영역 | CCA SLOC | opencodex 대응 | -|---|---|---| -| `internal/auth` | 4,755 | `src/oauth/*` (anthropic/chatgpt/google-antigravity/kimi/kiro/xai) | -| `internal/translator` | 18,086 | adapter별 body 변환 (`google.ts`, `openai-*.ts`, `anthropic.ts`) | -| `internal/runtime/executor` | 22,805 | `buildRequest`/`parseStream` + 안정화 + `ws-bridge.ts` | - -## auth/executor 커버리지 대조 - -CCA auth provider: `antigravity, claude, codex, kimi, vertex, xai` (+empty). -opencodex oauth: `anthropic, chatgpt(=codex), google-antigravity, kimi, kiro, xai`. - -| provider | CCA | opencodex | 차이 | -|---|---|---|---| -| antigravity | auth + executor + reasoning replay | oauth + `google-antigravity-{wire,replay}.ts` | **양쪽 보유** | -| codex/chatgpt | `codex_executor.go` + `codex_websockets_executor.go` | oauth chatgpt + `ws-bridge.ts` | **양쪽 보유** (WS parity) | -| claude | `claude_executor.go` + `claude_signing.go` | `anthropic.ts` oauth | 양쪽 보유 | -| kimi | `kimi_executor.go` | oauth kimi | 양쪽 보유 | -| vertex | `gemini_vertex_executor.go` + vertex auth | `google` adapter `googleMode:vertex` (**key 기반, OAuth 아님**) | **gap: vertex OAuth 경로 없음** | -| xai | `xai_executor.go` + `xai_websockets_executor.go` + `xai_reasoning_replay.go` | registry `xai`(forward), oauth xai | **gap: xai reasoning replay / WS executor 미확인** | -| kiro | (없음) | `kiro` adapter 풀세트 | opencodex 우위 | - -## 핵심 wire quirk 교차검증 - -| quirk | CCA 위치 | opencodex 위치 | 상태 | -|---|---|---|---| -| Antigravity thoughtSignature replay | `antigravity_reasoning_replay.go` (667줄) | `google-antigravity-replay.ts` (136줄) | opencodex가 훨씬 가벼움 — **replay 캐시 깊이 차이 점검 필요** | -| Antigravity signature validation | `translator/antigravity/claude/signature_validation.go` | `google-antigravity-wire.ts` `isLikelyRealThoughtSignature` | 접근 동일(synthetic id 거부), CCA가 더 세분 | -| Responses WS→SSE | HEAD 머지(#4048) `codex_websockets_executor` | `ws-bridge.ts` | **양쪽 최신, delta 점검 대상** | -| Antigravity translators | gemini/claude/openai 3종 | `google` adapter 단일 | opencodex는 단일 adapter로 흡수 | - -## 따라잡을 우선순위 - -1. **Antigravity reasoning replay 깊이** — CCA 667줄 vs opencodex 136줄. clear-on-invalid, - 캐시 키, 멀티턴 시그니처 보존 로직을 라인 대조해 누락 확인. -2. **vertex OAuth** — opencodex는 key/ADC만. CCA `vertex_credentials.go` 흐름을 참조해 - 서비스계정 OAuth 경로 보강 여부 결정. -3. **xai reasoning replay / WS** — CCA에 전용 모듈 존재. opencodex xai는 forward-only인지 확인. - -## chase 로그 - -| 날짜 | CCA HEAD | 분석 내용 | 결과 | -|---|---|---|---| -| 2026-07-01 | 00114be | auth/executor 커버리지 대조, replay 깊이 gap, vertex OAuth/xai WS gap 식별 | 이 문서 | diff --git a/devlog/_chase/30_litellm.md b/devlog/_chase/30_litellm.md deleted file mode 100644 index 8cbb1c0f6..000000000 --- a/devlog/_chase/30_litellm.md +++ /dev/null @@ -1,76 +0,0 @@ -# 30 — litellm parity - -- repo: `BerriAI/litellm` (Python, MIT) -- 로컬: `devlog/_chase/_litellm/` (shallow clone, gitignored) -- **분석 HEAD: `be4d0d8`** (2026-06-30 12:25, `fix(redis): re-establish async cluster connections after a node restart (#31577)`) -- 역할: 모델 카탈로그 / provider 커버리지 폭 **참조** (3차). 직접 포팅 출처 아님. - -## chase MOC - -> 상태: 🟡 운영 중 · **의미**: litellm 대비 opencodex 카탈로그 폭 **부분 뒤처짐(G3)**, 범위는 다름 -> 상태 어휘: `⬜` 미착수 · `🟡` 부분 · `✅` opencodex 선행/완료 · `—` 범위 밖 - -### Reviewed through - -| litellm | opencodex | -|---|---| -| `be4d0d8` (2026-06-30) | registry 48 provider (worktree) | - -### Recent litellm deltas - -| 항목 | litellm fact | opencodex 처리 | -|---|---|---| -| 모델 가격/컨텍스트 맵 | `model_prices_and_context_window_backup.json` 2910 모델 | 🟡 교차검증 소스로 활용 | -| chat 롱테일 provider | cohere·databricks·ai21·friendliai 등 130 폴더 | ⬜ openai-호환 후보 미추가 | -| 멀티모달 (embedding/tts/image/rerank/ocr) | mode 분포 chat 2247 외 ~660 | — opencodex 범위 밖 | -| OAuth/구독 IDE 백엔드 | (litellm 약함) | ✅ kiro·antigravity·umans 등 (opencodex 선행) | - -## 규모 (HEAD 기준) - -| 지표 | litellm | opencodex | -|---|---|---| -| provider 폴더 (`litellm/llms/*/`) | 130 | adapter 6종 / registry 48 provider | -| 모델 맵 엔트리 (`model_prices_and_context_window_backup.json`) | 2,910 | registry/카탈로그 기반 (provider별 동적) | -| distinct `litellm_provider` | 121 | 48 | - -## 결정적 차이: 범위(scope)가 다르다 - -litellm 모델 맵 mode 분포: `chat` 2247, `image_generation` 203, `embedding` 124, -`responses` 82, `audio_transcription` 61, `completion` 36, `image_edit` 31, -`audio_speech` 27, `rerank` 25, `video_generation` 25, `search` 18, `ocr` 13, -`moderation` 5, `realtime` 2. - -opencodex는 **코딩 에이전트용 chat/responses 라우팅 프록시**다. litellm의 -embedding/tts/stt/image/rerank/ocr/moderation은 opencodex 범위 밖 — parity 대상이 아니다. -의미 있는 교집합은 litellm의 **chat(2247) + responses(82)** 약 2.3k 모델뿐이다. - -## chat-relevant provider 교집합 - -litellm chat provider 중 opencodex와 겹치는 것: -`openai, anthropic, azure, azure_ai, bedrock, cerebras, cohere, deepseek, -fireworks_ai, gemini, groq, mistral, together_ai, vertex_ai, xai` 등. - -opencodex에 있고 litellm 표준 provider에 없는 것(IDE/구독 프록시 특화): -`kiro`, `google-antigravity`, `umans`, `opencode-go`, `neuralwatt`, `cursor`(미포팅), -`zai`, `qwen-portal`, `kimi-code` 등 — opencodex의 차별점은 **OAuth/구독 기반 IDE 백엔드**다. - -litellm에 있고 opencodex가 안 다루는 chat provider(폭): `cohere`, `databricks`, -`ai21`, `baseten`, `friendliai`, `featherless_ai`, `galadriel`, `gradient_ai` 등 -롱테일. 대부분 openai-호환이라 필요 시 `openai-chat` adapter + registry 한 줄로 추가 가능. - -## 따라잡을 표면 (좁게) - -1. **모델 가격/컨텍스트 윈도우 맵** — litellm `model_prices_and_context_window_backup.json`은 - 2910 모델의 컨텍스트 윈도우·가격의 사실상 업계 레퍼런스. opencodex 카탈로그의 컨텍스트 - 윈도우 값 교차검증 소스로 유용. -2. **롱테일 openai-호환 provider** — registry 한 줄 추가로 커버 가능한 후보 목록. -3. **provider별 endpoint/헤더 quirk** — `litellm/llms//`의 transformation 로직을 - opencodex가 새 provider 추가 시 wire 레퍼런스로 사용. - -범위가 다르므로 litellm 전체 parity는 목표 아님. **카탈로그 정확도 + openai-호환 롱테일**만 따라간다. - -## chase 로그 - -| 날짜 | litellm HEAD | 분석 내용 | 결과 | -|---|---|---|---| -| 2026-07-01 | be4d0d8 | 130 provider/2910 모델 맵, mode 분포로 scope 차이 확정, chat 교집합·카탈로그 활용처 식별 | 이 문서 | diff --git a/devlog/_chase/README.md b/devlog/_chase/README.md deleted file mode 100644 index 21d7a753b..000000000 --- a/devlog/_chase/README.md +++ /dev/null @@ -1,95 +0,0 @@ -# _chase: upstream 따라잡기 노트 - -opencodex 프록시 레이어가 따라잡는 두 upstream의 추적 노트. 코드를 통째로 -vendoring하지 않는다. 실제로 따라잡을 표면은 좁다: API 변환(translator/wire)과 -모델명/카탈로그 컨텍스트 두 갈래뿐이라, 무거운 클론 대신 노트 + on-demand 대조로 간다. - -## 두 upstream - -| slug | repo | 역할 | 로컬 | -|---|---|---|---| -| `gjc` (jawcode) | `lidge-jun/jawcode` (`packages/ai/src/providers/*.ts`) | opencodex 어댑터의 직접 포팅 출처 (1차 SOT) | `/Users/jun/Developer/new/700_projects/jawcode` 에 이미 존재 | -| `cca` | `router-for-me/CLIProxyAPI` (Go) | wire/auth/quirks 외부 교차검증 SOT (2차) | 없음. 필요 시 `_cca/`로 shallow clone | - -주의: jawcode는 gajae-code(에이전트)가 아니다. 프록시 프로바이더 레이어는 -jawcode에 있고, gajae-code는 별개 코딩 에이전트다. - -## Layout - -```text -devlog/_chase/ - README.md # 이 문서 (tracked) - 00_parity-index.md # 3 upstream parity 인덱스 + 분석 HEAD 기록 (tracked) - 01_overview.md # chase 정의 · 갭 4종(G1-G4) · 우선순위 (tracked) - 02_gap_inventory.md # 축별 앞섬/뒤처짐 + 항목별 G-tag·상태 (tracked) - 03_follow_index.md # 실행 우선순위 Tier 1/2/3 (tracked) - 10_jawcode.md # gjc parity: provider/wire/model 1:1 대조 (tracked) - 20_cli-proxy-api.md # cca parity: executor/auth/translator 교차검증 (tracked) - 30_litellm.md # litellm parity: 카탈로그/커버리지 폭 참조 (tracked) - _model/ # 모델/provider 현재 구조·변경 절차·upstream delta (tracked docs) - _gjc/ # (옵션) jawcode 로컬 클론/심볼릭: gitignored - _cca/ # (옵션) CLIProxyAPI 로컬 클론: gitignored - _litellm/ # (옵션) litellm 로컬 클론: gitignored -``` - -## form vs action (jawcode struct_har ↔ chase와 동형) - -jawcode는 `struct_har/`(형태 스냅샷, 자동생성)와 `struct_har/chase/`(행동, 수동)를 나눈다. -opencodex는 이미 `structure/`(patched SoT)와 `src/`(코드 정본)가 form 역할을 하므로, -chase는 **행동 레이어만** 둔다. - -| form (스냅샷) | action (행동) | -|---|---| -| `10_/20_/30_` dossier + repo `structure/`·`src/` | `01_overview` · `02_gap_inventory` · `03_follow_index` | -| upstream 규모·HEAD·wire 정본 | 갭·다음 경로·완료 기준·우선순위 | - -`devlog/` 전체가 root `.gitignore`에서 무시된다. 노트(`*.md`)는 의도적으로 -force-add(`git add -f`)해서 추적하고, 클론 디렉터리(`_gjc/`, `_cca/`)는 그대로 -무시된 채 둔다. - -## 따라잡는 방식 (vendoring 안 함) - -두 repo 모두 공개라 `cxc-search` 사다리로 원본을 열어 대조한다. clone은 옵션이다. - -1. discover: Tier 1 hosted `web_search`로 해당 영역의 upstream 최신 커밋/파일을 찾는다. -2. prove: 후보 URL(또는 raw 파일)을 열어 실제 wire/모델 목록을 확인한다. 스니펫만으로 확정하지 않는다. -3. diff: opencodex 해당 지점(아래 표)과 동작을 대조하고, 차이를 해당 phase devlog(`_plan/`·`_fin/`)에 file:line 근거로 기록한다. - -GitHub raw로 바로 여는 예: - -```bash -# CLIProxyAPI translator 한 파일 열기 (clone 없이) -curl -fsSL https://raw.githubusercontent.com/router-for-me/CLIProxyAPI/main/internal/translator/antigravity/gemini/antigravity_gemini_request.go -``` - -## (옵션) 로컬 클론 - -대량 grep이나 file:line 인용이 잦아지면 그때만 shallow clone 한다. 자동으로 무시된다. - -```bash -# cca: 로컬에 없으므로 필요 시 -git clone --depth 1 https://github.com/router-for-me/CLIProxyAPI devlog/_chase/_cca - -# gjc: 이미 워크스페이스에 있어 보통 심볼릭이면 충분 -ln -s /Users/jun/Developer/new/700_projects/jawcode devlog/_chase/_gjc -``` - -착수/대조 전 최신화: - -```bash -git -C devlog/_chase/_cca fetch origin && git -C devlog/_chase/_cca log -1 --oneline -git -C /Users/jun/Developer/new/700_projects/jawcode fetch origin && \ - git -C /Users/jun/Developer/new/700_projects/jawcode log -1 --oneline -``` - -## opencodex 대조 표면 (따라잡을 지점) - -| 표면 | opencodex | gjc(jawcode) | cca(CLIProxyAPI) | -|---|---|---|---| -| API 변환 (Google/Antigravity) | `src/adapters/google.ts`, `google-antigravity-wire.ts`, `google-antigravity-replay.ts` | `packages/ai/src/providers/*.ts` | `internal/translator/antigravity/**`, `internal/runtime/executor/antigravity_executor.go` | -| API 변환 (Kiro) | `src/adapters/kiro*.ts` | `packages/ai/src/providers/amazon-bedrock.ts` 등 | (없음) | -| auth/refresh | `src/oauth/*`, `src/lib/gcp-adc.ts` | jawcode oauth utils | `internal/auth/**` | -| 모델/카탈로그 | `src/providers/antigravity-models.ts`, `kiro-models.ts`, `src/codex/catalog.ts` | jawcode `models.json` + static lists | `cmd/fetch_antigravity_models/main.go` | - -모델/provider 변경 전에는 `_model/README.md`를 먼저 읽는다. upstream 비교의 상세 -file:line은 `10_jawcode.md`(gjc), `20_cli-proxy-api.md`(cca) 참고. diff --git a/devlog/_chase/_model/001_provider_inventory.md b/devlog/_chase/_model/001_provider_inventory.md deleted file mode 100644 index 1381f756d..000000000 --- a/devlog/_chase/_model/001_provider_inventory.md +++ /dev/null @@ -1,67 +0,0 @@ -# 001 — Provider inventory - -> Snapshot: 2026-07-17, `dev` @ `31fabf96` - -이 문서는 모델 dump가 아니다. built-in provider의 현재 구조와 소유권만 기록한다. 정확한 모델 ID와 capability는 `src/providers/registry.ts`, live `/models`, `src/generated/jawcode-model-metadata.ts`를 순서대로 확인한다. - -## 현재 수치 - -`PROVIDER_REGISTRY`를 직접 import해 센 결과다. - -| 축 | 수치 | -|---|---| -| built-in provider | 52 | -| auth | key 42, oauth 6, local 3, forward 1 | -| adapter | openai-chat 38, anthropic 5, google 3, openai-responses 2, cursor 1, kiro 1, azure-openai 1, mimo-free 1 | - -Provider ID: - -```text -openai, cursor, xai, anthropic, anthropic-apikey, kimi, kiro, openai-apikey, -umans, opencode-go, neuralwatt, openrouter, groq, google, google-vertex, -google-antigravity, azure-openai, ollama, vllm, lm-studio, deepseek, cerebras, -together, fireworks, firepass, moonshot, huggingface, nvidia, venice, zai, -nanogpt, synthetic, qwen-portal, qianfan, alibaba, parallel, zenmux, litellm, -ollama-cloud, mistral, minimax, minimax-cn, kimi-code, opencode-zen, -vercel-ai-gateway, opencode-free, xiaomi, kilo, mimo-free, -cloudflare-ai-gateway, github-copilot, gitlab-duo -``` - -## Provider 계층 - -| 계층 | 역할 | 현재 소유자 | -|---|---|---| -| registry | id, adapter, base URL, auth kind, static model/capability seed | `src/providers/registry.ts:9-77`, `src/providers/registry.ts:221-679` | -| derived preset | init, dashboard, key-login, OAuth config로 registry 값을 복제 | `src/providers/derive.ts:59-199` | -| persisted config | 사용자 override, selected/disabled models, context cap, key pool | `src/types.ts:348-428`, `src/types.ts:559-607` | -| router | 명시적 namespace와 bare model을 활성 provider에 연결 | `src/router.ts:162-234` | -| adapter | OpenAI/Anthropic/Google/Cursor/Kiro/Azure/MiMo wire로 변환 | `src/server/adapter-resolve.ts:27-49` | -| catalog | live discovery, static fallback, metadata augmentation, Codex sync | `src/codex/catalog.ts:990-1328`, `src/codex/catalog.ts:1478-1569` | - -## Provider군별 성격 - -| 군 | 예 | 주의점 | -|---|---|---| -| native passthrough | `openai` | Pool(기본)은 메인+추가 계정 풀을, Direct는 caller/main만 사용한다. 모드는 provider option이다. | -| OAuth/product token | `xai`, `anthropic`, `kimi`, `kiro`, `google-antigravity`, `cursor` | registry seed와 `src/oauth/index.ts` 구현이 모두 있어야 한다. | -| direct API key | `openai-apikey`, `anthropic-apikey`, `google`, `zai`, `openrouter` | 대부분 registry + 기존 adapter로 충분하다. | -| local/self-hosted | `ollama`, `vllm`, `lm-studio`, `litellm` | private destination 허용과 optional key 정책을 따로 본다. | -| product-specific adapter | `cursor`, `kiro`, `mimo-free` | OpenAI-compatible로 가정하면 안 된다. | -| gateway/aggregator | `openrouter`, `vercel-ai-gateway`, `cloudflare-ai-gateway`, `litellm` | upstream model metadata가 서로 다른 형태로 올 수 있다. | - -OpenAI의 공개 provider는 Pool/Direct 옵션을 가진 `openai`와 API `openai-apikey`다. -레거시 `chatgpt`와 과거 Multi id는 migration 입력일 뿐 public registry provider가 아니다. - -## jawcode metadata bridge - -현재 registry alias가 가리키는 jawcode bundle은 7개다: `xai`, `anthropic`, `moonshot`, `opencode-go`, `openrouter`, `google`, `minimax`. - -2026-07-17 생성 snapshot의 row 수는 각각 30, 25, 1, 19, 349, 34, 9다. 이 수치는 provider 지원 수가 아니라 catalog metadata row 수이며, jawcode 생성물을 갱신하면 달라진다. - -## 검증 - -```bash -bun -e 'import { PROVIDER_REGISTRY } from "./src/providers/registry.ts"; console.log(PROVIDER_REGISTRY.length)' -bun -e 'import { PROVIDER_REGISTRY } from "./src/providers/registry.ts"; console.log(Object.fromEntries(Object.entries(Object.groupBy(PROVIDER_REGISTRY, x => x.adapter)).map(([k,v]) => [k,v.length])))' -rg -n "export const PROVIDER_REGISTRY|export function routeModel|export function resolveAdapter" src/providers/registry.ts src/router.ts src/server/adapter-resolve.ts -``` diff --git a/devlog/_chase/_model/002_catalog_contract.md b/devlog/_chase/_model/002_catalog_contract.md deleted file mode 100644 index 197e1449e..000000000 --- a/devlog/_chase/_model/002_catalog_contract.md +++ /dev/null @@ -1,72 +0,0 @@ -# 002 — Catalog contract - -OpenCodex catalog는 한 파일의 정적 목록이 아니다. 네 입력을 합쳐 Codex가 읽는 catalog와 `/v1/models`를 만든다. - -## 입력과 우선순위 - -```text -PROVIDER_REGISTRY seed - + persisted provider config - + live provider /models (가능한 경우) - + generated jawcode metadata hints - -> visibility/capability normalization - -> Codex catalog sync + models cache invalidation -``` - -| 단계 | 소유자 | 계약 | -|---:|---|---| -| 1 | `src/providers/registry.ts:221` | built-in fallback 모델과 model-scoped capability를 제공한다. | -| 2 | `src/router.ts:79-159` | 오래된 저장 config에 registry metadata를 backfill하고 사용자 override를 보존한다. | -| 3 | `src/codex/catalog.ts:1126` | live `/models`를 TTL cache로 읽는다. 정상 live 응답은 ID 목록의 권위 있는 결과다. | -| 4 | `src/codex/catalog.ts:1094` | `context_length`, `max_model_len`, `metadata.capabilities/limits`를 OCX catalog hint로 바꾼다. | -| 5 | `src/codex/catalog.ts:1297` | registry의 `jawcodeBundle` alias가 있는 provider에 생성된 jawcode metadata를 보강한다. | -| 6 | `src/codex/catalog.ts:1247` | `disabledModels`, provider별 `selectedModels`, provider 특수 필터를 적용한다. | -| 7 | `src/codex/catalog.ts:1478` | routed `provider/model` entry를 Codex catalog에 병합한다. | -| 8 | `src/codex/catalog.ts:1557` | `$CODEX_HOME/models_cache.json`을 무효화한다. | - -## 정적 seed와 live discovery - -- `liveModels: true`인 provider는 registry의 `models`를 fallback으로 쓴다. -- live fetch가 성공하고 schema가 정상이면 live ID가 권위 있는 목록이다. 설정에만 남은 ID는 무조건 유지하지 않는다. -- fetch 실패나 malformed 응답이면 last-known-good cache, 그다음 static config로 내려간다. -- context와 modality는 사용자/registry hint로 보강할 수 있지만, provider context cap은 live 값보다 크게 올리지 않는다. -- media generation 모델은 coding model picker에서 분리한다. - -## jawcode metadata 생성 계약 - -입력 기본값은 `../jawcode/packages/ai/src/models.json`이고, `JAWCODE_MODELS_JSON`으로 다른 snapshot을 지정할 수 있다 (`scripts/generate-jawcode-metadata.ts:16-18`). 출력은 `src/generated/jawcode-model-metadata.ts`다 (`scripts/generate-jawcode-metadata.ts:19`). - -```bash -bun run generate:jawcode-metadata -# 또는 -JAWCODE_MODELS_JSON=/abs/path/models.json bun run generate:jawcode-metadata -``` - -생성 파일을 직접 고치지 않는다. source model metadata가 틀렸다면 jawcode generator/descriptor를 먼저 고치고 다시 생성한다. OpenCodex만의 routing/capability 예외라면 registry나 catalog normalization이 소유한다. - -## 모델 변경 분류 - -| 변경 | 첫 수정 지점 | -|---|---| -| 기존 provider에 fallback model 추가 | 해당 `PROVIDER_REGISTRY` entry | -| live model metadata shape 추가 | `ProviderModelsApiItem` + `catalogHintsFromModelsApiItem()` | -| context/reasoning/modality 보정 | registry의 model-scoped metadata 또는 jawcode generator source | -| 모델별 wire protocol 변경 | `src/server/adapter-resolve.ts` | -| picker 노출/숨김 | `disabledModels`, `selectedModels`, catalog filter | -| native OpenAI snapshot | `src/codex/data/upstream-models.json`의 owning generation/update 절차 | - -## 완료 기준 - -- catalog focused test가 통과한다. -- `/api/models`와 `/v1/models`가 같은 routed model source를 사용한다. -- `provider/model` slug, context window, input modalities, reasoning ladder가 기대값과 맞는다. -- 생성 파일을 바꿨다면 generator 명령과 diff 요약을 남긴다. -- `bun run typecheck`가 통과한다. - -## 검증 - -```bash -bun test --isolate tests/codex-catalog.test.ts tests/provider-live-models.test.ts -bun run typecheck -git diff --check -``` diff --git a/devlog/_chase/_model/003_auth_routing_flow.md b/devlog/_chase/_model/003_auth_routing_flow.md deleted file mode 100644 index 31d199902..000000000 --- a/devlog/_chase/_model/003_auth_routing_flow.md +++ /dev/null @@ -1,66 +0,0 @@ -# 003 — Auth and routing flow - -Provider를 추가할 때 인증, 모델 선택, wire adapter를 한 덩어리로 보면 잘못된 소유자에 코드를 넣기 쉽다. OpenCodex는 세 단계를 분리한다. - -## 인증 종류 - -`ProviderAuthKind`는 `forward | oauth | key | local` 네 가지다 (`src/providers/registry.ts:9`). 2026-07-17 registry 분포는 forward 1, OAuth 6, key 42, local 3이다. - -| 종류 | 예 | 실제 경로 | -|---|---|---| -| forward | `openai` | Pool(기본)은 메인 포함 계정 풀, Direct는 caller/main 전달만 사용한다. 일반 provider API key가 아니다. | -| OAuth | `xai`, `anthropic`, `kimi`, `kiro`, `google-antigravity`, `cursor` | `src/oauth/index.ts:62`의 controller와 `src/oauth/store.ts`의 계정별 credential store를 사용한다. | -| key | `openrouter`, `zai`, `google`, `anthropic-apikey` | provider config의 key 또는 key pool을 사용한다. | -| local | `ollama`, `vllm`, `lm-studio` | private destination을 명시적으로 허용하며 보통 key가 없다. | - -`OAUTH_PROVIDERS`에는 registry OAuth 6개 외에 ChatGPT auth 관리용 특수 `chatgpt` controller가 있다. registry provider 수와 OAuth controller 수를 같은 값으로 세면 안 된다. - -### OpenAI 권한 경계 - -- bare GPT id는 `openai`이며 `codexAccountMode`가 Pool(기본) 또는 Direct를 선택한다. -- Pool은 메인+추가 계정의 affinity/quota/cooldown 라우팅을 활성화하고 Direct는 풀 상태를 건너뛴다. -- `openai-apikey/`은 API key/key pool만 사용한다. -- 두 credential 경로는 서로 fallback하지 않는다. legacy provider id는 migration 입력에만 남는다. - -## 모델 라우팅 우선순위 - -`routeModel()`의 실제 순서는 다음과 같다 (`src/router.ts:162-222`). - -1. 설정된 provider와 일치하는 명시적 `/`. -2. 활성 provider의 `defaultModel`과 정확히 일치. -3. 알려진 bare-model prefix: Claude, GPT/o-series, Groq 계열. -4. 활성 provider의 static/configured `models`에 포함. -5. `defaultProvider` fallback. -6. 어느 provider도 없으면 오류. - -Slash가 들어간 upstream model ID는 prefix가 실제 configured provider일 때만 namespace로 분리한다. 예를 들어 `anthropic/claude-*`가 OpenRouter 모델 ID라면 `anthropic` provider가 설정되지 않은 경우 그대로 다음 단계로 내려간다. - -## Config backfill - -`routedProviderConfig()`는 registry와 저장 config를 합친다 (`src/router.ts:79-159`). - -- adapter와 built-in base URL/auth kind는 registry 정본을 따른다. -- 사용자 model metadata override는 registry seed보다 우선한다. -- local/self-hosted처럼 override가 허용된 base URL은 비어 있거나 placeholder가 남으면 거부한다. -- private destination은 `assertProviderDestinationAllowed()` 경계를 통과해야 한다. -- OAuth provider가 `allowKeyAuthOverride`를 명시한 경우만 key billing mode를 허용한다. - -## Adapter 선택 - -`resolveAdapter()`가 8개 adapter family를 선택한다 (`src/server/adapter-resolve.ts:27-49`). 모델 하나가 provider 기본 wire와 다를 때는 새 provider를 만들지 않고 `resolveWireProtocolOverride()`에서 좁게 바꾼다. 현재 예는 OpenCode Go의 일부 MiniMax 모델이다 (`src/server/adapter-resolve.ts:13-24`). - -## 429와 credential 동작 - -- 일반 upstream pre-stream retry는 connection reset과 선택된 5xx만 다룬다. 429는 generic transient status가 아니다 (`src/lib/upstream-retry.ts:37-40`). -- API-key pool은 429를 받으면 실패한 key를 cooldown하고 다음 key로 회전한다 (`src/providers/key-failover.ts:1-9`, `src/providers/key-failover.ts:67`). -- OAuth account refresh는 provider/account/generation 단위로 serialize해 회전된 refresh token을 덮어쓰지 않는다 (`src/oauth/store.ts:180-185`, `src/oauth/store.ts:321-322`). -- provider별 quota 조회는 `src/providers/quota.ts`가 맡고, wire retry taxonomy와 섞지 않는다. - -## 인증 변경 체크리스트 - -- [ ] API key, OAuth, forward, local, product token 중 하나로 먼저 분류했다. -- [ ] 기존 adapter가 이미 필요한 Authorization/header shape를 지원하는지 확인했다. -- [ ] OAuth라면 login, refresh, store, account switch, reauth 상태를 함께 확인했다. -- [ ] key pool이라면 failed-key CAS와 cooldown을 보존했다. -- [ ] 로그, chase 문서, test fixture에 raw secret을 넣지 않았다. -- [ ] private/self-hosted base URL이 destination policy를 우회하지 않는다. diff --git a/devlog/_chase/_model/004_patch_index.md b/devlog/_chase/_model/004_patch_index.md deleted file mode 100644 index 09d06c14a..000000000 --- a/devlog/_chase/_model/004_patch_index.md +++ /dev/null @@ -1,67 +0,0 @@ -# 004 — Model/provider patch index - -## 빠른 분류 - -| 변경 종류 | 첫 소유자 | 함께 확인할 곳 | -|---|---|---| -| built-in provider 추가 | `src/providers/registry.ts` | derive, auth, router, adapter, catalog, management API, tests/docs | -| 기존 provider 모델 추가 | registry 또는 live metadata parser | adapter override, catalog tests, picker visibility | -| OAuth/login 변경 | `src/oauth/` | registry `authKind/oauthId`, store concurrency, management API | -| API-key pool/429 변경 | `src/providers/key-failover.ts` | relay retry 경계, tests, quota display | -| wire request/stream 변경 | `src/adapters/` | `src/server/adapter-resolve.ts`, bridge/parser tests | -| Codex picker metadata | `src/codex/catalog.ts` | registry hints, generated jawcode metadata, `/api/models` | -| jawcode model metadata sync | jawcode generator source | `bun run generate:jawcode-metadata`, catalog diff | -| GUI provider preset | registry → derive 경로 | management API, GUI는 파생값 소비 | - -## 새 built-in provider - -1. `src/providers/registry.ts`에 stable id, label, adapter, base URL, auth kind를 추가한다. -2. 기존 adapter로 충분한지 먼저 확인한다. 새 wire shape일 때만 `src/adapters/`와 `resolveAdapter()`를 확장한다. -3. key provider는 `dashboardUrl`과 `deriveKeyLoginMap()` 계약을 확인한다. OAuth면 `src/oauth/index.ts`에 login/refresh controller를 등록한다. -4. bare model prefix가 꼭 필요한 경우에만 `src/router.ts`의 prefix table을 넓힌다. 기본은 명시적 `provider/model`이다. -5. static fallback 모델, live discovery, context/modality/reasoning metadata의 소유자를 결정한다. -6. `/api/providers`, `/api/models`, selected/disabled models, Codex catalog sync를 확인한다. -7. focused provider/adapter/catalog test와 typecheck를 실행하고 사용자 docs를 갱신한다. - -## 기존 provider에 모델 추가 - -1. upstream 모델 ID와 실제 wire protocol을 확인한다. -2. `liveModels: true`면 static allowlist를 만들지 말고 fallback/capability hint만 추가한다. -3. `modelContextWindows`, `modelInputModalities`, `modelReasoningEfforts`, parameter exclusion 목록 중 필요한 값만 해당 registry entry에 둔다. -4. 모델별 wire 차이가 있을 때만 `resolveWireProtocolOverride()`를 수정한다. -5. media-generation 모델인지 vision-input chat 모델인지 구분해 catalog filter를 확인한다. -6. jawcode metadata를 재사용해야 하면 생성 source를 고친 뒤 snapshot을 다시 만든다. - -## 새 config field - -새 필드가 정말 필요한 경우에만 아래 순서를 따른다. - -1. `src/types.ts`의 `ProviderRegistryEntry` 또는 `OcxProviderConfig` 계약. -2. `src/config.ts`의 validation/default/migration. -3. `providerConfigSeed()`와 `enrichProviderFromRegistry()` 파생 경로. -4. router/adapter/catalog의 실제 소비자. -5. management API DTO와 GUI editor. -6. round-trip config test와 backward-compat test. - -## jawcode와의 경계 - -| 질문 | jawcode 소유 | OpenCodex 소유 | -|---|---|---| -| JWC 자체가 provider를 호출하는가 | `packages/ai/src/providers/`, descriptor, auth storage | 해당 없음 | -| Codex 요청을 provider로 proxy하는가 | 참고 구현 | registry, router, adapters, bridge | -| JWC bundled model metadata | generator + `packages/ai/src/models.json` | generated metadata snapshot 소비 | -| Codex App picker 노출 | 해당 없음 | `src/codex/catalog.ts`와 sync/cache | -| OpenCodex OAuth 계정/키 pool | 해당 없음 | `src/oauth/`, `src/providers/api-keys.ts` | - -Provider patch를 가져올 때 `JWC native`, `OCX proxy`, `both`, `docs-only` 중 하나로 먼저 분류한다. jawcode에 provider가 생겼다는 이유만으로 OCX registry를 자동 추가하지 않고, OCX가 route할 수 있다는 이유만으로 jawcode `KnownProvider`를 늘리지 않는다. - -## 최소 검증 묶음 - -```bash -bun test --isolate tests/provider-registry-parity.test.ts tests/provider-live-models.test.ts -bun test --isolate tests/router.test.ts tests/codex-catalog.test.ts -bun run typecheck -git diff --check -``` - -실제 test 파일명은 변경 전 `rg --files tests | rg 'provider|router|catalog|oauth'`로 다시 확인한다. diff --git a/devlog/_chase/_model/005_upstream_delta_backlog.md b/devlog/_chase/_model/005_upstream_delta_backlog.md deleted file mode 100644 index d96b293c2..000000000 --- a/devlog/_chase/_model/005_upstream_delta_backlog.md +++ /dev/null @@ -1,42 +0,0 @@ -# 005 — Upstream model/provider delta backlog - -> Re-triaged: 2026-07-17 against OpenCodex `f779f3d9` (original model-import baseline `0167b415`) and the fingerprinted local jawcode snapshot in `devlog/_fin/260717_jawcode_model_import_audit/001_research_snapshot.md` - -초기 후보는 jawcode의 로컬 미커밋 `struct_har/chase/model/`과 실제 `packages/ai` 변경을 함께 대조했다. 이 문서는 실행 우선순위만 요약하며, 판정 근거는 [006](./006_jawcode_import_matrix.md), 모델명은 [007](./007_model_id_delta.md), 로직은 [008](./008_logic_delta.md)를 정본으로 삼는다. - -## 현재 분류 - -| 항목 | 상태 | 현재 OCX 근거 | 다음 행동 | -|---|---|---|---| -| Fugu/Sakana | `PLAN` direct / `PARTIAL` OpenRouter | 2026-06-22 Sakana가 `https://api.sakana.ai/v1`, Bearer API key, Responses/Chat Completions, `fugu`/`fugu-ultra` 계약을 공개했다. OCX direct preset은 아직 없다. | direct provider는 [`010_fugu_sakana_direct.md`](../../_plan/260717_non_openai_provider_chase/010_fugu_sakana_direct.md)에서 첫 PABCD로 `ADAPT`. OpenRouter는 live discovery를 유지한다. | -| Z.AI weekly-limit taxonomy | `PARTIAL` | `zai` registry/model metadata는 있음 (`src/providers/registry.ts:556`). exact weekly-exhaustion classifier는 없음 | 실제 Z.AI error body를 확보해 provider-scoped 분류가 필요한지 결정한다. | -| Cursor shared version owner | `OPEN` | discovery와 run의 상수가 갈라져 있다. jawcode는 한 owner를 사용한다. | owner 통합은 `ADAPT`; 사용할 버전 값은 인증된 discovery/run probe 후 결정한다. | -| OpenAI/Azure bounded 429 retry | `REJECT` direct port | generic retry가 429를 재시도하지 않고 key pool만 별도로 회전한다. jawcode wrapper는 OpenAI SDK 내부 retry를 막기 위한 코드다. | 현 구조에서는 복사하지 않는다. retry 정책이 바뀌면 quota fixture로 재검토한다. | -| OpenCode Go Kimi effort | `RESEARCH` | OCX는 `kimi-k2.7-code`와 `-highspeed` reasoning을 모두 숨긴다. jawcode는 기본 모델의 일부 effort를 허용·보정한다. | 두 모델을 별도 live probe해 지원표가 확인될 때만 `ADAPT`. | -| Anthropic disabled-thinking omission | `VERIFIED` / `NOOP` | 일반 adapter는 non-`none`일 때만 thinking을 보낸다. web-search sidecar의 explicit disabled는 별도 계약이다. | 일반 경로 이식 없음. sidecar를 함께 바꾸지 않는다. | -| LiteLLM rich/vision metadata 보존 | `PARTIAL` | keyless/self-hosted route와 live discovery는 있음. parser는 context, reasoning boolean, vision boolean을 읽지만 임의 rich metadata를 보존하지 않음 (`src/codex/catalog.ts:990-1124`) | OMP가 보존하는 필드와 Codex가 실제 소비하는 필드의 교집합을 정한다. | -| Codex credential rotation/self-heal | `VERIFIED` 기반, delta 확인 필요 | multi-account affinity, cooldown, generation-safe OAuth persist가 이미 별도 구현됨 | OMP 변경을 auth/account outcome 단위로 비교하고 중복 추상화를 피한다. | -| response terminal/replay 호환 | `PARTIAL` | Responses parser/bridge에 기존 terminal handling이 있으나 OMP `response.done`, Anthropic replay commit과 line-by-line 대조하지 않음 | `src/responses/`, `src/adapters/openai-responses.ts`, Anthropic replay test를 함께 비교한다. | -| Antigravity `gemini-3.1-pro-high` | `RESEARCH` | jawcode는 picker에서 retire하지만 OCX는 `gemini-pro-agent` 호환 alias로 노출·테스트한다. | picker 제거와 inbound alias 보존을 분리해 인증된 가용성 probe 후 결정한다. | -| GPT-5.6 context/cost | `ADAPT` implemented / cost `REJECT` | 세 ID와 3-tier 계약이 구현됐다. Direct/Multi는 372K Codex 계약, API는 1.05M context / 922K max input이며 OpenRouter도 1.05M metadata를 유지한다. OCX는 jawcode cost를 소비하지 않는다. | tier/context/max-input 작업은 종료. billing/가격 consumer가 생기기 전에는 jawcode cost를 복사하지 않는다. | - -## 실행 순서 - -단일 정본은 [`260717_non_openai_provider_chase/000_plan.md`](../../_plan/260717_non_openai_provider_chase/000_plan.md)의 WP1-WP15 표다. 이 backlog는 별도 번호나 우선순위를 유지하지 않는다. - -## 갱신 절차 - -1. jawcode `struct_har/chase/model/005_upstream_model_delta.md`의 source range와 현재 HEAD를 확인한다. -2. candidate commit의 실제 diff를 연다. commit 제목만으로 import하지 않는다. -3. OCX owner를 registry, auth, router, adapter, catalog, docs-only 중 하나로 분류한다. -4. 현재 focused test가 이미 같은 계약을 증명하는지 먼저 찾는다. -5. 구현하거나 `REJECT` 근거를 남기고 이 표의 상태를 갱신한다. - -## 재검증 검색 - -```bash -rg -n -i "fugu|sakana|fish_|weekly limit|client-version|litellm|response.done|thinking.*disabled" src tests -rg -n "isTransientUpstreamStatus|rotateKeyOn429|CURSOR_DISCOVERY_CLIENT_VERSION|CURSOR_CLIENT_VERSION" src tests -git -C ../jawcode status --short struct_har/chase/model -git -C ../jawcode log -1 --oneline -``` diff --git a/devlog/_chase/_model/006_jawcode_import_matrix.md b/devlog/_chase/_model/006_jawcode_import_matrix.md deleted file mode 100644 index b7b63a262..000000000 --- a/devlog/_chase/_model/006_jawcode_import_matrix.md +++ /dev/null @@ -1,47 +0,0 @@ -# 006 — jawcode import decision matrix - -> Snapshot: 2026-07-17 local jawcode working tree. It is uncommitted and fingerprinted in `devlog/_fin/260717_jawcode_model_import_audit/001_research_snapshot.md`. - -이 표는 “jawcode에 있으니 복사한다”가 아니라, 같은 문제를 OCX가 실제로 갖는지와 어느 owner가 책임져야 하는지를 결정한다. `IMPORT`/`ADAPT`도 코드 작업 승인이 아니라 표의 gate를 통과했을 때의 방향이다. - -## 판정표 - -| 후보 | 근거 | OCX 현재 상태 | 결정 | 구현 전 gate | -|---|---|---|---|---| -| Cursor client-version 공용 owner | `local-source` | discovery `cli-2026.02.13-41ac335`, run `cli-2026.07.08-0c04a8a`로 drift | `ADAPT` | 두 경로가 같은 owner를 import하는 focused test | -| Cursor에 사용할 정확한 version | `live-unverified` | 두 값 중 어느 것이 양 endpoint의 현재 계약인지 미확인 | `RESEARCH` | 인증된 discovery와 run이 같은 값으로 성공하는 probe. jawcode의 02 값을 그대로 채택하지 않음 | -| OpenAI/Azure bounded 429 wrapper | `local-source` | OCX generic retry는 429를 재시도하지 않으며, key pool만 429에서 회전 | `NOOP` / direct port `REJECT` | 향후 SDK retry를 도입하거나 429 지연이 재현될 때만 provider-scoped fixture로 재검토 | -| GPT-5.6 Luna/Sol/Terra model IDs | `local-source` | 세 ID 모두 native, OpenAI API-key, OpenRouter에 이미 존재 | `NOOP` | 없음. 신규 모델 추가 대상이 아님 | -| GPT-5.6 context window | `local-source`, OCX contract implemented | jawcode 최종 373K와 달리 OCX Pool/Direct는 같은 bare 372K 그룹, API-key는 1.05M context / 922K max input, OpenRouter는 1.05M | `ADAPT` implemented | 없음. account-mode/API ownership과 compact cap이 테스트로 고정됨 | -| GPT-5.6 cost rows | `local-source` | OCX model catalog/runtime에 jawcode cost consumer가 없음 | `REJECT` current scope | billing/가격 UI owner가 생기기 전에는 생성물에 필드를 더하지 않음 | -| Antigravity picker에서 `gemini-3.1-pro-high` retire | `local-source`, `live-unverified` | OCX는 아직 picker seed에 노출 | `RESEARCH` | 인증된 available-models 또는 inference 실패 증거 | -| Antigravity inbound/wire alias 보존 | `local-source` | `gemini-3.1-pro-high -> gemini-pro-agent`를 테스트로 고정 | `NOOP` now | picker를 retire하더라도 기존 config 호환 alias는 별도 deprecation 없이 삭제하지 않음 | -| OpenCode Go `kimi-k2.7-code` effort map | `local-source`, `live-unverified` | OCX는 reasoning을 완전히 숨김; jawcode는 `xhigh|max -> high` 보정 | `RESEARCH`, then `ADAPT` | low/medium/high와 tool-choice 조합 live matrix | -| OpenCode Go `kimi-k2.7-code-highspeed` effort map | `live-unverified` | OCX는 기본 모델과 같이 reasoning을 숨기지만 jawcode 근거는 없음 | `RESEARCH` | highspeed 자체 probe. 기본 모델 결과를 전이하지 않음 | -| Anthropic 일반 요청의 disabled-thinking omission | `local-source` | OCX는 reasoning이 non-`none`일 때만 thinking을 보냄 | `NOOP` | 기존 `anthropic-reasoning` test 유지 | -| Anthropic web-search sidecar explicit disabled | OCX local source | 별도 sidecar가 의도적으로 `{type:"disabled"}` 전송 | `REJECT` coupled change | 해당 endpoint 계약의 독립 증거 없이는 일반 adapter 변경과 묶지 않음 | -| Anthropic OAuth organization identity | `chase-only` | OCX token parser는 account UUID/email만 저장 | `RESEARCH` | 실제 token response schema와 multi-org collision 재현 | -| Anthropic tool argument/stream index hardening | `local-source` | OCX adapter/parser 구조가 jawcode SDK event loop와 다름 | `RESEARCH`, then `ADAPT` | malformed tool JSON 및 interleaved block fixture를 OCX parser에서 먼저 재현 | -| Google tool-argument JSON sanitization | `local-source` | provider shape가 다르고 같은 malformed payload 재현 없음 | `RESEARCH` | OCX Google adapter의 failing fixture | -| Gemini CLI version header bump | `local-source` | OCX는 jawcode Gemini CLI transport와 같은 owner가 아님 | `REJECT` direct port | 동일 endpoint/header를 소유하는 경로가 확인될 때만 재검토 | -| LiteLLM rich metadata | `chase-only` + OCX local source | OCX는 context, reasoning boolean, vision boolean만 소비 | `ADAPT` consumer-first | Codex catalog가 실제 소비할 필드와 fixture를 먼저 추가. 임의 metadata passthrough 금지 | -| jawcode generated `maxTokens`, `reasoning`, `wireModelId` | `local-source` | 생성물에는 있으나 OCX catalog application은 소비하지 않음 | `RESEARCH` | 각 필드의 runtime consumer와 precedence contract 정의 | -| OpenRouter source-only 17 IDs | `local-source` | metadata regeneration만으로 OpenRouter row를 append하지 않음 | `NOOP` refresh-only / `RESEARCH` exposure | live `/models` 결과 또는 의도적 static seed 결정. [007](./007_model_id_delta.md) 참조 | -| xAI `grok-4.5` | `local-source` | OCX xAI registry에 이미 모델·reasoning·context가 있음 | `NOOP` | OpenRouter namespaced row와 direct xAI row를 혼동하지 않음 | -| Z.AI weekly-limit classifier | `chase-only`, `live-unverified` | 정확한 body classifier 없음 | `RESEARCH` | 실제 provider error body와 현재 key/failover 결과 fixture | -| invalid-prompt/refusal circuit breaker | `chase-only` | 400은 generic transient retry 대상이 아니며 refusal parsing도 존재 | `NOOP` for retry / `RESEARCH` normalization | 반복 retry가 실제 재현되거나 provider별 safety stop 통합 요구가 생길 때 | -| model hub, floating selection, custom role models | `chase-only` | jawcode agent UI semantics; OCX는 proxy catalog/management API 제품 | `REJECT` direct port | OCX GUI에 같은 사용자 요구가 정의될 때 별도 제품 계획 | -| prompt cap / agent dispatch guard | `chase-only` | jawcode agent prompt assembly와 OCX proxy context contract가 다름 | `REJECT` direct port | OCX-owned overflow 재현이 있을 때 catalog/context owner에서 설계 | -| Fugu/Sakana standalone provider/login | `official-source` + user direction | Sakana가 direct endpoint, Bearer auth, Responses/Chat, 두 모델을 공개했고 사용자가 first-class provider를 요청함 | `ADAPT` | registry/Responses fixture + authenticated smoke; [`010`](../../_plan/260717_non_openai_provider_chase/010_fugu_sakana_direct.md) | -| OpenRouter `sakana/fugu-ultra` | `local-source` | jawcode metadata row는 있으나 OCX는 OpenRouter row를 metadata로 append하지 않음 | `NOOP` static import / `RESEARCH` visibility | OpenRouter live discovery 결과를 권위로 사용 | - -## 다음 코드 작업 후보 - -첫 구현 후보는 공식 direct 계약이 확인된 **Sakana Fugu provider**다. 그다음 명확한 hardening 후보는 **Cursor version owner 통합**이며, 값 자체는 여전히 `RESEARCH`다. - -1. discovery와 run을 같은 version parameter로 probe한다. -2. 성공한 값을 공용 owner로 추출한다. -3. 두 caller와 override behavior를 focused test로 고정한다. -4. 실패 시 version을 추측해 통일하지 않고 현재 분리를 유지한 채 증거를 기록한다. - -나머지는 [`260717_non_openai_provider_chase`](../../_plan/260717_non_openai_provider_chase/000_plan.md)의 decade 문서와 live-fixture gate 없이 코드 phase로 승격하지 않는다. diff --git a/devlog/_chase/_model/007_model_id_delta.md b/devlog/_chase/_model/007_model_id_delta.md deleted file mode 100644 index 8cc0d1e4a..000000000 --- a/devlog/_chase/_model/007_model_id_delta.md +++ /dev/null @@ -1,125 +0,0 @@ -# 007 — Provider and model ID delta - -> Comparison base: OpenCodex `0167b415` and the fingerprinted 2026-07-17 local jawcode snapshot. - -## Provider namespace - -jawcode `models.json`에는 48개 top-level provider key가 있고, OCX `PROVIDER_REGISTRY`에는 53개 provider ID가 있다. 숫자나 문자열 차이는 곧 missing provider가 아니다. - -### jawcode generated catalog에만 있는 ID - -`alibaba-coding-plan`, `amazon-bedrock`, `deepinfra`, `google-gemini-cli`, `minimax-code`, `minimax-code-cn`, `openai-codex`, `opencode` - -### OCX registry에만 있는 ID - -`alibaba`, `anthropic-apikey`, `kimi`, `lm-studio`, `mimo-free`, `neuralwatt`, `ollama`, `openai-apikey`, `opencode-free`, `parallel`, `umans`, `vllm` - -### 의미상 대응을 먼저 봐야 하는 이름 - -| OCX | jawcode 쪽 비교 대상 | 주의점 | -|---|---|---| -| `alibaba` | `alibaba-coding-plan` | ID가 아니라 endpoint/auth/plan 계약으로 비교 | -| `openai` | `openai-codex` | forwarded Codex auth; Pool(기본)/Direct는 provider option으로 선택 | -| `openai-apikey` | `openai` | API key Responses transport가 가까움 | -| `anthropic`, `anthropic-apikey` | `anthropic` | OCX는 auth mode를 provider ID로 분리 | -| `kimi`, `kimi-code`, `moonshot` | `kimi-code`, `moonshot` | OAuth/code endpoint/API endpoint를 분리해 비교 | -| `opencode-free`, `opencode-go`, `opencode-zen` | `opencode`, `opencode-go`, `opencode-zen` | 무료 catalog, Go plan, Zen endpoint를 이름만으로 합치지 않음 | - -## jawcode metadata bridge의 실제 범위 - -OCX는 `anthropic`, `google`, `minimax`, `moonshot`, `opencode-go`, `openrouter`, `xai`의 7개 jawcode bundle을 매핑한다. `src/codex/catalog.ts:304`의 append allowlist는 `opencode-go` 하나뿐이다. - -- `opencode-go`: jawcode에만 있는 row를 OCX routed catalog에 추가할 수 있다. -- 나머지 6개: 이미 live/static discovery에 존재하는 row의 context/input만 보강한다. -- generated metadata의 `maxTokens`, `reasoning`, `wireModelId`는 현재 catalog에 적용되지 않는다. - -따라서 `bun run generate:jawcode-metadata`로 파일이 바뀌었다고 새 OpenRouter 모델이 자동 노출되는 것은 아니다. - -## OpenRouter source-only 17 IDs - -jawcode source `models.json`에는 있으나 현재 OCX generated snapshot에는 없는 ID다. - -| 분류 | ID | OCX 효과 | -|---|---|---| -| 이미 OCX static seed | `openai/gpt-5.6-luna`, `openai/gpt-5.6-sol`, `openai/gpt-5.6-terra` | 이름 추가 불필요. source metadata refresh는 기존 row 보강만 가능 | -| OpenRouter source-only tier variant | `openai/gpt-5.6-luna-pro`, `openai/gpt-5.6-sol-pro`, `openai/gpt-5.6-terra-pro` | OpenRouter에는 metadata만으로 append되지 않음. OCX `openai-apikey`에는 별도 static virtual rows가 구현됨 | -| xAI namespaced/alias | `x-ai/grok-4.5`, `~x-ai/grok-latest` | direct OCX `xai/grok-4.5`와 별개. OpenRouter live result로만 노출 판단 | -| Aion | `aion-labs/aion-2.0`, `aion-labs/aion-3.0`, `aion-labs/aion-3.0-mini` | discovery-only candidate | -| Nex AGI | `nex-agi/nex-n2-mini`, `nex-agi/nex-n2-pro` | discovery-only candidate | -| Poolside | `poolside/laguna-xs-2.1`, `poolside/laguna-xs-2.1:free` | discovery-only candidate | -| Tencent | `tencent/hy3`, `tencent/hy3:free` | discovery-only candidate; 기존 `hy3-preview`와 동일시하지 않음 | - -`sakana/fugu-ultra`는 이미 현재 OCX generated jawcode snapshot에도 존재하므로 이 17개에는 포함되지 않는다. 다만 OpenRouter는 metadata append 대상이 아니어서, 그 row가 picker에 나타나려면 live/static discovery가 먼저 모델을 제공해야 한다. - -## OpenRouter `maxTokens` delta 11개 - -아래 값은 `source -> current generated snapshot` 비교다. 현재 OCX catalog가 generated `maxTokens`를 소비하지 않으므로 **즉시 런타임 변화가 아닌 contract gap**이다. - -| ID | jawcode source | OCX generated snapshot | -|---|---:|---:| -| `minimax/minimax-m2` | 131,072 | 196,608 | -| `minimax/minimax-m2.1` | 131,072 | 196,608 | -| `minimax/minimax-m3` | 131,072 | 512,000 | -| `moonshotai/kimi-k2-thinking` | 100,352 | 262,144 | -| `moonshotai/kimi-k2.7-code` | 262,144 | 16,384 | -| `nvidia/nemotron-3-super-120b-a12b` | 262,144 | 16,384 | -| `openai/gpt-oss-120b` | 65,536 | 131,072 | -| `qwen/qwen3-235b-a22b-thinking-2507` | 8,888 | 262,144 | -| `qwen/qwen3-30b-a3b-thinking-2507` | 32,768 | 131,072 | -| `z-ai/glm-5.1` | 128,000 | 131,072 | -| `z-ai/glm-5.2` | 131,072 | 32,768 | - -이 값을 사용하려면 먼저 `CatalogModel`/Codex consumer에서 max output의 의미와 precedence를 정의해야 한다. context window와 output cap을 섞어 적용하면 안 된다. - -## GPT-5.6 Luna/Sol/Terra - -### 이름 - -세 모델 이름은 양쪽에 이미 있다: `gpt-5.6-luna`, `gpt-5.6-sol`, `gpt-5.6-terra`. 신규 ID 이식 대상이 아니다. - -### jawcode 단계별 값 - -| 단계/transport | context | max output | -|---|---:|---:| -| generated OpenAI API row, policy 적용 전 | 1,050,000 | 128,000 | -| generated `openai-codex` row, policy 적용 전 | 373,000 | 128,000 | -| `applyGpt56ContextWindow` 적용 후 | 373,000 | 기존 max output 유지 | - -### OCX 현재 값 - -| OCX route | context | -|---|---:| -| native Codex catalog | 372,000 | -| `openai-apikey` base + Pro | 1,050,000 context / 922,000 max input | -| `openrouter/openai/*` | 1,050,000 | -| Cursor tier rows | live/registry owner, 현재 test seed 1,000,000 | - -Direct/Multi의 372K Codex 계약과 API/OpenRouter의 1.05M metadata는 transport별로 고정됐다. -API Pro id는 public selected identity를 보존하고 wire에서 base model + `reasoning.mode: "pro"`로 변환된다. - -### jawcode cost metadata - -| ID | input | output | cache read | cache write | -|---|---:|---:|---:|---:| -| Luna | 1 | 6 | 0.1 | 1.25 | -| Sol | 5 | 30 | 0.5 | 6.25 | -| Terra | 2.5 | 15 | 0.25 | 3.125 | - -OCX에는 이 jawcode cost shape의 runtime/catalog consumer가 없으므로 현재 범위에서는 가져오지 않는다. - -## Anthropic source/generated delta - -- ID 수는 25개로 같다. -- OCX generator는 `claude-sonnet-4-6`과 `[1m]`의 context를 의도적으로 200K로 override한다. -- Sonnet 4.5에는 같은 override가 없으므로 source refresh 전에 1M 계약을 별도 검증해야 한다. - -| ID | jawcode source context/output | OCX snapshot context/output | 해석 | -|---|---:|---:|---| -| `claude-sonnet-4-5` | 1,000,000 / 64,000 | 200,000 / 64,000 | override 없음; refresh 전 live proof 필요 | -| `claude-sonnet-4-5-20250929` | 1,000,000 / 64,000 | 200,000 / 64,000 | override 없음; refresh 전 live proof 필요 | -| `claude-sonnet-4-6` | 1,000,000 / 128,000 | 200,000 / 64,000 | context override는 의도적; output은 unconsumed delta | -| `claude-sonnet-4-6[1m]` | 1,000,000 / 64,000 | 200,000 / 64,000 | 이름과 snapshot context가 불일치하므로 별도 계약 확인 | - -## xAI delta - -jawcode source의 `grok-4.5`는 current generated xAI bundle에는 아직 없지만, OCX `xai` registry가 이미 `grok-4.5`와 500K context, low/medium/high reasoning을 명시한다. generated refresh는 현재 direct xAI 노출에 필수 조건이 아니다. diff --git a/devlog/_chase/_model/008_logic_delta.md b/devlog/_chase/_model/008_logic_delta.md deleted file mode 100644 index 07a75e824..000000000 --- a/devlog/_chase/_model/008_logic_delta.md +++ /dev/null @@ -1,139 +0,0 @@ -# 008 — Model/provider logic delta - -> Evidence boundary: jawcode paths below refer to the fingerprinted local uncommitted snapshot, not a merged upstream release. - -## 1. Cursor client version - -### jawcode - -`packages/ai/src/providers/cursor/client-version.ts` owns one `CURSOR_CLIENT_VERSION = "cli-2026.02.13-41ac335"`, used by model discovery and run transport. Its invariant is valuable: one backend integration should not silently drift because only one caller was updated. - -### OCX - -- discovery: `src/adapters/cursor/live-models.ts:20` — `cli-2026.02.13-41ac335` -- run: `src/adapters/cursor/live-transport.ts:56` — `cli-2026.07.08-0c04a8a` - -### Decision - -- Shared owner: `ADAPT`. -- Exact value: `RESEARCH`. - -The centralization can be imported as an invariant, but choosing the older jawcode value would be an unsupported behavior change. Probe both endpoints with one value first. - -## 2. OpenAI/Azure bounded 429 - -### jawcode - -`openai-bounded-rate-limits.ts` wraps the OpenAI SDK fetch. A 429 is marked `x-should-retry:false` when Retry-After exceeds 60 seconds or the body matches permanent quota phrases such as `out_of_credits`, `insufficient_quota`, or monthly usage limit. - -### OCX - -- `src/lib/upstream-retry.ts:38` retries selected 500/502/503/504/520/521/522 statuses only. -- 429 is not a generic transient retry. -- `src/providers/key-failover.ts:73` can rotate an API-key pool on 429; with no alternative it surfaces the failure. - -### Decision - -Direct port is `REJECT`; outcome is currently `NOOP`. The jawcode wrapper solves SDK-internal retry behavior that OCX does not use. If OCX later adopts an SDK or begins retrying 429, bring over the **classification tests**, not necessarily the wrapper. - -## 3. GPT-5.6 policy - -jawcode adds curated Luna/Sol/Terra descriptors and constructs 1.05M OpenAI API rows plus 373K Codex rows, then `applyGpt56ContextWindow` resets both to 373K. OCX now has explicit route contracts: Pool(default)/Direct account modes share one bare `openai` 372K group; `openai-apikey` uses 1.05M context / 922K max input and static Pro virtual ids. - -Decision: account-mode/API identity, context, max input, and Pro wire rewrite are `ADAPT` and implemented. -Cost remains `REJECT` in current scope. See [007](./007_model_id_delta.md). - -## 4. Antigravity retired model - -jawcode `model-manager.ts` filters `google-antigravity/gemini-3.1-pro-high` from bundled, cached, and dynamic model lists. OCX still maps it to `gemini-pro-agent` in `src/providers/antigravity-models.ts:21` and tests both picker presence and wire mapping. - -Two contracts must not be collapsed: - -1. **Picker exposure:** `RESEARCH`; remove only after authenticated availability/inference proof. -2. **Inbound compatibility alias:** `NOOP` now; preserve existing saved config even if the picker row is retired later. - -## 5. OpenCode Go Kimi effort - -jawcode's local compatibility logic records: - -- `kimi-k2.5`: `minimal -> low` -- `kimi-k2.7-code`: `xhigh -> high`, `max -> high` -- Kimi reasoning is disabled for forced tool-choice cases. - -OCX currently sets an empty effort list and `noReasoningModels` for both `kimi-k2.7-code` and `kimi-k2.7-code-highspeed`, while preserving reasoning content for replay. - -Decision: keep the conservative behavior until live probes establish a matrix. The base model could become `ADAPT`; highspeed remains independent `RESEARCH` because jawcode evidence does not cover it. - -## 6. Anthropic thinking and stream hardening - -### Disabled thinking - -OCX `src/adapters/anthropic.ts:629` only sends thinking for a string effort other than `none`. This already matches jawcode's general-path omission, so it is `NOOP`. - -`tests/web-search-anthropic.test.ts:196` proves the web-search sidecar intentionally sends `{type:"disabled"}`. This is a separate endpoint contract and is not evidence of a bug in the general adapter. - -### Tool argument and block-index changes - -jawcode also sanitizes tool argument JSON and tracks stream block indices. Those are SDK/event-shape-specific. Before adapting them, reproduce malformed JSON or interleaved block loss in OCX's own adapter/parser fixtures. Decision: `RESEARCH`, then architecture-specific `ADAPT` only if reachable. - -## 7. Google request compatibility - -jawcode changes Gemini CLI version headers and sanitizes JSON strings in tool arguments. OCX's Google AI Studio/Vertex/Antigravity ownership does not automatically share the Gemini CLI fingerprint. - -- Header version: direct port `REJECT` unless OCX sends the same endpoint/header contract. -- Tool JSON sanitizer: `RESEARCH`; require a failing OCX fixture before adding normalization. - -## 8. Generated metadata consumption - -The generator writes six model fields: provider, id, context window, max tokens, input modalities, reasoning, and optional wire model ID. OCX catalog application at `src/codex/catalog.ts:517` consumes only context window and input modalities. `CatalogModel` has no max-output or wire-ID field, and missing rows are appended only for `opencode-go`. - -Consequences: - -- Refreshing the generated file can update context/input on an already discovered row. -- It cannot by itself expose new OpenRouter/xAI/Anthropic rows. -- `maxTokens` diffs have no current runtime effect. -- Adding consumers requires an explicit precedence contract against live metadata and registry hints. - -Decision: generator refresh and consumer expansion are separate tasks. The former is mechanical; the latter is `RESEARCH`/`ADAPT` with focused tests. - -## 9. Anthropic organization identity - -jawcode chase cites upstream organization-scoped auth behavior, but the inspected local source does not establish that full change. OCX `src/oauth/anthropic.ts:30-67` parses account UUID/email and the credential store uses those identities. - -Decision: `RESEARCH`. Obtain the actual token response schema and a multi-org collision scenario before changing storage identity. This is `chase-only`, not a confirmed local implementation delta. - -## 10. Safety, invalid prompt, and terminality - -Chase items describe invalid-prompt breakers, refusal/safety stops, and fallback boundaries. In OCX, 400 errors are not generic retry candidates and the Responses parser already understands refusals. Therefore a retry circuit-breaker port is `NOOP`; cross-provider safety normalization remains `RESEARCH` only if a concrete terminality bug appears. - -## 11. LiteLLM metadata - -OCX live model parsing accepts ID/owner, context length variants, and limited reasoning/vision capabilities. It intentionally drops arbitrary rich metadata. - -Decision: `ADAPT` consumer-first. Add only a field that a Codex-visible catalog or request adapter consumes, together with precedence and fixture tests. Do not add a lossless passthrough bag merely to mirror OMP. - -## 12. Sakana Fugu direct provider - -2026-06-22 이후 전제가 바뀌었다. Sakana 공식 setup은 direct base URL `https://api.sakana.ai/v1`, Bearer API key, Responses/Chat Completions, `fugu`와 `fugu-ultra`, `high/xhigh` effort와 `max -> xhigh` 호환을 공개한다. - -Decision: standalone direct provider는 `ADAPT`. 기존 `openai-responses` keyed path를 재사용하며, OpenRouter `sakana/fugu-ultra` 노출은 계속 live discovery가 소유한다. 실행 계약은 [`010_fugu_sakana_direct.md`](../../_plan/260717_non_openai_provider_chase/010_fugu_sakana_direct.md)다. - -## 13. Product-boundary rejects - -The following chase ideas belong to jawcode/OMP agent products and are not direct OCX proxy imports: - -- floating model selection and model hub UX; -- custom role models and task-agent resolution; -- agent prompt caps and dispatch preprocessing; - -They are `REJECT` for direct port, not claims that the ideas are intrinsically invalid. A future OCX product requirement must start a separate owner-first design. - -## 14. Recommended implementation order - -1. Add direct Sakana Fugu through the existing keyed Responses owner. -2. Probe and centralize Cursor client version. -3. Probe Antigravity picker retirement while preserving inbound alias compatibility. -4. Probe OpenCode Go Kimi base/highspeed effort support separately. -5. Add Z.AI and Anthropic changes only from provider-specific fixtures. -6. Decide consumer-backed generated metadata fields and precedence. -7. Add registry-compatible providers before workspace/auth/signing providers; use the full map in [`000_plan.md`](../../_plan/260717_non_openai_provider_chase/000_plan.md). diff --git a/devlog/_chase/_model/README.md b/devlog/_chase/_model/README.md deleted file mode 100644 index ce9924f9e..000000000 --- a/devlog/_chase/_model/README.md +++ /dev/null @@ -1,62 +0,0 @@ -# chase/_model — OpenCodex 모델·provider 기준점 - -이 폴더는 OpenCodex가 실제로 소유하는 모델/provider 구조와 변경 절차를 모아 둔 chase 기준점이다. jawcode의 `struct_har/chase/model/`에서 문서 분리 방식을 가져왔지만, 내용은 OpenCodex의 프록시·라우팅·Codex catalog 소유권에 맞춰 다시 작성했다. - -현재 OpenAI/xAI 외 provider 구현 로드맵은 [`devlog/_plan/260717_non_openai_provider_chase/000_plan.md`](../../_plan/260717_non_openai_provider_chase/000_plan.md)다. 이 로드맵은 direct Sakana를 첫 work-phase로 두고, 기존 hardening 뒤에 OpenAI-compatible preset, workspace auth, native AWS 순서로 진행한다. - -## 읽는 순서 - -1. [001_provider_inventory.md](./001_provider_inventory.md) — 현재 provider 수, adapter/auth 분포, 정본 경로. -2. [002_catalog_contract.md](./002_catalog_contract.md) — registry, live `/models`, jawcode metadata, Codex catalog가 합쳐지는 순서. -3. [003_auth_routing_flow.md](./003_auth_routing_flow.md) — 인증 종류, 모델 라우팅 우선순위, wire adapter 선택. -4. [004_patch_index.md](./004_patch_index.md) — 새 provider/모델/인증/wire 변경 시 실제 수정 지점. -5. [005_upstream_delta_backlog.md](./005_upstream_delta_backlog.md) — 실행 순서만 남긴 요약 backlog. -6. [006_jawcode_import_matrix.md](./006_jawcode_import_matrix.md) — jawcode 후보별 가져오기·조정·제외 판정과 구현 gate. -7. [007_model_id_delta.md](./007_model_id_delta.md) — provider namespace, 정확한 모델 ID, context/output metadata 차이. -8. [008_logic_delta.md](./008_logic_delta.md) — Cursor, retry, reasoning, auth, metadata bridge의 실제 로직 대조. - -## OpenCodex의 주요 소유자 - -| 표면 | 정본 | -|---|---| -| built-in provider와 모델 capability seed | `src/providers/registry.ts:9`, `src/providers/registry.ts:221` | -| registry → init/GUI/key-login/OAuth 파생 | `src/providers/derive.ts:59`, `src/providers/derive.ts:101`, `src/providers/derive.ts:151` | -| 저장 config 계약 | `src/types.ts:348`, `src/types.ts:559` | -| `provider/model` 및 bare model 라우팅 | `src/router.ts:162` | -| wire adapter 선택과 모델별 override | `src/server/adapter-resolve.ts:13`, `src/server/adapter-resolve.ts:27` | -| live discovery와 Codex-visible catalog | `src/codex/catalog.ts:1126`, `src/codex/catalog.ts:1270`, `src/codex/catalog.ts:1478` | -| jawcode metadata snapshot 생성 | `scripts/generate-jawcode-metadata.ts:16`, `package.json:42` | -| provider 관리와 model picker API | `src/server/management-api.ts:400`, `src/server/management-api.ts:480` | - -## 운영 규칙 - -- built-in provider의 정본은 `PROVIDER_REGISTRY`다. GUI preset, key-login 목록, OAuth 기본 config에 같은 값을 따로 복사하지 않는다. -- `src/generated/jawcode-model-metadata.ts`는 직접 수정하지 않는다. jawcode `packages/ai/src/models.json`을 입력으로 `bun run generate:jawcode-metadata`를 실행한다. -- live `/models`가 있는 provider는 live 결과를 모델 ID의 권위 있는 목록으로 취급하고, registry는 fallback과 capability hint를 맡는다. -- provider 추가와 모델 추가를 구분한다. 기존 adapter로 호출 가능한 새 모델 때문에 새 adapter나 새 auth flow를 만들지 않는다. -- jawcode native provider와 OpenCodex proxy provider는 별도 결정이다. 이름이 같아도 transport, auth, retry, catalog 소유권은 자동으로 공유되지 않는다. - -## 출처와 신선도 - -초기 구조는 2026-07-17에 로컬 jawcode의 미커밋 `struct_har/chase/model/` 7개 문서를 읽고 만들었다. jawcode 문서는 참고 근거일 뿐 OpenCodex 정본이 아니다. provider 수, 모델 ID, upstream commit은 바뀔 수 있으므로 변경 작업을 시작할 때 이 폴더의 검증 명령을 다시 실행한다. - -상태 표기는 다음 네 가지로 통일한다. - -| 상태 | 뜻 | -|---|---| -| `VERIFIED` | 현재 소스와 focused test로 확인됨 | -| `PARTIAL` | 일부 경로는 있으나 upstream 계약 전체는 확인되지 않음 | -| `OPEN` | 현재 OCX 소유 경로에 구현이 없음 | -| `REJECT` | OCX 경계 밖이거나 의도적으로 가져오지 않음 | - -구현 후보를 대조할 때는 위의 현재 상태와 별도로 다음 **결정**을 사용한다. - -| 결정 | 뜻 | -|---|---| -| `IMPORT` | 같은 계약을 OCX owner에 구현할 가치가 확인됨. 표에 적힌 gate를 통과한 뒤 작업한다. | -| `ADAPT` | 목적은 유효하지만 jawcode 코드를 그대로 복사하지 않고 OCX 구조에 맞춘다. | -| `NOOP` | OCX가 이미 같은 결과를 내거나 구조상 해당 문제가 발생하지 않는다. | -| `REJECT` | 현재 OCX 제품/transport 경계에는 넣지 않는다. | -| `RESEARCH` | chase-only 또는 live-unverified라 구현 결정을 내릴 증거가 부족하다. | - -근거 종류는 `local-source`, `chase-only`, `live-unverified`로 표시한다. `chase-only`와 `live-unverified`는 이 문서만으로 `IMPORT`나 `ADAPT` 승인을 만들 수 없다. diff --git a/devlog/_fin/100_codex-native-parity/00_overview.md b/devlog/_fin/100_codex-native-parity/00_overview.md deleted file mode 100644 index 9a9f3bf5e..000000000 --- a/devlog/_fin/100_codex-native-parity/00_overview.md +++ /dev/null @@ -1,128 +0,0 @@ -# 100 — Codex Native Parity Plan - -Date: 2026-06-20 - -## Goal - -Phase 90 proved that opencodex can make Codex CLI/App treat the proxy as a native Codex -provider by injecting the right provider config and model catalog. Phase 100 is the next -parity pass: audit every Codex-native catalog/runtime selector that opencodex currently -inherits from a template, then decide which fields should be preserved, rewritten, or stripped -for routed non-OpenAI models. - -This started as planning-only research. Implementation now proceeds in independent PABCD slices, -with each slice recorded in this devlog and committed separately. - -## Research Split - -The investigation was split lexicographically by decade: - -| File | Topic | -| --- | --- | -| `10_search-and-tool-discovery.md` | `supports_search_tool`, `web_search_tool_type`, hosted vs deferred search fallback | -| `11_search-defaults-and-inherited-state.md` | Native Codex search defaults and current opencodex inherited catalog state | -| `12_catalog-normalization-implementation-plan.md` | Phase 100.1 implementation plan for routed catalog selector normalization | -| `13_catalog-normalization-completion.md` | Phase 100.1 completion evidence and verification record | -| `20_personality-model-messages.md` | `model_messages`, `supports_personality`, prompt identity/personality support | -| `21_model-messages-strip-first.md` | Follow-up decision: strip `model_messages` from routed models first | -| `30_tool-mode-multi-agent.md` | `tool_mode`, `multi_agent_version`, code-mode and subagent selector behavior | -| `40_responses-lite-websockets.md` | `use_responses_lite`, `supports_websockets`, HTTP/SSE vs WS viability | -| `41_responses-lite-policy.md` | Follow-up policy for `use_responses_lite` inheritance | -| `50_streaming-thinking-context.md` | intermediate text, thinking blocks, token usage, context window metadata | -| `51_raw-reasoning-bridge.md` | Raw `response.reasoning_text.delta` evidence and required bridge shape | -| `60_jawcode-metadata-snapshot.md` | jawcode metadata reuse plan for context/capability defaults | -| `90_phase-plan.md` | implementation order and verification gates | - -## Primary Finding - -opencodex intentionally clones a native Codex model template so Codex's strict catalog parser and -App/TUI picker recognize routed model entries. That was necessary for Phase 90, but it means routed -models can also inherit native-only runtime selectors: - -- hosted/deferred search capabilities; -- `model_messages` identity/personality templates; -- code-mode tool exposure; -- multi-agent V1/V2/disabled selection; -- responses-lite behavior; -- context-window and token accounting defaults; -- websocket capability hints. - -For routed models, every inherited field should be considered unsafe until opencodex either proves -it is provider-neutral or normalizes it deliberately. - -## Recommended Principle - -Use native Codex metadata only as a structural template. For routed entries: - -1. Preserve fields that are purely parser/picker compatibility. -2. Rewrite fields that mention model identity, provider identity, reasoning semantics, context size, - tool exposure, or runtime transport. -3. Strip fields that advertise native OpenAI-only capability unless opencodex implements an - equivalent bridge. - -## Highest Priority Fixes - -1. Strip `model_messages` from routed non-OpenAI models first so GPT/Codex/OpenAI identity does not - leak through `instructions_template`. -2. Normalize `tool_mode`, `multi_agent_version`, and `use_responses_lite` instead of inheriting them - silently from the native template. -3. Do not implement websocket support in Phase 100. Keep `supports_websockets` absent/false for - routed providers because the routed path is upstream HTTP/SSE, and a websocket first hop cannot - make non-websocket upstream models websocket-capable. -4. Add provider/model-specific context-window metadata instead of inheriting native GPT limits. -5. Extend usage/reasoning streaming parity so Codex receives cached/reasoning token details and the - correct reasoning channel shape. - -## Websocket Decision Update - -Earlier Phase 100 docs treated websocket as a possible later spike. That is now explicitly out of -scope for Phase 100. - -```text -Decision: no 100.6 websocket spike -Policy: routed providers keep supports_websockets absent/false -Reason: upstream routed providers are HTTP/SSE, so websocket is not end-to-end -``` - -If websocket-native provider support is ever useful, it should be a separate transport project after -a provider exposes a real websocket endpoint that opencodex can bridge without converting back to -HTTP/SSE internally. - -## Follow-up Decision Update - -The follow-up investigation changed the first implementation recommendation: - -```text -Earlier: try rewriting model_messages first. -Now: strip model_messages from routed non-OpenAI entries first. -``` - -Reason: Codex prefers `model_messages.instructions_template` over `base_instructions`. The current -opencodex catalog rewrite only changes `base_instructions`, so routed models cloned from native -`gpt-5.5` can still receive the native GPT/Codex template. Stripping is the lowest-risk way to make -Codex use the already-rewritten `base_instructions` and disables `/personality` only until -provider-safe templates exist. - -## Source Baseline - -The upstream Codex source inspected for this planning pass was cloned at: - -```text -/tmp/opencodex-codex-src -``` - -The inspected commit was: - -```text -c83618ab2098525d343df2160d98b2449dca6d5d -``` - -The main opencodex implementation surfaces referenced by the plan are: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts -``` diff --git a/devlog/_fin/100_codex-native-parity/10_search-and-tool-discovery.md b/devlog/_fin/100_codex-native-parity/10_search-and-tool-discovery.md deleted file mode 100644 index dfb47562c..000000000 --- a/devlog/_fin/100_codex-native-parity/10_search-and-tool-discovery.md +++ /dev/null @@ -1,106 +0,0 @@ -# 100.10 — Search and Tool Discovery - -## Questions - -- What does `supports_search_tool` do? -- What does `web_search_tool_type` do? -- If opencodex omits or inherits these fields, what fallback does Codex use? -- Does this map to opencodex's current web-search sidecar correctly? - -## Codex RS Behavior - -`supports_search_tool` is not the hosted OpenAI web-search tool. It gates Codex's deferred -`tool_search` discovery surface for MCP/app/extension tools. If it is false or omitted, Codex does -not expose deferred `tool_search` to the model. Direct tools can still be exposed through the normal -tool plan. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:408 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:328 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:941 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:992 -``` - -`web_search_tool_type` controls the hosted web-search tool shape. The default is text-only. A -`text_and_image` value causes Codex to emit a hosted search tool with image search content types. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:279 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/hosted_spec.rs:20 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/hosted_spec.rs:28 -/tmp/opencodex-codex-src/codex-rs/core/src/config/mod.rs:2403 -/tmp/opencodex-codex-src/codex-rs/core/src/config/mod.rs:3283 -/tmp/opencodex-codex-src/codex-rs/tools/src/tool_spec.rs:36 -``` - -## Fallbacks - -If metadata is missing or unknown: - -- `supports_search_tool` falls back to false. -- `web_search_tool_type` falls back to text-only. -- hosted web-search availability is still gated by Codex config, model transport, and hosted-tool - mode. -- unknown model fallback metadata does not enable deferred tool discovery. - -## Current opencodex Behavior - -opencodex clones a native Codex catalog template in: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -``` - -The routed entries currently do not explicitly remove or override: - -- `supports_search_tool` -- `web_search_tool_type` - -That means routed models can inherit native OpenAI search/tool-discovery capability hints. - -For hosted web search, opencodex mitigates at request time: - -- the Responses parser recognizes hosted `web_search` tools; -- routed-model translation drops hosted OpenAI web-search from upstream tool calls; -- the sidecar can inject a synthetic `web_search(query)` tool only when its prerequisites pass. - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:134 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:141 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:142 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:380 -/Users/jun/Developer/new/700_projects/opencodex/src/web-search/index.ts:30 -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:159 -/Users/jun/Developer/new/700_projects/opencodex/src/web-search/synthetic-tool.ts:11 -``` - -## Gap - -The catalog may advertise native hosted search semantics even when the routed upstream provider -does not have native OpenAI-hosted search. opencodex's sidecar makes this partially usable, but the -metadata is still not explicit. - -If sidecar prerequisites are missing, Codex may plan around a capability that gets removed before -the routed provider sees the request. - -## Phase 100 Recommendation - -Keep the two search concepts separate: - -1. `supports_search_tool`: preserve only if opencodex intentionally wants routed models to use - Codex deferred tool discovery. -2. `web_search_tool_type`: set based on opencodex sidecar capability, not native template - inheritance. -3. Add regression tests proving hosted `web_search` is either translated to the synthetic sidecar - tool or suppressed with predictable behavior. -4. Document that native OpenAI passthrough can keep hosted search metadata, while non-OpenAI routed - models depend on opencodex sidecar search. - -See `11_search-defaults-and-inherited-state.md` for the concrete native model defaults and current -observed opencodex catalog state. diff --git a/devlog/_fin/100_codex-native-parity/11_search-defaults-and-inherited-state.md b/devlog/_fin/100_codex-native-parity/11_search-defaults-and-inherited-state.md deleted file mode 100644 index 74a851989..000000000 --- a/devlog/_fin/100_codex-native-parity/11_search-defaults-and-inherited-state.md +++ /dev/null @@ -1,136 +0,0 @@ -# 100.11 — Search Defaults and Inherited State - -## Question - -What do native Codex default models actually use for `web_search_tool_type`, and what are routed -opencodex models currently inheriting? - -## Native Codex Defaults - -Concrete native catalog values live in: - -```text -/tmp/opencodex-codex-src/codex-rs/models-manager/models.json -``` - -That file is bundled by: - -```text -/tmp/opencodex-codex-src/codex-rs/models-manager/src/lib.rs:12 -``` - -Observed native rows: - -| Model | `web_search_tool_type` | `supports_search_tool` | Notes | -| --- | --- | --- | --- | -| `gpt-5.5` | `text_and_image` | `true` | priority 0, picker-visible, current native default | -| `gpt-5.4` | `text_and_image` | `true` | picker-visible | -| `gpt-5.4-mini` | `text_and_image` | `true` | picker-visible/helper | -| `gpt-5.3-codex` | `text` | `true` | picker-visible | -| `gpt-5.2` | `text` | `true` | picker-visible | -| `codex-auto-review` | `text_and_image` | `true` | hidden helper | - -Codex chooses the default model by sorting available models by `priority`, then picking the first -picker-visible model. With the inspected catalog, that makes `gpt-5.5` the native default, and its -native hosted web-search shape is `text_and_image`. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/models-manager/src/manager.rs:117 -/tmp/opencodex-codex-src/codex-rs/models-manager/src/manager.rs:145 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:625 -``` - -## Missing-Field Fallbacks - -If a model row omits these fields: - -- `web_search_tool_type` falls back to `text`. -- `supports_search_tool` falls back to `false`. - -Unknown model fallback metadata also sets: - -```text -web_search_tool_type = Text -supports_search_tool = false -``` - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:279 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:376 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:408 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:909 -/tmp/opencodex-codex-src/codex-rs/models-manager/src/model_info.rs:65 -/tmp/opencodex-codex-src/codex-rs/models-manager/src/model_info.rs:90 -/tmp/opencodex-codex-src/codex-rs/models-manager/src/model_info.rs:102 -``` - -Runtime interpretation: - -- `text` emits no `search_content_types`. -- `text_and_image` emits `search_content_types: ["text", "image"]`. -- `supports_search_tool` is separate and gates deferred `tool_search`. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:312 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:328 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/hosted_spec.rs:28 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/hosted_spec_tests.rs:20 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan_tests.rs:700 -``` - -## Current opencodex Catalog Observation - -The current opencodex catalog intentionally normalizes routed search metadata after native-template -cloning. - -Observed from: - -```text -/Users/jun/.codex/opencodex-catalog.json -``` - -Representative entries: - -| Slug | `web_search_tool_type` | `supports_search_tool` | -| --- | --- | --- | -| `gpt-5.5` | `text_and_image` | `true` | -| `gpt-5.3-codex-spark` | `text` | `true` | -| `opencode-go/kimi-k2.7-code` | `text_and_image` | `true` | -| `opencode-go/glm-5.2` | `text_and_image` | `true` | -| `opencode-go/deepseek-v4-pro` | `text_and_image` | `true` | - -This is not because opencode-go models have native OpenAI hosted image-search support. It is because -`normalizeRoutedCatalogEntry()` deliberately rewrites routed catalog entries to `text_and_image` -after cloning. opencodex then executes hosted search through the native `gpt-5.4-mini` sidecar and -passes routed models a synthetic search tool plus textual summaries of any image results. - -Local source: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:73-88 -``` - -## Decision - -The current behavior is now the resolved Phase 100.2/100.16 policy: - -- native OpenAI passthrough may keep `text_and_image` and `supports_search_tool`; -- routed non-OpenAI models should not inherit hosted OpenAI search semantics silently; -- Phase 100.2 now deliberately sets routed `web_search_tool_type = "text_and_image"` because - opencodex executes hosted search through the native `gpt-5.4-mini` sidecar; -- routed upstream providers still do not receive OpenAI hosted image-search tools directly; - opencodex suppresses the hosted tool and exposes a synthetic search function to the routed model; -- for text-only routed models, image search results are verbalized as text with source URLs; -- the earlier text-only recommendation is superseded by `16_search-image-sidecar-correction.md`. - -## Implementation Note - -`supports_search_tool` should not be used as a web-search flag. It is for deferred tool discovery. -It remains enabled for routed models because opencodex intentionally relays Codex's deferred -tool-discovery surface through parser/bridge handling. diff --git a/devlog/_fin/100_codex-native-parity/12_catalog-normalization-implementation-plan.md b/devlog/_fin/100_codex-native-parity/12_catalog-normalization-implementation-plan.md deleted file mode 100644 index c832554b0..000000000 --- a/devlog/_fin/100_codex-native-parity/12_catalog-normalization-implementation-plan.md +++ /dev/null @@ -1,127 +0,0 @@ -# 100.12 — Catalog Normalization Implementation Plan - -## PABCD Cycle - -This document is the P-phase artifact for Phase 100.1. - -Goal: - -```text -Make every routed non-OpenAI catalog entry safe after native Codex template cloning. -``` - -This is the first implementation slice because it blocks the most dangerous accidental inheritance: -native `model_messages`, `tool_mode`, `multi_agent_version`, and `use_responses_lite`. - -## Current Shape - -Primary implementation path: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -``` - -Current behavior: - -1. `deriveEntry()` clones a native Codex template. -2. It rewrites `slug`, `display_name`, `description`, `priority`, `visibility`, and `base_instructions`. -3. For routed slugs, it already removes speed/service-tier metadata. -4. Before this Phase 100.1 implementation, it did not strip `model_messages`, `tool_mode`, - `multi_agent_version`, or `use_responses_lite`. - -Risk: - -native templates can carry prompt/runtime selectors that are valid for OpenAI native models but wrong -for routed models. - -## Implementation Plan - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -``` - -Add a dedicated helper: - -```ts -export function normalizeRoutedCatalogEntry(entry: RawEntry): RawEntry { - delete entry.model_messages; - delete entry.tool_mode; - delete entry.multi_agent_version; - delete entry.use_responses_lite; - delete entry.supports_websockets; - delete entry.additional_speed_tiers; - delete entry.service_tier; - delete entry.service_tiers; - delete entry.default_service_tier; - return entry; -} -``` - -Then call it from the routed branch inside `deriveEntry()` after the identity rewrite and reasoning -ladder setup. - -Native bare GPT entries must not pass through this helper. - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -``` - -Add Bun tests that: - -1. build a routed entry from a native-like template containing: - - `model_messages` - - `tool_mode` - - `multi_agent_version` - - `use_responses_lite` - - `supports_websockets` - - speed/service-tier fields -2. assert routed output strips those fields; -3. assert routed `base_instructions` no longer claims GPT/OpenAI identity; -4. assert native bare GPT entries preserve native-only fields. - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/package.json -``` - -Add: - -```json -"test": "bun test" -``` - -so each Phase 100 cycle can use a stable test command. - -## Verification - -Run: - -```bash -bun test -bun x tsc --noEmit -git diff --check -``` - -Manual catalog assertions: - -- routed entries have no `model_messages`; -- routed entries have no `tool_mode`; -- routed entries have no `multi_agent_version`; -- routed entries have no `use_responses_lite`; -- routed entries have no `supports_websockets`; -- native bare GPT entries preserve native catalog fields. - -## Commit - -Commit message: - -```text -fix: normalize routed Codex catalog entries -``` - -This commit should include code, tests, and this devlog file. diff --git a/devlog/_fin/100_codex-native-parity/13_catalog-normalization-completion.md b/devlog/_fin/100_codex-native-parity/13_catalog-normalization-completion.md deleted file mode 100644 index f493b0ec1..000000000 --- a/devlog/_fin/100_codex-native-parity/13_catalog-normalization-completion.md +++ /dev/null @@ -1,76 +0,0 @@ -# 100.13 — Catalog Normalization Completion - -## Scope - -Phase 100.1 implemented routed catalog selector normalization. - -Primary files: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -/Users/jun/Developer/new/700_projects/opencodex/package.json -``` - -## Implemented Behavior - -Routed non-OpenAI catalog entries now pass through `normalizeRoutedCatalogEntry()` after native -template cloning. - -The helper strips: - -- `model_messages` -- `tool_mode` -- `multi_agent_version` -- `use_responses_lite` -- `supports_websockets` -- `additional_speed_tiers` -- `service_tier` -- `service_tiers` -- `default_service_tier` - -Native bare GPT entries do not pass through this helper, so native OpenAI passthrough metadata is -preserved. - -## Test Coverage - -Added: - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -``` - -Covered assertions: - -1. direct `normalizeRoutedCatalogEntry()` stripping; -2. routed entries produced by `buildCatalogEntries()` strip native-only selectors; -3. native bare GPT entries preserve native-only fields and still normalize `fast` catalog tier ids - to `priority`. - -## Verification - -Commands: - -```bash -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Results: - -```text -bun test tests: 3 pass, 0 fail -bun x tsc --noEmit: pass -git diff --check: pass -Backend verifier: DONE -``` - -## Remaining Phase 100 Work - -Phase 100.1 only blocks accidental native selector inheritance in the catalog. Remaining slices: - -- 100.2 search capability policy; -- 100.3 thinking and usage parity; -- 100.4 jawcode-backed context metadata; -- 100.5 error and header fidelity. diff --git a/devlog/_fin/100_codex-native-parity/14_search-policy-implementation-plan.md b/devlog/_fin/100_codex-native-parity/14_search-policy-implementation-plan.md deleted file mode 100644 index f439caa6c..000000000 --- a/devlog/_fin/100_codex-native-parity/14_search-policy-implementation-plan.md +++ /dev/null @@ -1,116 +0,0 @@ -# 100.14 — Search Policy Implementation Plan - -## PABCD Cycle - -This document is the P-phase artifact for Phase 100.2. - -Goal: - -```text -Make routed search catalog metadata deliberate instead of inherited from native OpenAI templates. -``` - -## Current Shape - -Current request-time behavior is mostly correct: - -- hosted `web_search` tools are extracted by the Responses parser; -- hosted `web_search` is dropped from routed upstream tools; -- `planWebSearch()` injects a synthetic function tool only when sidecar prerequisites are present; -- `tool_search` is already translated through parser/bridge as a client-executed tool discovery call. - -Current catalog behavior is not explicit: - -- routed entries can still inherit native `web_search_tool_type = "text_and_image"`; -- routed entries can still inherit `supports_search_tool` accidentally. - -## Policy - -For routed non-OpenAI catalog entries: - -```text -web_search_tool_type = "text_and_image" -supports_search_tool = true -``` - -Reason: - -- `web_search_tool_type = "text_and_image"` is truthful for opencodex's routed path because hosted - search is executed by the default `gpt-5.4-mini` sidecar, and native Codex marks `gpt-5.4-mini` as - text+image search capable. -- Routed upstream models still do not receive OpenAI hosted image-search tools directly. opencodex - intercepts hosted web search, then the sidecar verbalizes image results for text-only routed models. -- `supports_search_tool = true` is deliberate, not inherited: opencodex already supports Codex's - deferred `tool_search` surface through parser/bridge relaying. - -If future evidence shows routed `tool_search` is unsafe for a provider class, this should become a -provider/model capability flag, not a copied native template value. - -## Implementation Plan - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -``` - -Add: - -```ts -function normalizeRoutedSearchMetadata(entry: RawEntry): void { - entry.web_search_tool_type = "text_and_image"; - entry.supports_search_tool = true; -} -``` - -Call it from `normalizeRoutedCatalogEntry()` after native-only selector stripping. - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -``` - -Extend the native-like template to include `web_search_tool_type = "text_and_image"` and -`supports_search_tool = true`. - -Add tests proving: - -1. routed entries normalize `web_search_tool_type` to `text_and_image`; -2. routed entries deliberately keep `supports_search_tool = true`; -3. native bare GPT entries preserve `text_and_image`. - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/web-search.test.ts -``` - -Add request-time tests proving: - -1. `parseRequest()` stashes hosted `web_search`; -2. `planWebSearch()` returns a plan only when: - - hosted web search was requested; - - route is not passthrough; - - a forward ChatGPT provider exists; - - incoming authorization exists; - - sidecar is not disabled; -3. `planWebSearch()` returns `undefined` when these prerequisites are absent. - -## Verification - -Run: - -```bash -bun test tests -bun x tsc --noEmit -git diff --check -``` - -## Commit - -Commit message: - -```text -fix: normalize routed search metadata -``` diff --git a/devlog/_fin/100_codex-native-parity/15_search-policy-completion.md b/devlog/_fin/100_codex-native-parity/15_search-policy-completion.md deleted file mode 100644 index 9767a6505..000000000 --- a/devlog/_fin/100_codex-native-parity/15_search-policy-completion.md +++ /dev/null @@ -1,79 +0,0 @@ -# 100.15 — Search Policy Completion - -## Scope - -Phase 100.2 implemented explicit routed search metadata and request-time sidecar regression tests. - -Primary files: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -/Users/jun/Developer/new/700_projects/opencodex/tests/web-search.test.ts -``` - -## Implemented Behavior - -Routed non-OpenAI catalog entries now set: - -```text -web_search_tool_type = "text_and_image" -supports_search_tool = true -``` - -This is deliberate policy, not native-template inheritance. - -Meaning: - -- `web_search_tool_type = "text_and_image"` advertises the capability opencodex can actually provide - through the default `gpt-5.4-mini` sidecar. The routed upstream model does not run OpenAI hosted - image search directly; the sidecar runs it and verbalizes image results when the downstream model is - text-only. -- `supports_search_tool = true` keeps Codex deferred `tool_search` available. It does not enable - hosted web search by itself. - -Hosted web search still depends on request-time sidecar prerequisites: - -- Codex must send a hosted `web_search` tool; -- the request must be routed, not native passthrough; -- a forward ChatGPT provider must exist; -- incoming ChatGPT authorization must be present; -- `webSearchSidecar.enabled` must not be false. - -If any prerequisite is absent, the hosted `web_search` tool remains suppressed and is not exposed to -the routed upstream model. - -## Test Coverage - -Added or extended tests: - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -/Users/jun/Developer/new/700_projects/opencodex/tests/web-search.test.ts -``` - -Covered assertions: - -1. routed catalog entries force `web_search_tool_type = "text_and_image"`; -2. routed catalog entries deliberately set `supports_search_tool = true`; -3. native bare GPT entries preserve native `text_and_image`; -4. template-less fallback routed entries still receive explicit search metadata; -5. `parseRequest()` stashes hosted web search while preserving normal function tools; -6. `planWebSearch()` activates only when all sidecar prerequisites pass; -7. `planWebSearch()` returns `undefined` predictably when prerequisites are absent. - -## Verification - -Commands: - -```bash -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Expected result: - -```text -all pass -``` diff --git a/devlog/_fin/100_codex-native-parity/16_search-image-sidecar-correction.md b/devlog/_fin/100_codex-native-parity/16_search-image-sidecar-correction.md deleted file mode 100644 index 15126c604..000000000 --- a/devlog/_fin/100_codex-native-parity/16_search-image-sidecar-correction.md +++ /dev/null @@ -1,60 +0,0 @@ -# 100.16 — Search Image Sidecar Correction - -## Correction - -After the Phase 100.2 implementation, user review caught an important distinction: - -```text -gpt-5.4-mini supports native text+image web search. -``` - -The first 100.2 implementation normalized routed entries to: - -```text -web_search_tool_type = "text" -``` - -That was too conservative for opencodex's actual web-search path. - -## Final Policy - -Routed non-OpenAI catalog entries should advertise: - -```text -web_search_tool_type = "text_and_image" -supports_search_tool = true -``` - -This does not mean the routed upstream provider receives OpenAI hosted image-search tools directly. -The runtime path remains: - -1. Codex enables hosted web search from catalog metadata. -2. opencodex parses and suppresses the hosted `web_search` tool before sending the request upstream. -3. opencodex exposes a synthetic `web_search(query)` function to the routed model when sidecar - prerequisites are met. -4. The sidecar executes real hosted web search through the native ChatGPT forward provider, using - `gpt-5.4-mini` by default. -5. If the routed target model is text-only, the sidecar verbalizes relevant image results and includes - source URLs. - -## Why `text_and_image` Is Truthful - -Native Codex metadata marks `gpt-5.4-mini` as: - -```text -web_search_tool_type = "text_and_image" -supports_search_tool = true -``` - -opencodex's sidecar uses that model for the real hosted search call. Therefore the catalog capability -is true for the opencodex route as a whole, even when the final routed upstream model only receives a -textual summary of image results. - -## Verification Scope - -Regression tests now assert: - -1. routed catalog entries normalize `web_search_tool_type` to `text_and_image`; -2. native bare GPT entries still preserve `text_and_image`; -3. template-less routed fallback entries receive the same explicit metadata; -4. request-time sidecar prerequisites still control whether hosted search is actually executed. diff --git a/devlog/_fin/100_codex-native-parity/17_thinking-usage-implementation-plan.md b/devlog/_fin/100_codex-native-parity/17_thinking-usage-implementation-plan.md deleted file mode 100644 index d7a7cd5b4..000000000 --- a/devlog/_fin/100_codex-native-parity/17_thinking-usage-implementation-plan.md +++ /dev/null @@ -1,291 +0,0 @@ -# 100.17 — Thinking and Usage Parity Implementation Plan - -## Easy Summary - -Phase 100.3 makes translated providers look more like native Codex Responses output. The proxy will -keep existing reasoning summary events, add a separate raw reasoning text path for providers that -actually emit raw reasoning content, and report cached/reasoning token details when providers expose -them. This should improve Codex CLI/App thinking display and token accounting without pretending -unknown providers support metadata they do not expose. - -## Current State - -Relevant source files: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/anthropic.ts -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts -``` - -Current gaps: - -1. `AdapterEvent` only has `thinking_delta`, and the bridge always emits it as - `response.reasoning_summary_text.delta`. -2. OpenAI-compatible `delta.reasoning_content` is raw reasoning-like content but currently gets - downgraded to a summary. -3. `OcxUsage` only has `inputTokens` and `outputTokens`, so Responses usage lacks: - - `input_tokens_details.cached_tokens` - - `output_tokens_details.reasoning_tokens` -4. Non-streaming JSON responses ignore reasoning deltas entirely. - -## Policy - -Use two reasoning event classes: - -```ts -| { type: "thinking_delta"; thinking: string } -| { type: "reasoning_raw_delta"; text: string } -``` - -Provider mapping: - -1. Anthropic `thinking_delta` stays `thinking_delta` for now because its signed thinking blocks also - serve provider-specific continuity and are already represented as summary in opencodex history. -2. OpenAI-compatible `reasoning_content` becomes `reasoning_raw_delta` because that field is a raw - reasoning stream on compatible chat APIs. -3. Google remains text/tool only until a provider-specific raw-thinking field is observed. - -Usage details policy: - -1. Preserve existing totals. -2. Add optional details only when upstream reports them. -3. Do not fabricate cache or reasoning-token values. - -## Diff-Level Plan - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts -``` - -Change `AdapterEvent`: - -```diff - export type AdapterEvent = - | { type: "text_delta"; text: string } - | { type: "thinking_delta"; thinking: string } -+ | { type: "reasoning_raw_delta"; text: string } - | { type: "tool_call_start"; id: string; name: string } -``` - -Extend `OcxUsage`: - -```diff - export interface OcxUsage { - inputTokens: number; - outputTokens: number; -+ cachedInputTokens?: number; -+ reasoningOutputTokens?: number; - } -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -``` - -Add a helper: - -```ts -function responsesUsage(usage: OcxUsage | undefined): Record { - if (!usage) return { input_tokens: 0, output_tokens: 0, total_tokens: 0 }; - const out: Record = { - input_tokens: usage.inputTokens, - output_tokens: usage.outputTokens, - total_tokens: usage.inputTokens + usage.outputTokens, - }; - if (usage.cachedInputTokens !== undefined) { - out.input_tokens_details = { cached_tokens: usage.cachedInputTokens }; - } - if (usage.reasoningOutputTokens !== undefined) { - out.output_tokens_details = { reasoning_tokens: usage.reasoningOutputTokens }; - } - return out; -} -``` - -Replace duplicated stream/non-stream usage literals with `responsesUsage(event.usage)`. - -Add raw reasoning state alongside the existing summary state: - -```ts -let currentRawReasoning: { itemId: string; outputIndex: number; text: string } | null = null; -``` - -Add a `closeCurrentRawReasoning()` that finalizes: - -```ts -{ - type: "reasoning", - id: currentRawReasoning.itemId, - summary: [], - content: [{ type: "reasoning_text", text: currentRawReasoning.text }], -} -``` - -Handle `reasoning_raw_delta` by emitting: - -```text -response.output_item.added -response.reasoning_text.delta -``` - -The raw delta payload must include the fields Codex RS requires: - -```ts -emit("response.reasoning_text.delta", { - item_id: currentRawReasoning.itemId, - output_index: currentRawReasoning.outputIndex, - content_index: 0, - delta: event.text, -}); -``` - -Do not emit summary events for raw reasoning. - -Raw and summary reasoning state must be mutually exclusive: - -1. `reasoning_raw_delta` closes `currentReasoning` and `currentToolCall`. -2. `thinking_delta` closes `currentRawReasoning` and `currentToolCall`. -3. `text_delta`, `tool_call_start`, `done`, and `error` close both reasoning states. -4. `closeCurrentRawReasoning()` increments `outputIndex` exactly once, matching - `closeCurrentReasoning()`. - -Update `buildResponseJSON()` so non-streaming raw reasoning produces a completed reasoning item -before the assistant message. - -Non-streaming ordering: - -```text -reasoning_raw_delta events -> one reasoning output item with content[] -thinking_delta events -> one reasoning output item with summary[] -text_delta events -> one assistant message item -done -> usage -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts -``` - -Change streaming mapping: - -```diff -- yield { type: "thinking_delta", thinking: delta.reasoning_content }; -+ yield { type: "reasoning_raw_delta", text: delta.reasoning_content }; -``` - -Change non-streaming mapping: - -```ts -if (typeof msg.reasoning_content === "string" && msg.reasoning_content.length > 0) { - events.push({ type: "reasoning_raw_delta", text: msg.reasoning_content }); -} -``` - -Map usage details: - -```ts -const promptDetails = usage.prompt_tokens_details as Record | undefined; -const completionDetails = usage.completion_tokens_details as Record | undefined; -usage: { - inputTokens: usage.prompt_tokens ?? 0, - outputTokens: usage.completion_tokens ?? 0, - ...(promptDetails?.cached_tokens !== undefined ? { cachedInputTokens: promptDetails.cached_tokens } : {}), - ...(completionDetails?.reasoning_tokens !== undefined ? { reasoningOutputTokens: completionDetails.reasoning_tokens } : {}), -} -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/anthropic.ts -``` - -Keep `thinking_delta` as summary. Extend usage mapping: - -```ts -const cacheRead = usage.cache_read_input_tokens ?? 0; -const cacheCreation = usage.cache_creation_input_tokens ?? 0; -const cachedInputTokens = cacheRead + cacheCreation; -``` - -Include `cachedInputTokens` only when either upstream field is present: - -```ts -const hasCache = - usage.cache_read_input_tokens !== undefined || - usage.cache_creation_input_tokens !== undefined; -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts -``` - -Extend usage mapping when Gemini returns known metadata: - -```ts -cachedInputTokens: usageMeta.cachedContentTokenCount -reasoningOutputTokens: usageMeta.thoughtsTokenCount -``` - -Keep both optional. - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/bridge.test.ts -``` - -Add focused bridge tests: - -1. streaming `reasoning_raw_delta` emits `response.reasoning_text.delta` and final reasoning content; -2. streaming `thinking_delta` still emits summary events; -3. usage details serialize into `input_tokens_details.cached_tokens` and - `output_tokens_details.reasoning_tokens`; -4. non-streaming JSON includes raw reasoning item and usage details. -5. raw reasoning closes before later text output, preserving output ordering and indexes. - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/adapter-usage.test.ts -``` - -Add adapter-level unit tests for: - -1. OpenAI-compatible usage details and `reasoning_content` mapping; -2. Anthropic cache-token mapping; -3. Google cached/thoughts-token mapping. - -## Verification - -Run: - -```bash -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Expected result: - -```text -all pass -``` - -## Acceptance Criteria - -1. Existing summary thinking behavior does not regress. -2. OpenAI-compatible raw `reasoning_content` reaches Codex as raw `reasoning_text`. -3. Non-streaming translated responses preserve raw reasoning content. -4. Usage details are present when upstream providers expose them and absent when unknown. -5. No provider gets fabricated cache/reasoning token counts. diff --git a/devlog/_fin/100_codex-native-parity/18_jawcode-context-metadata-implementation-plan.md b/devlog/_fin/100_codex-native-parity/18_jawcode-context-metadata-implementation-plan.md deleted file mode 100644 index 185a85ca2..000000000 --- a/devlog/_fin/100_codex-native-parity/18_jawcode-context-metadata-implementation-plan.md +++ /dev/null @@ -1,239 +0,0 @@ -# 100.18 — jawcode Context Metadata Implementation Plan - -## Easy Summary - -Phase 100.4 gives routed models real context-window metadata instead of accidentally inheriting the -native GPT catalog window. opencodex will not import jawcode at runtime. Instead, a build-time script -will read jawcode's static `models.json`, generate a tiny opencodex-owned snapshot, and the catalog -builder will apply exact provider/model matches to Codex catalog fields. - -## Policy - -Do: - -```text -jawcode models.json -> generated opencodex snapshot -> exact provider/model lookup -> Codex fields -``` - -Do not: - -```text -runtime import @jawcode-dev/ai -spread full jawcode model objects into Codex catalog -guess unknown provider/model metadata -``` - -Exact match behavior: - -1. mapped provider + exact model id -> apply jawcode metadata; -2. mapped provider + unknown model -> keep current conservative/template metadata; -3. unmapped provider -> keep current metadata. - -## Provider Mapping - -Initial mapping: - -```ts -const PROVIDER_ALIASES = { - "xai": "xai", - "anthropic": "anthropic", - "google": "google", - "gemini": "google", - "moonshot": "moonshot", - "kimi": "moonshot", - "openrouter": "openrouter", - "opencode-go": "opencode-go", -} as const; -``` - -This includes the user-confirmed correction that `opencode-go` exists in jawcode and maps directly. - -## Codex Field Mapping - -From jawcode: - -```text -contextWindow -> context_window -contextWindow -> max_context_window -floor(contextWindow * 0.9) -> auto_compact_token_limit -input -> input_modalities -``` - -Do not map: - -```text -maxTokens -> context_window -max_context_window_tokens -> context_window -``` - -`maxTokens` is output-token metadata, not context capacity. - -## Diff-Level Plan - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/scripts/generate-jawcode-metadata.ts -``` - -Responsibilities: - -1. Read jawcode model registry from: - - `process.env.JAWCODE_MODELS_JSON`, or - - `../jawcode/packages/ai/src/models.json` relative to opencodex cwd. -2. Filter to the unique jawcode provider ids used by `PROVIDER_ALIASES` only: - - `xai` - - `anthropic` - - `google` - - `moonshot` - - `openrouter` - - `opencode-go` -3. Project each allowed model to compact tuples: - - provider - - id - - contextWindow - - maxTokens - - input - - reasoning - - wireModelId -4. Write: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/generated/jawcode-model-metadata.ts -``` - -The generated file must be deterministic: sorted providers and sorted model ids. -It must not emit the full jawcode registry; large unrelated provider catalogs stay out of opencodex. - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/generated/jawcode-model-metadata.ts -``` - -Generated exports: - -```ts -export interface JawcodeModelMetadata { - provider: string; - id: string; - contextWindow?: number; - maxTokens?: number; - input?: ("text" | "image")[]; - reasoning?: boolean; - wireModelId?: string; -} - -export function getJawcodeModelMetadata(provider: string, modelId: string): JawcodeModelMetadata | undefined; -``` - -The generated module also owns alias resolution: - -```ts -const PROVIDER_ALIASES: Record = { - "xai": "xai", - "anthropic": "anthropic", - "google": "google", - "gemini": "google", - "moonshot": "moonshot", - "kimi": "moonshot", - "openrouter": "openrouter", - "opencode-go": "opencode-go", -}; - -export function resolveJawcodeProvider(provider: string): string | undefined { - return PROVIDER_ALIASES[provider]; -} -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -``` - -Add import: - -```ts -import { getJawcodeModelMetadata, resolveJawcodeProvider } from "./generated/jawcode-model-metadata"; -``` - -Add helper: - -```ts -function applyJawcodeCatalogMetadata(entry: RawEntry, slug: string): void { - const slash = slug.indexOf("/"); - if (slash < 0) return; - const provider = slug.slice(0, slash); - const modelId = slug.slice(slash + 1); - const jawcodeProvider = resolveJawcodeProvider(provider); - if (!jawcodeProvider) return; - const meta = getJawcodeModelMetadata(jawcodeProvider, modelId); - if (!meta) return; - if (typeof meta.contextWindow === "number" && meta.contextWindow > 0) { - entry.context_window = meta.contextWindow; - entry.max_context_window = meta.contextWindow; - entry.auto_compact_token_limit = Math.floor(meta.contextWindow * 0.9); - } - if (Array.isArray(meta.input) && meta.input.length > 0) { - entry.input_modalities = meta.input; - } -} -``` - -Call sites: - -1. Template-backed routed path: call `applyJawcodeCatalogMetadata(e, slug)` immediately after - `normalizeRoutedCatalogEntry(e)` and before `return normalizeServiceTiers(e)`. -2. Template-less fallback path: build the fallback object in a local `const entry`, call - `applyJawcodeCatalogMetadata(entry, slug)`, then return `normalizeServiceTiers(entry)`. - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/package.json -``` - -Add script: - -```json -"generate:jawcode-metadata": "bun scripts/generate-jawcode-metadata.ts" -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -``` - -Add tests proving: - -1. exact `opencode-go/deepseek-v4-pro` gets jawcode context metadata; -2. provider alias `kimi/kimi-k2.5` resolves through jawcode `moonshot`; -3. unknown provider/model keeps existing template/fallback context values and does not guess; -4. generated snapshot does not include unrelated providers outside the alias allowlist. - -## Verification - -Run: - -```bash -bun run generate:jawcode-metadata -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Expected: - -```text -all pass -``` - -## Acceptance Criteria - -1. No runtime dependency on `@jawcode-dev/ai`. -2. Generated snapshot is deterministic and committed. -3. Routed exact matches receive jawcode context metadata. -4. Unknown routed models are not guessed. -5. Existing catalog normalization/search/service-tier behavior does not regress. diff --git a/devlog/_fin/100_codex-native-parity/19_search-policy-doc-reconciliation-plan.md b/devlog/_fin/100_codex-native-parity/19_search-policy-doc-reconciliation-plan.md deleted file mode 100644 index f09eb8890..000000000 --- a/devlog/_fin/100_codex-native-parity/19_search-policy-doc-reconciliation-plan.md +++ /dev/null @@ -1,149 +0,0 @@ -# 100.19 — Search Policy Documentation Reconciliation Plan - -## Objective - -Reconcile Phase 100 search-policy documentation after the confirmed `gpt-5.4-mini` text+image -sidecar correction. - -The implementation is already correct on `dev`: - -```text -routed web_search_tool_type = "text_and_image" -routed supports_search_tool = true -``` - -The remaining issue is documentation drift: older research docs still contain the pre-correction -recommendation that routed entries should prefer `text` unless image-search is proven end to end. -That statement is now superseded because opencodex executes hosted search through the native -`gpt-5.4-mini` sidecar and verbalizes image results for text-only routed models. - -## Files - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/devlog/100_codex-native-parity/11_search-defaults-and-inherited-state.md -``` - -Update the stale current-observation opening sentence so it no longer says routed entries are -accidentally inheriting native metadata: - -```diff --The live local opencodex catalog currently shows routed entries inheriting native search metadata. -+The current opencodex catalog intentionally normalizes routed search metadata after native-template -+cloning. -``` - -Keep the current example table and local catalog path block, then replace the stale explanation below -that block: - -```diff --This is not because opencode-go models have native OpenAI hosted image-search support. It is because -+opencodex cloned a native template and did not normalize these fields. -+This is not because opencode-go models have native OpenAI hosted image-search support. It is because -+`normalizeRoutedCatalogEntry()` deliberately rewrites routed catalog entries to `text_and_image` -+after cloning. opencodex then executes hosted search through the native `gpt-5.4-mini` sidecar and -+passes routed models a synthetic search tool plus textual summaries of any image results. -``` - -Update the source pointer: - -```diff --/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:108 -+/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:73-88 -``` - -Update the decision preamble: - -```diff --Leave the current behavior documented for now, but treat it as explicit technical debt: -+The current behavior is now the resolved Phase 100.2/100.16 policy: -``` - -Replace the stale decision bullets: - -```diff -- native OpenAI passthrough may keep `text_and_image` and `supports_search_tool`; -- routed non-OpenAI models should not inherit hosted OpenAI search semantics silently; -- if opencodex keeps `web_search_tool_type = text_and_image`, the docs must state it means Codex -- will offer a hosted text+image web-search shape, while opencodex currently converts/drops hosted -- tools and can only provide sidecar synthetic web-search behavior; -- first implementation pass should prefer setting routed `web_search_tool_type` to `text` unless -- the sidecar actually supports image-search semantics end to end. -+ native OpenAI passthrough may keep `text_and_image` and `supports_search_tool`; -+ routed non-OpenAI models should not inherit hosted OpenAI search semantics silently; -+ Phase 100.2 now deliberately sets routed `web_search_tool_type = "text_and_image"` because -+ opencodex executes hosted search through the native `gpt-5.4-mini` sidecar; -+ routed upstream providers still do not receive OpenAI hosted image-search tools directly; -+ opencodex suppresses the hosted tool and exposes a synthetic search function to the routed model; -+ for text-only routed models, image search results are verbalized as text with source URLs; -+ the earlier text-only recommendation is superseded by `16_search-image-sidecar-correction.md`. -``` - -Update the implementation note: - -```diff --It can remain enabled for routed models only if opencodex wants routed providers to see Codex's --deferred tool-discovery surface. -+It remains enabled for routed models because opencodex intentionally relays Codex's deferred -+tool-discovery surface through parser/bridge handling. -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/devlog/100_codex-native-parity/90_phase-plan.md -``` - -Mark the search-policy question as resolved while keeping deferred `tool_search` and hosted -`web_search_tool_type` conceptually separate: - -```diff --3. Should routed models expose deferred `tool_search` by default? -+3. Resolved in Phase 100.2/100.16: -+ - routed models expose deferred `tool_search` by default; -+ - routed hosted web-search metadata is `text_and_image` because actual hosted search runs via -+ native `gpt-5.4-mini` sidecar. -``` - -### NO SOURCE CHANGE - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts -``` - -No code patch is planned because the current source and tests already assert the corrected policy: - -```text -normalizeRoutedCatalogEntry(entry).web_search_tool_type === "text_and_image" -normalizeRoutedCatalogEntry(entry).supports_search_tool === true -``` - -## Verification - -Run: - -```bash -bun test tests/codex-catalog.test.ts tests/web-search.test.ts -bun x tsc --noEmit -rg -n 'prefer setting routed|explicit technical debt|did not normalize these fields|codex-catalog.ts:108' devlog/100_codex-native-parity/11_search-defaults-and-inherited-state.md devlog/100_codex-native-parity/90_phase-plan.md -git diff --check -``` - -Expected result: - -```text -all targeted catalog/search tests pass -typecheck passes -stale doc phrase search returns no matches -diff whitespace check passes -``` - -## Commit - -If approved, commit as: - -```text -docs: reconcile routed search policy -``` diff --git a/devlog/_fin/100_codex-native-parity/20_personality-model-messages.md b/devlog/_fin/100_codex-native-parity/20_personality-model-messages.md deleted file mode 100644 index 6e6f31782..000000000 --- a/devlog/_fin/100_codex-native-parity/20_personality-model-messages.md +++ /dev/null @@ -1,123 +0,0 @@ -# 100.20 — Model Messages and Personality - -## Questions - -- What is `model_messages`? -- What is `supports_personality`? -- Does inheriting a native Codex template affect routed model identity? - -## Codex RS Behavior - -`model_messages` is prompt assembly metadata. It is not display copy. It can contain: - -- `instructions_template` -- `instructions_variables` - -If `instructions_template` is present, Codex uses it through `ModelInfo::get_model_instructions()` -before falling back to `base_instructions`. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:346 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:446 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:452 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:474 -``` - -`supports_personality` is derived, not a raw field. Codex reports personality support only when: - -1. `model_messages` exists; -2. the template contains `{{ personality }}`; -3. all three variables exist: - - `personality_default` - - `personality_friendly` - - `personality_pragmatic` - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:482 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:505 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:553 -/tmp/opencodex-codex-src/codex-rs/tui/src/chatwidget/settings.rs:288 -/tmp/opencodex-codex-src/codex-rs/tui/src/chatwidget/settings_popups.rs:23 -/tmp/opencodex-codex-src/codex-rs/tui/src/chatwidget/input_submission.rs:333 -``` - -## Current opencodex Behavior - -Routed entries are cloned from a native template in: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:108 -``` - -For namespaced routed models, opencodex currently rewrites `base_instructions` identity text: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:122 -``` - -It does not rewrite `model_messages.instructions_template`. - -## Gap - -If a native template includes `model_messages.instructions_template`, Codex may use that template -instead of the rewritten `base_instructions`. That can leak GPT/Codex/OpenAI identity text into -routed models such as Claude, Grok, Gemini, Kimi, OpenRouter, local vLLM, or Ollama. - -This also means the Codex personality UI can appear for routed models because the template supports -personality, even though the routed provider identity text was not normalized in the active prompt -template. - -## Options - -### Option A — Strip `model_messages` for Routed Models - -Pros: - -- safest identity behavior; -- avoids pretending routed models support Codex-native personality templates; -- simple implementation. - -Cons: - -- disables `/personality` UX for routed models; -- loses any useful prompt-template scaffolding in Codex native metadata. - -### Option B — Rewrite `model_messages.instructions_template` - -Pros: - -- preserves personality UX; -- keeps Codex prompt assembly structure; -- avoids identity leak if every identity-bearing string is rewritten correctly. - -Cons: - -- requires careful template-specific rewrite logic; -- templates may change upstream; -- provider-specific model families may need different identity wording. - -## Superseded Recommendation - -The initial recommendation was to try Option B first if rewrite could be made robust. The follow-up -investigation supersedes that: routed non-OpenAI models should strip `model_messages` first. - -See: - -```text -/Users/jun/Developer/new/700_projects/opencodex/devlog/100_codex-native-parity/21_model-messages-strip-first.md -``` - -Reason: `model_messages.instructions_template` is not a cosmetic field. Codex uses it before -`base_instructions`, so the existing routed `base_instructions` rewrite can be bypassed. - -Minimum acceptance criteria: - -- routed `provider/model` catalog entries must not mention OpenAI/GPT/Codex as the model identity - unless the provider is native OpenAI passthrough; -- `supports_personality` should be true only when the final prompt template is known to be safe for - routed providers; -- tests should inspect generated catalog JSON, not only runtime UI output. diff --git a/devlog/_fin/100_codex-native-parity/21_model-messages-strip-first.md b/devlog/_fin/100_codex-native-parity/21_model-messages-strip-first.md deleted file mode 100644 index 410d4790d..000000000 --- a/devlog/_fin/100_codex-native-parity/21_model-messages-strip-first.md +++ /dev/null @@ -1,140 +0,0 @@ -# 100.21 — Model Messages Strip-First Decision - -## Decision - -For routed non-OpenAI models, strip `model_messages` first. - -Do not try to preserve `/personality` on routed models until opencodex owns provider-safe -`model_messages` templates. - -## Why This Changed - -Earlier Phase 100 docs left two options open: - -1. strip `model_messages`; -2. rewrite `model_messages.instructions_template`. - -Follow-up Codex RS inspection makes the safer path clear. `model_messages` is live prompt assembly -metadata. Codex calls: - -```text -model_info.get_model_instructions(config.personality) -``` - -and that method prefers: - -```text -model_messages.instructions_template -``` - -over: - -```text -base_instructions -``` - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:592 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:452 -``` - -## Native Model State - -Most current native picker models use `model_messages`: - -| Model | `model_messages` | -| --- | --- | -| `gpt-5.5` | present | -| `gpt-5.4` | present | -| `gpt-5.4-mini` | present | -| `gpt-5.3-codex` | present | -| `gpt-5.2` | null | -| `codex-auto-review` | present | - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/models-manager/models.json:24 -/tmp/opencodex-codex-src/codex-rs/models-manager/models.json:118 -/tmp/opencodex-codex-src/codex-rs/models-manager/models.json:207 -/tmp/opencodex-codex-src/codex-rs/models-manager/models.json:291 -/tmp/opencodex-codex-src/codex-rs/models-manager/models.json:375 -/tmp/opencodex-codex-src/codex-rs/models-manager/models.json:455 -``` - -The template shape is simple but large: - -- an identity header that starts with Codex/GPT-5 wording; -- `{{ personality }}`; -- the main Codex instruction body; -- three personality variables. - -`supports_personality` is derived from the template, not an independent flag: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:446 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:474 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:490 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:512 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:553 -/tmp/opencodex-codex-src/codex-rs/app-server/src/models.rs:25 -``` - -## Current opencodex Risk - -opencodex currently deep-clones the native template and rewrites only `base_instructions`: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:108 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:122 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:135 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:157 -``` - -That rewrite is probably ineffective for routed entries cloned from modern native models, because -Codex can ignore `base_instructions` when `model_messages.instructions_template` exists. - -There is also uneven proxy-side mitigation: - -- OpenAI-compatible chat requests rewrite the first GPT-5 identity line; -- Anthropic and Google pass `systemPrompt` through directly. - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:206 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts:10 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/anthropic.ts:51 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts:13 -``` - -## Practical Policy - -For routed entries where `slug.includes("/")`: - -```text -delete e.model_messages -``` - -Expected result: - -- Codex falls back to `base_instructions`; -- the existing routed identity rewrite becomes active; -- `supports_personality` becomes false for routed models; -- no OpenAI/GPT/Codex identity leaks through a cloned native template. - -Native OpenAI passthrough entries should keep `model_messages`. - -## Future Option - -Provider-safe personality can come back later through opencodex-owned templates. That should be a -separate feature: - -1. create a small routed template independent of upstream GPT wording; -2. keep `{{ personality }}` support; -3. supply all three personality variables; -4. snapshot-test generated catalog entries for identity leaks. - -Until then, strip-first is the correct default. diff --git a/devlog/_fin/100_codex-native-parity/30_tool-mode-multi-agent.md b/devlog/_fin/100_codex-native-parity/30_tool-mode-multi-agent.md deleted file mode 100644 index bbcdfec84..000000000 --- a/devlog/_fin/100_codex-native-parity/30_tool-mode-multi-agent.md +++ /dev/null @@ -1,196 +0,0 @@ -# 100.30 — Tool Mode and Multi-Agent Version - -## Questions - -- What is `tool_mode`? -- What is `multi_agent_version`? -- If opencodex omits or inherits them, what fallback does Codex use? -- Do these fields affect the subagent picker and tool exposure? - -## Codex RS Behavior: `tool_mode` - -`tool_mode` is an optional per-model runtime selector in Codex's model catalog metadata. - -Known values: - -```text -direct -code_mode -code_mode_only -``` - -Relevant upstream definitions: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:297 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:414 -``` - -Semantics: - -- `direct`: expose normal tool specs directly to the model. -- `code_mode`: add code-mode wrapper tools while allowing direct tool exposure where applicable. -- `code_mode_only`: hide nested tools from direct model exposure and force code-mode entrypoints. - -Codex computes the effective mode from `model_info.tool_mode` first. If omitted, it falls back to -feature flags: `CodeModeOnly`, then `CodeMode`, then `Direct`. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/core/src/tools/mod.rs:63 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/mod.rs:64 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:272 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:452 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:475 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:987 -``` - -Unknown `tool_mode` strings deserialize as omitted, not as hard errors. - -Fallback metadata for unknown model slugs sets: - -```text -tool_mode: None -``` - -## Codex RS Behavior: `multi_agent_version` - -`multi_agent_version` is an optional per-model selector for Codex collaboration/subagent tooling. - -Known values: - -```text -disabled -v1 -v2 -``` - -Relevant upstream definitions: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:420 -/tmp/opencodex-codex-src/codex-rs/protocol/src/protocol.rs:2891 -``` - -Semantics: - -- `disabled`: no collaboration/subagent tools. -- `v1`: legacy namespaced multi-agent surface. -- `v2`: newer direct or namespaced agent-control surface with richer agent lifecycle tools. - -For a real turn, Codex resolves and stores one multi-agent version for the session. Once stored, -later model changes do not necessarily change multi-agent behavior inside the same session. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:2916 -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:2921 -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:2925 -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:2927 -/tmp/opencodex-codex-src/codex-rs/core/src/session/turn_context.rs:727 -``` - -If omitted, Codex falls back to feature flags: - -- `multi_agent_v2` enabled -> `V2` -- stable `multi_agent` / `Collab` enabled -> `V1` -- otherwise -> `Disabled` - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/core/src/config/mod.rs:1363 -/tmp/opencodex-codex-src/codex-rs/features/src/lib.rs:1011 -/tmp/opencodex-codex-src/codex-rs/features/src/lib.rs:1017 -``` - -## Tool-Spec Impact - -`multi_agent_version` changes which collaboration tools are exposed: - -- V2 can expose `spawn_agent`, `send_message`, `followup_task`, `wait_agent`, - `interrupt_agent`, and `list_agents`. -- V1 exposes legacy handlers such as `spawn_agent`, `send_input`, `resume_agent`, - `wait_agent`, and `close_agent`. -- `disabled` suppresses collaboration tools. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:343 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:347 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:766 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:769 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:819 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/handlers/multi_agents_spec.rs:48 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/handlers/multi_agents_spec.rs:80 -``` - -## Spawn-Agent Model List - -`tool_mode` does not decide which models appear in `spawn_agent` descriptions. Spawn-agent model -override descriptions come from available `ModelPreset`s sorted by priority, filtered for picker -visibility, and capped at five entries. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/models-manager/src/manager.rs:80 -/tmp/opencodex-codex-src/codex-rs/models-manager/src/manager.rs:117 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/handlers/multi_agents_spec.rs:19 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/handlers/multi_agents_spec.rs:747 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/handlers/multi_agents_spec.rs:751 -``` - -## Current opencodex Behavior - -opencodex derives routed entries by deep-copying a native template: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:82 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:108 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:118 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:128 -``` - -Live `/v1/models?client_version=...` and on-disk catalog sync both use the same derived entries: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:473 -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:480 -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:484 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:273 -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:292 -``` - -The routed entries currently do not normalize: - -- `tool_mode` -- `multi_agent_version` - -## Gap - -If the chosen native template has `tool_mode = "code_mode"` or `"code_mode_only"`, every routed -model inherits that tool-exposure behavior. - -If the template has `multi_agent_version = "v2"` or `"disabled"`, every routed model inherits that -collaboration behavior. A model marked `v2` can expose V2 tools even when the feature flag is off. -A model marked `disabled` can suppress collaboration even when the feature is on. - -Because Codex stores resolved multi-agent version on the first real turn, metadata fixes may require -a new Codex session to take effect. - -## Phase 100 Recommendation - -Normalize both fields deliberately in `deriveEntry()`: - -1. For native OpenAI passthrough entries, preserve native `tool_mode` and `multi_agent_version`. -2. For non-OpenAI routed entries, either delete both fields and let Codex feature defaults apply, or - set a project-wide explicit policy. -3. Prefer deletion first unless opencodex has a strong reason to force code-mode or V2 for routed - models. Deletion does not disable the user's configured/default multi-agent behavior; it lets - Codex resolve the feature default (including V2 when the user has enabled V2) instead of letting - the cloned native template force a selector. -4. Add catalog snapshot tests proving routed entries do not inherit these selectors accidentally. diff --git a/devlog/_fin/100_codex-native-parity/40_responses-lite-websockets.md b/devlog/_fin/100_codex-native-parity/40_responses-lite-websockets.md deleted file mode 100644 index 9750f933d..000000000 --- a/devlog/_fin/100_codex-native-parity/40_responses-lite-websockets.md +++ /dev/null @@ -1,126 +0,0 @@ -# 100.40 — Responses Lite and Websockets - -## Questions - -- What is `use_responses_lite`? -- What is `supports_websockets`? -- Can opencodex gain speed by setting `supports_websockets = true` even though most upstreams are - Chat Completions? - -## Codex RS Behavior: `use_responses_lite` - -`use_responses_lite` is model metadata. It defaults to false. - -When true, Codex changes outgoing Responses behavior: - -- strips image `detail` fields; -- sets reasoning `context = all_turns`; -- disables `parallel_tool_calls`; -- sends `x-openai-internal-codex-responses-lite: true` on HTTP/compact paths; -- sends websocket client metadata `responses_lite = true` on WS paths; -- suppresses hosted Responses tools and exposes standalone image-generation / web-search tools. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:408 -/tmp/opencodex-codex-src/codex-rs/models-manager/src/model_info.rs:68 -/tmp/opencodex-codex-src/codex-rs/core/src/client_common.rs:52 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:700 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:759 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:811 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:840 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:1644 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:1746 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:291 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:397 -/tmp/opencodex-codex-src/codex-rs/core/src/tools/spec_plan.rs:618 -``` - -## Codex RS Behavior: `supports_websockets` - -`supports_websockets` is provider metadata. It defaults to false. Codex chooses the websocket path -only when all of these are true: - -- `wire_api == Responses` -- provider `supports_websockets == true` -- no active fallback flag prevents websocket use - -The built-in OpenAI provider sets `supports_websockets = true`. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/model-provider-info/src/lib.rs:134 -/tmp/opencodex-codex-src/codex-rs/model-provider-info/src/lib.rs:324 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/endpoint/responses_websocket.rs:334 -``` - -## Current opencodex Behavior - -opencodex currently injects a Responses provider, but it does not inject -`supports_websockets = true`. - -That means Codex should stay on the HTTP/SSE Responses path for opencodex. - -For routed non-OpenAI providers, opencodex usually translates Codex Responses requests into -provider-specific HTTP APIs, frequently OpenAI-compatible Chat Completions streaming. There is no -end-to-end Responses websocket proxy today. - -## Gap - -If routed catalog entries inherit `use_responses_lite = true` from a future native template, Codex -could change request semantics in ways opencodex did not deliberately implement: - -- altered image detail handling; -- altered reasoning context; -- disabled parallel tool calls; -- standalone hosted-tool assumptions. - -If opencodex sets `supports_websockets = true` without implementing the protocol, Codex may attempt -WS and incur fallback overhead or fail with a protocol mismatch. - -## Speed Assessment - -Setting `supports_websockets = true` is not expected to materially improve speed for most routed -models today. The upstream provider path is still HTTP/SSE Chat Completions or another HTTP stream, -so a Codex-to-opencodex websocket would only change the first hop while opencodex still waits on the -same provider stream. - -A real speed benefit is only plausible for: - -1. native OpenAI Responses passthrough with an end-to-end Responses websocket proxy; or -2. providers that themselves expose a low-latency websocket protocol opencodex can bridge without - converting back to HTTP/SSE internally. - -## Decision: No Phase 100 Websocket Work - -Phase 100 should not include a websocket spike. - -Reason: - -- opencodex's routed upstreams are mostly HTTP/SSE Chat Completions or HTTP/SSE-compatible streams; -- enabling websocket only between Codex and opencodex cannot make a provider/model websocket-capable - when the upstream provider is not websocket-capable end-to-end; -- a websocket first hop would still block on the same upstream SSE chunks, so it is unlikely to - improve first-token latency or throughput; -- setting `supports_websockets = true` would advertise a native capability opencodex does not - actually provide for routed models. - -The stable policy is therefore: - -```text -routed providers -> supports_websockets absent/false -> HTTP/SSE path only -``` - -If a future provider exposes a real websocket-native protocol, that should be designed as a separate -provider-specific transport project, not Phase 100 Codex-native parity work. - -## Phase 100 Recommendation - -1. Do not set `supports_websockets = true` for opencodex routed providers. -2. Strip or explicitly set `use_responses_lite = false` for routed non-OpenAI models unless - opencodex implements the related request-shape differences. -3. Preserve `use_responses_lite` only for native OpenAI passthrough models if Codex's native template - requires it. -4. Add a regression test that routed entries cannot inherit `use_responses_lite` silently. diff --git a/devlog/_fin/100_codex-native-parity/41_responses-lite-policy.md b/devlog/_fin/100_codex-native-parity/41_responses-lite-policy.md deleted file mode 100644 index 3e2b2e218..000000000 --- a/devlog/_fin/100_codex-native-parity/41_responses-lite-policy.md +++ /dev/null @@ -1,84 +0,0 @@ -# 100.41 — Responses Lite Policy - -## Decision - -For routed non-OpenAI models: - -```text -delete use_responses_lite -``` - -or explicitly set: - -```text -use_responses_lite = false -``` - -Either is acceptable. Deleting is cleaner because Codex's default is false. - -## Why - -`use_responses_lite` is not just a UI flag. When true, Codex changes the outgoing request shape: - -- strips image detail fields; -- sets reasoning context differently; -- disables parallel tool calls; -- sends a Responses Lite internal header on HTTP paths; -- sends `responses_lite = true` metadata on websocket paths; -- changes hosted tool exposure. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:408 -/tmp/opencodex-codex-src/codex-rs/models-manager/src/model_info.rs:68 -/tmp/opencodex-codex-src/codex-rs/core/src/client_common.rs:52 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:700 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:759 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:811 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:840 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:1644 -/tmp/opencodex-codex-src/codex-rs/core/src/client.rs:1746 -``` - -## Current Observation - -The current local opencodex catalog shows: - -```text -use_responses_lite: false -``` - -for both native and routed sample entries in: - -```text -/Users/jun/.codex/opencodex-catalog.json -``` - -That is currently safe, but still inherited. The Phase 100 policy is to make the safe value -deliberate instead of accidental. - -## Websocket Policy - -Do not set provider-level: - -```text -supports_websockets = true -``` - -for opencodex until there is an end-to-end Responses websocket proxy. - -Current routed providers mostly end in HTTP/SSE provider APIs. A websocket first hop from Codex to -opencodex would not remove the upstream HTTP/SSE bottleneck and can introduce fallback or protocol -mismatch risk. - -## Implementation Rule - -In routed-entry normalization: - -```text -if slug includes "/": - delete use_responses_lite -``` - -Keep native OpenAI passthrough untouched. diff --git a/devlog/_fin/100_codex-native-parity/50_streaming-thinking-context.md b/devlog/_fin/100_codex-native-parity/50_streaming-thinking-context.md deleted file mode 100644 index 0494f7fd2..000000000 --- a/devlog/_fin/100_codex-native-parity/50_streaming-thinking-context.md +++ /dev/null @@ -1,246 +0,0 @@ -# 100.50 — Streaming, Thinking, and Context Metadata - -## Questions - -- Is intermediate streamed response text faithful? -- Are thinking/reasoning blocks represented correctly? -- Is context-window and token accounting metadata complete enough for Codex? - -## Intermediate Response Text - -For streamed translated adapters, opencodex is reasonably faithful. - -The routed streaming path is: - -```text -adapter.parseStream(...) -> bridgeToResponsesSSE(...) -``` - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:193 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:54 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:60 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:142 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:155 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:217 -``` - -The bridge emits the core Responses SSE sequence Codex expects: - -- `response.created` -- `response.output_item.added` -- `response.content_part.added` -- `response.output_text.delta` -- `response.output_text.done` -- `response.content_part.done` -- `response.output_item.done` -- `response.completed` - -Upstream Codex parses these events in: - -```text -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:302 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:310 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:342 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:393 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:411 -``` - -OpenAI/Azure Responses passthrough has the best stream fidelity because opencodex forwards the -upstream Responses body and sanitized headers directly. - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-responses.ts:31 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/azure.ts:5 -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:141 -``` - -## Thinking / Reasoning Blocks - -opencodex normalized stream events include: - -```text -thinking_delta -``` - -Relevant local path: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:149 -``` - -The bridge currently emits provider thinking as reasoning summaries: - -- `response.output_item.added` with `type: "reasoning"` -- `response.reasoning_summary_part.added` -- `response.reasoning_summary_text.delta` -- summary done events - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:162 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:165 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:169 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:175 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:81 -``` - -Upstream Codex distinguishes summary reasoning from raw reasoning content: - -```text -/tmp/opencodex-codex-src/codex-rs/codex-api/src/common.rs:101 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/common.rs:105 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:326 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:334 -``` - -Gap: opencodex never emits `response.reasoning_text.delta`, so every provider thinking stream is -presented as a summary, even when the upstream provider is returning raw reasoning-like content. - -Incoming previous-turn reasoning is parsed into local assistant `thinking` with a JSON signature: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/responses/schema.ts:42 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:240 -``` - -That is useful, but it does not round-trip provider-specific opaque reasoning metadata natively. - -## Non-Streaming Gap - -Translated non-streaming responses are lower fidelity. `buildResponseJSON()` only accumulates text -and usage, then emits a message item if text exists. - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:216 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:260 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:269 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:274 -``` - -Known examples: - -- Anthropic streaming maps thinking deltas, but non-streaming currently handles text/tool-use only. -- OpenAI-compatible streaming maps `delta.reasoning_content`, but non-streaming handles - `message.content` and `tool_calls` only. -- Google maps text and function calls, with no equivalent thinking channel today. - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/anthropic.ts:233 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/anthropic.ts:283 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts:202 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts:237 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts:137 -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts:170 -``` - -## Usage and Context Metadata - -Current local usage type has only: - -```text -inputTokens -outputTokens -``` - -Relevant local paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:158 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:221 -``` - -Upstream Codex can consume richer usage: - -- `input_tokens` -- `input_tokens_details.cached_tokens` -- `output_tokens` -- `output_tokens_details.reasoning_tokens` -- `total_tokens` - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:100 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:119 -/tmp/opencodex-codex-src/codex-rs/protocol/src/protocol.rs:1999 -``` - -Gap: translated streams report cached/reasoning token counts as zero or absent, affecting status UI, -analytics, and context-budget behavior. - -Codex model metadata also includes context-window fields: - -- `context_window` -- `max_context_window` -- `auto_compact_token_limit` -- `effective_context_window_percent` -- `truncation_policy` - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:346 -/tmp/opencodex-codex-src/codex-rs/protocol/src/openai_models.rs:428 -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:3421 -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:3457 -/tmp/opencodex-codex-src/codex-rs/core/src/session/mod.rs:3529 -/tmp/opencodex-codex-src/codex-rs/protocol/src/protocol.rs:2013 -``` - -opencodex does not currently set provider/model-specific context-window fields for routed catalog -entries. Routed models either inherit native template limits or omit them in fallback mode. - -## Response Header and Error Gaps - -Upstream Codex can derive events from headers before SSE processing: - -- server model; -- rate limits; -- model etag; -- server reasoning included. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:31 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/common.rs:77 -``` - -Translated opencodex streams set minimal SSE headers only: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:206 -``` - -Error fidelity is also incomplete. opencodex emits `response.failed` with `last_error`, while the -upstream parser evidence suggests typed classification reads `response.error`. - -Relevant paths: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:231 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:347 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:350 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:382 -``` - -## Phase 100 Recommendation - -1. Add provider/model-specific catalog metadata for context windows and truncation behavior. -2. Extend `OcxUsage` and bridge output to include cached input tokens and reasoning output tokens. -3. Decide whether each provider's thinking stream should map to Codex reasoning summary or raw - reasoning text. -4. Enforce parsed `reasoning.summary = "none"` when building the stream. -5. Either improve translated non-streaming parity or explicitly document it as lower fidelity. -6. Synthesize or forward Codex-relevant headers where possible. -7. Align translated `response.failed` shape with upstream parser expectations. diff --git a/devlog/_fin/100_codex-native-parity/51_raw-reasoning-bridge.md b/devlog/_fin/100_codex-native-parity/51_raw-reasoning-bridge.md deleted file mode 100644 index 8cce62ed0..000000000 --- a/devlog/_fin/100_codex-native-parity/51_raw-reasoning-bridge.md +++ /dev/null @@ -1,175 +0,0 @@ -# 100.51 — Raw Reasoning Bridge - -## Question - -Can Codex RS accept raw reasoning text, or only reasoning summaries? - -## Answer - -Codex RS can accept raw reasoning text through: - -```text -response.reasoning_text.delta -``` - -This is separate from the summary path: - -```text -response.reasoning_summary_text.delta -``` - -## Required Raw Event Shape - -For raw reasoning, the SSE event needs: - -```json -{ - "type": "response.reasoning_text.delta", - "delta": "raw detail", - "content_index": 0 -} -``` - -Codex parses this into `ResponseEvent::ReasoningContentDelta`, then into raw reasoning app/server -events. - -Relevant upstream paths: - -```text -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:334 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:335 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:336 -/tmp/opencodex-codex-src/codex-rs/core/src/session/turn.rs:2310 -/tmp/opencodex-codex-src/codex-rs/core/src/session/turn.rs:2325 -/tmp/opencodex-codex-src/codex-rs/app-server-protocol/src/protocol/event_mapping.rs:384 -/tmp/opencodex-codex-src/codex-rs/app-server-protocol/src/protocol/event_mapping.rs:389 -/tmp/opencodex-codex-src/codex-rs/app-server-protocol/src/protocol/event_mapping.rs:390 -``` - -Upstream fixture helper: - -```text -/tmp/opencodex-codex-src/codex-rs/core/tests/common/responses.rs:766 -/tmp/opencodex-codex-src/codex-rs/core/tests/common/responses.rs:768 -/tmp/opencodex-codex-src/codex-rs/core/tests/common/responses.rs:769 -/tmp/opencodex-codex-src/codex-rs/core/tests/common/responses.rs:770 -``` - -## Required Sequence - -Raw reasoning must arrive while a reasoning output item is active: - -1. `response.created` -2. `response.output_item.added` with `type: "reasoning"` -3. `response.reasoning_text.delta` -4. `response.output_item.done` with final reasoning item containing raw content -5. `response.completed` - -Relevant upstream test source: - -```text -/tmp/opencodex-codex-src/codex-rs/core/tests/suite/items.rs:1150 -/tmp/opencodex-codex-src/codex-rs/core/tests/suite/items.rs:1152 -/tmp/opencodex-codex-src/codex-rs/core/tests/suite/items.rs:1153 -/tmp/opencodex-codex-src/codex-rs/core/tests/suite/items.rs:1154 -/tmp/opencodex-codex-src/codex-rs/core/tests/suite/items.rs:1155 -``` - -Final raw reasoning item shape: - -```json -{ - "type": "reasoning", - "id": "reasoning-raw", - "summary": [], - "content": [ - { "type": "reasoning_text", "text": "raw detail" } - ] -} -``` - -The model supports both `summary` and raw `content`: - -```text -/tmp/opencodex-codex-src/codex-rs/protocol/src/models.rs:947 -/tmp/opencodex-codex-src/codex-rs/protocol/src/models.rs:951 -/tmp/opencodex-codex-src/codex-rs/protocol/src/models.rs:954 -/tmp/opencodex-codex-src/codex-rs/protocol/src/models.rs:1604 -/tmp/opencodex-codex-src/codex-rs/protocol/src/models.rs:1605 -``` - -## Current opencodex State - -opencodex currently maps every provider `thinking_delta` into the summary path: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:167 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:169 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:176 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:83 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:86 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:90 -``` - -Current adapter event type: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:151 -``` - -```ts -| { type: "thinking_delta"; thinking: string } -``` - -Incoming historical reasoning is already parsed from either summary or raw content: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/responses/schema.ts:21 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/schema.ts:22 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/schema.ts:42 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/schema.ts:45 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/schema.ts:46 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:242 -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts:243 -``` - -So the missing piece is outbound bridge mode, not inbound schema awareness. - -## Test Status - -A targeted upstream runtime test was attempted by the research agent: - -```bash -cargo test -p codex-core --test all reasoning_raw_content_delta_respects_flag -- --nocapture -``` - -It selected the intended test but aborted with a stack overflow before assertion result: - -```text -thread 'tokio-rt-worker' has overflowed its stack -fatal runtime error: stack overflow, aborting -``` - -Therefore this document treats raw reasoning support as source-backed and fixture-backed, not as a -fresh passing runtime verification. - -## Phase 100 Recommendation - -Do not just rename `response.reasoning_summary_text.delta` to `response.reasoning_text.delta`. - -Add an explicit second bridge mode: - -```ts -| { type: "reasoning_raw_delta"; text: string } -``` - -or add source classification to `thinking_delta`. - -Then emit: - -- summary providers through `response.reasoning_summary_text.delta`; -- raw providers through `response.reasoning_text.delta`; -- final reasoning items with `content: [{ type: "reasoning_text", text }]` for raw mode. - -Provider mapping should come from adapter knowledge or jawcode metadata such as -`compat.reasoningContentField` / `thinkingFormat`, not from a global assumption. diff --git a/devlog/_fin/100_codex-native-parity/52_error-header-fidelity-implementation-plan.md b/devlog/_fin/100_codex-native-parity/52_error-header-fidelity-implementation-plan.md deleted file mode 100644 index 4cd6e9180..000000000 --- a/devlog/_fin/100_codex-native-parity/52_error-header-fidelity-implementation-plan.md +++ /dev/null @@ -1,302 +0,0 @@ -# 100.52 — Error and Header Fidelity Implementation Plan - -## Objective - -Phase 100.5 makes translated opencodex failures look more like native Codex Responses failures. - -Codex RS classifies streaming failures from: - -```text -response.failed.response.error.code -``` - -Current opencodex translated streams only emit: - -```text -response.failed.response.last_error -``` - -That makes context-window, quota, and rate-limit failures look like generic stream errors. This phase -adds a shared error classifier, emits both `error` and `last_error` for streaming compatibility, and -tightens passthrough header sanitization without fabricating rate-limit headers. - -## Evidence - -Upstream Codex parser evidence: - -```text -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:347 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:532 -/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs:557 -``` - -Local gap: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts -``` - -## Files - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/errors.ts -``` - -Complete content: - -```ts -export interface OcxErrorPayload { - message: string; - type: string; - code: string | null; -} - -export function classifyError(status: number, type: string, message: string): OcxErrorPayload { - const text = message.toLowerCase(); - if (text.includes("context_length_exceeded") || text.includes("context window") || text.includes("context length") || text.includes("maximum context") || text.includes("too many tokens")) { - return { message, type: "invalid_request_error", code: "context_length_exceeded" }; - } - if (text.includes("insufficient_quota") || text.includes("quota exceeded") || text.includes("exceeded your current quota")) { - return { message, type: "insufficient_quota", code: "insufficient_quota" }; - } - if (status === 429 || text.includes("rate limit") || text.includes("too many requests")) { - return { message, type: "rate_limit_error", code: "rate_limit_exceeded" }; - } - if (status === 401 || status === 403 || type === "authentication_error") { - return { message, type: "authentication_error", code: "invalid_api_key" }; - } - if (status >= 500) { - return { message, type: "server_error", code: "upstream_server_error" }; - } - if (status === 400 || type === "invalid_request_error") { - return { message, type: "invalid_request_error", code: "invalid_request_error" }; - } - return { message, type, code: type || null }; -} -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -``` - -Import classifier: - -```diff - import type { AdapterEvent, OcxUsage } from "./types"; -+import { classifyError } from "./errors"; -``` - -Add a helper near `responsesUsage()`: - -```diff -+function responseError(status: number, type: string, message: string): Record { -+ return classifyError(status, type, message); -+} -``` - -Update adapter error event: - -```diff - emit("response.failed", { - response: { - ...responseSnapshot("failed", finishedItems), -- last_error: { type: "upstream_error", message: event.message }, -+ error: responseError(502, "upstream_error", event.message), -+ last_error: responseError(502, "upstream_error", event.message), - }, - }); -``` - -Update caught bridge exception: - -```diff - emit("response.failed", { - response: { - ...responseSnapshot("failed", finishedItems), -- last_error: { type: "proxy_error", message: err instanceof Error ? err.message : String(err) }, -+ error: responseError(500, "proxy_error", err instanceof Error ? err.message : String(err)), -+ last_error: responseError(500, "proxy_error", err instanceof Error ? err.message : String(err)), - }, - }); -``` - -Update JSON error formatter: - -```diff --export function formatErrorResponse(status: number, type: string, message: string): Response { -- return new Response(JSON.stringify({ error: { message, type, code: null } }), { -+export function formatErrorResponse(status: number, type: string, message: string): Response { -+ return new Response(JSON.stringify({ error: classifyError(status, type, message) }), { - status, headers: { "Content-Type": "application/json" }, - }); - } -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts -``` - -Expand hop-by-hop/stale header drops: - -```diff -- const DROP = new Set(["content-encoding", "content-length", "transfer-encoding", "connection", "keep-alive"]); -+ const DROP = new Set([ -+ "content-encoding", -+ "content-length", -+ "transfer-encoding", -+ "connection", -+ "keep-alive", -+ "proxy-authenticate", -+ "proxy-authorization", -+ "te", -+ "trailer", -+ "upgrade", -+ ]); -``` - -This keeps truthful upstream headers such as `x-ratelimit-*`, `openai-*`, `request-id`, -`content-type`, and model/version headers. It does not synthesize rate-limit headers because -opencodex does not have complete upstream quota telemetry for translated providers. - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/error-fidelity.test.ts -``` - -Complete content: - -```ts -import { describe, expect, test } from "bun:test"; -import { bridgeToResponsesSSE, formatErrorResponse } from "../src/bridge"; -import { classifyError } from "../src/errors"; -import { sanitizePassthroughHeaders } from "../src/server"; -import type { AdapterEvent } from "../src/types"; - -async function* replay(events: AdapterEvent[]): AsyncGenerator { - for (const event of events) yield event; -} - -async function collectSse(stream: ReadableStream): Promise<{ event?: string; data: Record }[]> { - const reader = stream.getReader(); - const decoder = new TextDecoder(); - let text = ""; - while (true) { - const { done, value } = await reader.read(); - if (done) break; - text += decoder.decode(value, { stream: true }); - } - return text.split("\n\n") - .map(frame => frame.trim()) - .filter(frame => frame.length > 0 && frame !== "data: [DONE]") - .map(frame => { - const lines = frame.split("\n"); - const event = lines.find(line => line.startsWith("event: "))?.slice(7); - const dataLine = lines.find(line => line.startsWith("data: ")); - return { event, data: JSON.parse(dataLine?.slice(6) ?? "{}") as Record }; - }); -} - -describe("error fidelity", () => { - test("classifyError maps Codex-recognized context/quota/rate failures", () => { - expect(classifyError(400, "upstream_error", "Your input exceeds the context window")).toMatchObject({ - type: "invalid_request_error", - code: "context_length_exceeded", - }); - expect(classifyError(429, "upstream_error", "Rate limit reached for model")).toMatchObject({ - type: "rate_limit_error", - code: "rate_limit_exceeded", - }); - expect(classifyError(402, "upstream_error", "You exceeded your current quota")).toMatchObject({ - type: "insufficient_quota", - code: "insufficient_quota", - }); - }); - - test("formatErrorResponse returns OpenAI-compatible classified error envelope", async () => { - const response = formatErrorResponse(429, "upstream_error", "Rate limit reached for model"); - expect(response.status).toBe(429); - await expect(response.json()).resolves.toEqual({ - error: { - message: "Rate limit reached for model", - type: "rate_limit_error", - code: "rate_limit_exceeded", - }, - }); - }); - - test("streaming response.failed includes both error and last_error", async () => { - const frames = await collectSse(bridgeToResponsesSSE(replay([ - { type: "error", message: "Your input exceeds the context window" }, - ]), "routed/model")); - const failed = frames.find(frame => frame.event === "response.failed")?.data.response as Record; - expect(failed.error).toMatchObject({ - type: "invalid_request_error", - code: "context_length_exceeded", - }); - expect(failed.last_error).toEqual(failed.error); - }); - - test("sanitizePassthroughHeaders drops stale and hop-by-hop headers while preserving rate-limit metadata", () => { - const sanitized = sanitizePassthroughHeaders(new Headers({ - "content-encoding": "gzip", - "content-length": "12", - "connection": "keep-alive", - "keep-alive": "timeout=5", - "proxy-authenticate": "Basic", - "te": "trailers", - "trailer": "x-checksum", - "upgrade": "websocket", - "x-ratelimit-limit-requests": "100", - "openai-model": "gpt-5.5", - "content-type": "application/json", - })); - expect(sanitized.has("content-encoding")).toBe(false); - expect(sanitized.has("content-length")).toBe(false); - expect(sanitized.has("connection")).toBe(false); - expect(sanitized.has("keep-alive")).toBe(false); - expect(sanitized.has("proxy-authenticate")).toBe(false); - expect(sanitized.has("te")).toBe(false); - expect(sanitized.has("trailer")).toBe(false); - expect(sanitized.has("upgrade")).toBe(false); - expect(sanitized.get("x-ratelimit-limit-requests")).toBe("100"); - expect(sanitized.get("openai-model")).toBe("gpt-5.5"); - expect(sanitized.get("content-type")).toBe("application/json"); - }); -}); -``` - -## Verification - -Run: - -```bash -bun test tests/error-fidelity.test.ts tests/bridge.test.ts -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Expected result: - -```text -error fidelity tests pass -full test suite passes -typecheck passes -diff whitespace check passes -``` - -## Commit - -Commit as: - -```text -[agent] fix: align error and header fidelity -``` diff --git a/devlog/_fin/100_codex-native-parity/53_e2e-style-verification-plan.md b/devlog/_fin/100_codex-native-parity/53_e2e-style-verification-plan.md deleted file mode 100644 index bdcc464ce..000000000 --- a/devlog/_fin/100_codex-native-parity/53_e2e-style-verification-plan.md +++ /dev/null @@ -1,180 +0,0 @@ -# 100.53 — E2E-Style Verification Gap Closure Plan - -## Objective - -Close the final review gap: Phase 100 has unit/integration coverage, but no explicit -`tests/e2e-style/` artifact. Add a local e2e-style smoke test that exercises the Codex-native parity -path across catalog metadata, Responses request parsing, web-search sidecar planning, and bridge -failure output. - -No runtime source changes are planned. - -## Files - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/e2e-style/phase100-native-parity.test.ts -``` - -Complete content: - -```ts -import { describe, expect, test } from "bun:test"; -import { bridgeToResponsesSSE } from "../../src/bridge"; -import { buildCatalogEntries } from "../../src/codex-catalog"; -import { parseRequest } from "../../src/responses/parser"; -import { planWebSearch } from "../../src/web-search"; -import type { AdapterEvent, OcxConfig, OcxProviderConfig } from "../../src/types"; - -function nativeTemplate(): Record { - return { - slug: "gpt-5.5", - display_name: "gpt-5.5", - description: "Native GPT model", - priority: 1, - visibility: "list", - base_instructions: "You are Codex, a coding agent based on GPT-5.", - model_messages: { instructions_template: "You are Codex, a coding agent based on GPT-5." }, - tool_mode: "code", - multi_agent_version: "v2", - use_responses_lite: true, - supports_websockets: true, - web_search_tool_type: "text_and_image", - supports_search_tool: true, - supported_reasoning_levels: [ - { effort: "low", description: "native low" }, - { effort: "medium", description: "native medium" }, - { effort: "high", description: "native high" }, - { effort: "xhigh", description: "native xhigh" }, - ], - }; -} - -async function* replay(events: AdapterEvent[]): AsyncGenerator { - for (const event of events) yield event; -} - -async function collectSse(stream: ReadableStream): Promise<{ event?: string; data: Record }[]> { - const reader = stream.getReader(); - const decoder = new TextDecoder(); - let text = ""; - while (true) { - const { done, value } = await reader.read(); - if (done) break; - text += decoder.decode(value, { stream: true }); - } - return text.split("\n\n") - .map(frame => frame.trim()) - .filter(frame => frame.length > 0 && frame !== "data: [DONE]") - .map(frame => { - const lines = frame.split("\n"); - const event = lines.find(line => line.startsWith("event: "))?.slice(7); - const dataLine = lines.find(line => line.startsWith("data: ")); - return { event, data: JSON.parse(dataLine?.slice(6) ?? "{}") as Record }; - }); -} - -describe("Phase 100 Codex-native parity smoke", () => { - test("routed model keeps native-like catalog affordances while runtime routes through opencodex sidecars and bridge errors", async () => { - const routedProvider: OcxProviderConfig = { - adapter: "openai-chat", - baseUrl: "https://routed.example/v1", - apiKey: "routed-key", - noVisionModels: ["deepseek-v4-pro"], - }; - const forwardProvider: OcxProviderConfig = { - adapter: "openai-responses", - baseUrl: "https://chatgpt.example/v1", - authMode: "forward", - }; - const config: OcxConfig = { - port: 10100, - defaultProvider: "opencode-go", - providers: { - "opencode-go": routedProvider, - chatgpt: forwardProvider, - }, - }; - - const catalog = buildCatalogEntries(nativeTemplate(), ["gpt-5.5"], [ - { provider: "opencode-go", id: "deepseek-v4-pro" }, - ]); - const routed = catalog.find(entry => entry.slug === "opencode-go/deepseek-v4-pro"); - expect(routed).toMatchObject({ - web_search_tool_type: "text_and_image", - supports_search_tool: true, - context_window: 1_000_000, - auto_compact_token_limit: 900_000, - }); - expect(routed).not.toHaveProperty("model_messages"); - expect(routed).not.toHaveProperty("use_responses_lite"); - expect(routed).not.toHaveProperty("supports_websockets"); - - const parsed = parseRequest({ - model: "opencode-go/deepseek-v4-pro", - stream: true, - input: "Search current docs, then answer.", - tools: [ - { type: "web_search", search_context_size: "medium" }, - { type: "tool_search", description: "Load extra tools" }, - ], - }); - expect(parsed._webSearch).toMatchObject({ type: "web_search", search_context_size: "medium" }); - expect(parsed.context.tools?.some(tool => tool.toolSearch)).toBe(true); - - const searchPlan = planWebSearch( - config, - parsed, - false, - new Headers({ authorization: "Bearer forwarded-chatgpt" }), - routedProvider, - "deepseek-v4-pro", - ); - expect(searchPlan).toMatchObject({ - forwardProvider, - settings: { - model: "gpt-5.4-mini", - describeImages: true, - }, - }); - - const frames = await collectSse(bridgeToResponsesSSE(replay([ - { type: "error", message: "Your input exceeds the context window" }, - ]), "deepseek-v4-pro")); - const failed = frames.find(frame => frame.event === "response.failed")?.data.response as Record; - expect(failed.error).toMatchObject({ - code: "context_length_exceeded", - type: "invalid_request_error", - }); - }); -}); -``` - -## Verification - -Run: - -```bash -bun test tests/e2e-style/phase100-native-parity.test.ts -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Expected result: - -```text -e2e-style smoke test passes -full test suite passes -typecheck passes -diff whitespace check passes -``` - -## Commit - -Commit as: - -```text -[agent] test: add phase 100 parity smoke -``` diff --git a/devlog/_fin/100_codex-native-parity/60_jawcode-metadata-snapshot.md b/devlog/_fin/100_codex-native-parity/60_jawcode-metadata-snapshot.md deleted file mode 100644 index 728cb0e9e..000000000 --- a/devlog/_fin/100_codex-native-parity/60_jawcode-metadata-snapshot.md +++ /dev/null @@ -1,287 +0,0 @@ -# 100.60 — jawcode Metadata Snapshot Plan - -## Question - -Can opencodex dynamically reuse jawcode's existing model defaults for context windows, capabilities, -and usage metadata? - -## Short Answer - -Yes, jawcode has richer metadata, but opencodex should not import it directly at runtime first. - -Recommended approach: - -```text -jawcode rich metadata -> build-time generated opencodex snapshot -> small opencodex projection -> verified Codex catalog fields -``` - -Do not do: - -```text -dynamic import @jawcode-dev/ai at opencodex runtime -> spread jawcode Model into Codex catalog -``` - -## jawcode Sources - -Primary static model registry: - -```text -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.json -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.ts -``` - -`models.json` includes `contextWindow`, `maxTokens`, `reasoning`, `thinking`, `input`, `output`, -`cost`, `compat`, `wireModelId`, and `unlisted`. - -Relevant jawcode paths: - -```text -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.ts:22 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.ts:30 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.ts:34 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.ts:41 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.ts:50 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:874 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:898 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:899 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:936 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:938 -``` - -Provider descriptors: - -```text -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/descriptors.ts:48 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/descriptors.ts:60 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/descriptors.ts:64 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/descriptors.ts:128 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/descriptors.ts:296 -``` - -jawcode structure docs summarize the catalog fields: - -```text -/Users/jun/Developer/new/700_projects/jawcode/structure/30_providers.md:149 -/Users/jun/Developer/new/700_projects/jawcode/structure/30_providers.md:151 -/Users/jun/Developer/new/700_projects/jawcode/structure/30_providers.md:152 -/Users/jun/Developer/new/700_projects/jawcode/structure/30_providers.md:156 -``` - -## Current opencodex Shape - -opencodex currently carries routed catalog models as: - -```ts -export interface CatalogModel { - id: string; - provider: string; - owned_by?: string; -} -``` - -Path: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:34 -``` - -Provider config is also narrower: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:204 -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:208 -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:209 -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:222 -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:227 -``` - -## Direct Runtime Dependency Risk - -`@jawcode-dev/ai` exports useful pieces: - -```text -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/package.json:87 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/package.json:91 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/package.json:115 -``` - -But runtime importing it into opencodex is risky: - -- opencodex currently keeps runtime deps small; -- jawcode's package has a higher Bun floor than opencodex; -- it may pull in provider SDK and agent-specific surfaces; -- opencodex is meant to be a small proxy, not a full jawcode runtime; -- failures would have to degrade cleanly on every user install. - -Current opencodex dependency baseline: - -```text -/Users/jun/Developer/new/700_projects/opencodex/package.json:18 -/Users/jun/Developer/new/700_projects/opencodex/package.json:31 -``` - -## Recommended Projection - -Generate a small opencodex-owned metadata file, for example: - -```ts -export interface OcxModelMetadata { - provider: string; - id: string; - contextWindow?: number; - maxTokens?: number; - input?: ("text" | "image")[]; - reasoning?: boolean; - thinking?: unknown; - compat?: { - supportsReasoningEffort?: boolean; - reasoningContentField?: "reasoning_content" | "reasoning" | "reasoning_text"; - supportsUsageInStreaming?: boolean; - }; - wireModelId?: string; - unlisted?: boolean; -} -``` - -Generated source candidates: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/generated/jawcode-model-metadata.ts -/Users/jun/Developer/new/700_projects/opencodex/scripts/generate-jawcode-metadata.ts -``` - -Do not write these in this docs phase; this is the Phase 100 implementation target. - -## Provider-ID Mapping - -opencodex provider ids do not always match jawcode provider ids. The snapshot generator needs an -explicit mapping layer. - -Examples: - -| opencodex provider | possible jawcode source | -| --- | --- | -| `xai` | `xai` | -| `anthropic` | `anthropic` | -| `google` / `gemini` | Google descriptor ids | -| `moonshot` / `kimi` | Kimi/Moonshot descriptor ids | -| `openrouter` | `openrouter` | -| `alibaba` | `alibaba-coding-plan` or other Alibaba descriptor ids | -| `opencode-go` | `opencode-go` | - -Correction: `opencode-go` is already in jawcode. It is not an unmapped custom-only source. - -Evidence: - -```text -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:130 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/descriptors.ts:178 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:859 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:862 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:2112 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:2338 -``` - -jawcode models it as an OpenAI-compatible provider using: - -```text -https://opencode.ai/zen/go/v1 -``` - -with explicit API-resolution overrides for known OpenCode Go routing mismatches. The Phase 100 -metadata generator should therefore map opencodex `opencode-go` directly to jawcode `opencode-go` -and reuse jawcode's discovered context/capability metadata where exact model ids match. - -Policy: - -```text -mapped exact provider/model -> use jawcode metadata -mapped provider but unknown model -> use provider default/conservative values -unmapped provider -> keep current opencodex conservative defaults -``` - -The generator must report unmapped providers and models instead of silently guessing. - -## Context Window Mapping - -jawcode's `contextWindow` is the best candidate for Codex catalog `context_window`. - -jawcode's `maxTokens` is output-token metadata and should not be confused with prompt context. - -Important warning from jawcode OpenAI-compatible discovery: - -```text -max_prompt_tokens = prompt capacity -max_context_window_tokens = prompt + output -``` - -Do not map `max_context_window_tokens` directly to Codex `context_window`, because it can inflate -the prompt budget and break compaction thresholds. - -Relevant jawcode paths: - -```text -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:1694 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:1707 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:1742 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:1743 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts:1744 -``` - -## Usage Mapping - -jawcode has richer per-response usage: - -```text -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:493 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:495 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:497 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:499 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:501 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:503 -/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/types.ts:514 -``` - -opencodex currently has: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:158 -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:159 -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts:160 -``` - -Phase 100 implementation should extend opencodex response usage separately from catalog metadata: - -```ts -interface OcxUsage { - inputTokens: number; - outputTokens: number; - cachedInputTokens?: number; - reasoningOutputTokens?: number; -} -``` - -jawcode quota `UsageReport` is separate and should be documented for dashboard/status only, not -confused with Responses token usage. - -## First Implementation Slice - -1. Add generated jawcode metadata snapshot. -2. Extend internal `CatalogModel` with optional metadata. -3. Use metadata for opencodex runtime gates first: - - no vision if `input` lacks `image`; - - no reasoning if `reasoning` is false or compat says reasoning effort unsupported; - - reasoning raw/summary mapping from `compat.reasoningContentField` where reliable. -4. Only after parser verification, write Codex catalog fields: - - `context_window`; - - `max_context_window`; - - `auto_compact_token_limit` or related fields if the installed Codex catalog accepts them. - -## Risk Summary - -- jawcode shape is camelCase; Codex catalog shape is snake_case. -- provider ids do not always match. -- total context-window fields can be semantically wrong for prompt budget. -- direct runtime dependency is too heavy for the first pass. -- stale generated snapshots need provenance and regeneration checks. -- fast/service-tier fields must remain stripped for routed models. diff --git a/devlog/_fin/100_codex-native-parity/90_phase-plan.md b/devlog/_fin/100_codex-native-parity/90_phase-plan.md deleted file mode 100644 index df4bd06fe..000000000 --- a/devlog/_fin/100_codex-native-parity/90_phase-plan.md +++ /dev/null @@ -1,200 +0,0 @@ -# 100.90 — Phase Plan - -## Scope - -Phase 100 should convert this research into explicit catalog/runtime metadata policy for opencodex. -The goal is not to make every routed provider fully native in one pass. The goal is to prevent -accidental native-template inheritance from changing routed model behavior silently. - -## Implementation Order - -### 100.1 Catalog Selector Normalization - -Primary file: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -``` - -Tasks: - -1. Add a routed-entry normalization function after template cloning. -2. For non-OpenAI routed entries, explicitly handle: - - `model_messages` - - `tool_mode` - - `multi_agent_version` - - `use_responses_lite` - - service/speed tier fields already handled in Phase 90 -3. Preserve these fields only for native OpenAI passthrough entries. -4. Add catalog snapshot tests around generated routed entries. - -Expected first policy: - -- strip `model_messages` for routed models first; -- delete `tool_mode`; -- delete `multi_agent_version`; -- delete or force `use_responses_lite = false`; -- keep `supports_websockets` unset at provider level. - -### 100.2 Search Capability Policy - -Primary files: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts -/Users/jun/Developer/new/700_projects/opencodex/src/web-search/synthetic-tool.ts -``` - -Tasks: - -1. Decide whether routed models should expose deferred `tool_search`. -2. Set `web_search_tool_type` according to opencodex sidecar capability, not template inheritance. -3. Add tests proving hosted `web_search` is either converted to the synthetic sidecar tool or - suppressed predictably. - -### 100.3 Thinking and Usage Parity - -Primary files: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/anthropic.ts -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts -``` - -Tasks: - -1. Extend usage to include cached input and reasoning output tokens. -2. Emit nested Responses usage details. -3. Decide per adapter whether `thinking_delta` means summary or raw reasoning. -4. Honor `reasoning.summary = "none"` in stream output. -5. Add a regression fixture for streaming reasoning and final usage. - -### 100.4 Context Window Metadata - -Primary files: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -/Users/jun/Developer/new/700_projects/opencodex/src/generated/jawcode-model-metadata.ts -/Users/jun/Developer/new/700_projects/opencodex/scripts/generate-jawcode-metadata.ts -``` - -Tasks: - -1. Add a build-time/generated jawcode metadata snapshot with provider/model context metadata where - known. -2. Extend internal routed `CatalogModel` metadata before writing every field into Codex catalog JSON. -3. Populate `context_window`, `max_context_window`, `auto_compact_token_limit`, and related - metadata for routed entries. -4. Use conservative defaults when exact model limits are unknown. -5. Verify Codex token status and auto-compact behavior against at least one routed large-context - model. - -### 100.5 Error and Header Fidelity - -Primary files: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts -``` - -Tasks: - -1. Emit `response.error` in failure payloads where upstream Codex expects it. -2. Preserve `last_error` only if it is also useful for Responses compatibility. -3. Add context-window exceeded and quota/rate-limit fixtures. -4. Synthesize Codex-relevant headers only where opencodex has truthful data. - -### 100.6 Websocket Work Removed - -Phase 100 will not include a websocket spike. - -Final policy: - -```text -routed providers keep supports_websockets absent/false -``` - -Reason: - -opencodex routed models mostly end in upstream HTTP/SSE Chat Completions or HTTP/SSE-compatible -streams. Adding websocket only between Codex and opencodex does not make those upstream providers -websocket-native. It would advertise a capability the routed model path does not support end-to-end -and is not expected to materially improve speed. - -Future websocket work, if any, must be a separate provider-specific transport project for a provider -that actually exposes a websocket-native API. It is not part of Phase 100. - -## Verification Gates - -Minimum checks for the first implementation pass: - -```bash -bun x tsc --noEmit -``` - -Catalog checks: - -```bash -ocx sync -codex debug models -``` - -Expected routed-model assertions: - -- no OpenAI/GPT identity leak in active instructions; -- no `service_tiers` / `additional_speed_tiers`; -- no accidental `use_responses_lite`; -- no accidental `tool_mode` / `multi_agent_version` unless deliberately chosen; -- no `supports_websockets` provider flag; -- context-window fields are provider-appropriate or conservative. - -Runtime checks: - -- streamed text still arrives incrementally; -- thinking blocks render in the intended Codex channel; -- usage includes total/input/output and, where available, cached/reasoning details; -- web-search sidecar behavior is deterministic when prerequisites are present or absent; -- context-window errors classify as Codex-recognizable failures. - -## Open Decisions - -1. Resolved: routed models should strip `model_messages` first. Provider-safe personality templates - can be added later. -2. Should routed models inherit Codex feature-default multi-agent behavior by deleting - `multi_agent_version`, or should opencodex force a specific version? -3. Resolved in Phase 100.2/100.16: - - routed models expose deferred `tool_search` by default; - - routed hosted web-search metadata is `text_and_image` because actual hosted search runs via - native `gpt-5.4-mini` sidecar. -4. What conservative context-window default should apply when jawcode has no exact provider/model - match? -5. Resolved: Phase 100 will not implement websocket support or a websocket spike. Routed providers - keep `supports_websockets` absent/false. - -## Proposed First Build Slice - -Start with catalog normalization only. It has the highest leverage and lowest runtime risk: - -1. strip routed `model_messages`; -2. normalize `tool_mode`, `multi_agent_version`, and `use_responses_lite`; -3. preserve native OpenAI passthrough entries; -4. keep websocket disabled; -5. add catalog snapshot tests; -6. run `bun x tsc --noEmit`; -7. manually inspect `codex debug models`. - -Then add jawcode metadata snapshot support: - -1. generate a small opencodex-owned metadata projection from jawcode; -2. map provider ids explicitly; -3. enrich internal routed model metadata; -4. write only Codex-verified catalog fields. - -Streaming/context/error parity should follow after catalog semantics are stable. - -Websocket work is intentionally excluded from this phase. diff --git a/devlog/_fin/110_codex-stream-stability/00_overview.md b/devlog/_fin/110_codex-stream-stability/00_overview.md deleted file mode 100644 index 31a9a326f..000000000 --- a/devlog/_fin/110_codex-stream-stability/00_overview.md +++ /dev/null @@ -1,92 +0,0 @@ -# 110.00 — Overview: Codex Stream Errors over the opencodex Proxy - -## Symptom (user report) - -With `ocx` running, driving the **Codex CLI** through the proxy produces frequent -*stream errors*. The reporter suspected an SSE or WebSocket transport problem, asked whether -SSE multiplexing / WS would improve performance, observed that bolting WS onto the -`chat/completions` adapter "seems pointless," and asked whether the real cause is that -**Codex passthrough is not actually happening**. - -This phase reads the codebase against the upstream Codex SSE parser, identifies the root -cause, evaluates the transport question, and lays out a prioritized patch direction. This -phase ships **analysis + patch direction only** — the code changes in `30_patch-direction.md` -are a separate, approval-gated implementation phase. - -## TL;DR - -1. It is **not a transport-protocol problem**. It is an **SSE lifecycle / reliability** - problem. WebSockets and SSE multiplexing do not address any of the root causes. The - phase 100 "no WebSocket" decision stands (see `20_transport-evaluation.md`). -2. opencodex has **two response paths**, and the errors have **different causes on each**: - - **Passthrough** (native `gpt-*`): the ChatGPT backend body is relayed verbatim, so a - terminal `response.completed` cannot be dropped by opencodex. Errors here are - **header fidelity** + **no abort/cancel** (disconnect/leak). - - **Bridge** (routed models, e.g. `opencode-go/deepseek-v4-pro`): opencodex parses an - upstream chat/completions stream and **re-encodes** it into Responses SSE. Errors here - are **missing terminal event**, **idle timeout**, and **fidelity gaps**. -3. "Is passthrough happening?" — **Yes for native `gpt-*`** (default config). For **routed - models it is structurally impossible** (the upstream is chat/completions, not a - Responses-native endpoint), so opencodex *must* bridge. The fix is bridge fidelity, not - "forcing passthrough." - -## The two paths - -| | Passthrough | Bridge (translation) | -|---|---|---| -| Adapters | `openai-responses`, `azure` (`passthrough: true`) | `openai-chat`, `anthropic`, `google` | -| Trigger | `config.ts:60-76` default `openai` provider, `authMode: "forward"` | routed `provider/model` namespace → `router.ts:28-37` | -| Code | `server.ts:141-157` relays `upstreamResponse.body` + `sanitizePassthroughHeaders` | `server.ts:194,205` `adapter.parseStream()` → `bridgeToResponsesSSE()` | -| Fidelity | High — backend events relayed unchanged | Lossy — fixed event set re-emitted (`bridge.ts:38-311`) | -| `response.completed` origin | ChatGPT backend (verbatim) | Synthesized by the bridge on the `done` event | - -## What "stream error" means to Codex - -The Codex CLI consumes the proxy's SSE with a strict Rust parser. Every failure surfaces as -`ApiError::Stream(...)`. The authoritative trigger set -(`/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs`, see `10_…`): - -- `:459` **stream closed before `response.completed`** — stream ended with no terminal event -- `:466` **idle timeout waiting for SSE** — no event within `idle_timeout` -- `:454` **SSE frame decode error** — a malformed frame from the proxy -- `:349/:378` **`response.failed` with no classifiable `error`** (mitigated by phase 100.5) -- `:391` **`response.incomplete`**, `:406` **`response.completed` parse failure** - -## Root-cause summary - -| ID | Cause | Path | Codex trigger | Status | -|----|-------|------|---------------|--------| -| RC1 | Bridge ends stream with no terminal `response.completed` when an adapter returns without a `done` event | **Bridge** | `:459` | Open | -| RC2 | No `AbortSignal` on upstream fetch + no `cancel()` on the bridge stream → leak + re-throw on client disconnect | Both | leak / `:454` noise | Open | -| RC3 | No idle heartbeat during upstream stalls (slow routed providers) | **Bridge** | `:466` | Open | -| RC4 | Bridge fidelity: error envelope + dropped/malformed frames | **Bridge** | `:349`, `:454` | Partly fixed by 100.5 (`a0d4ec9`) | -| RC5 | Passthrough header fidelity (stale `content-encoding`/`content-length`) | **Passthrough** | `:454` | Mitigated by 100.5; verify | - -> **Misattribution guard:** RC1/RC3 bite the **bridge/routed** path only. Native `gpt-*` -> passthrough errors are RC2 + RC5. Do not attribute native-codex stream errors to RC1. - -## Answers to the four questions - -1. **SSE or WS problem?** SSE *lifecycle*, not transport. See `10_…`. -2. **Would SSE multiplexing / WS improve performance?** No — see `20_…`. Real wins: - keep native passthrough, fix abort, add heartbeat. -3. **Is WS on the chat/completions adapter pointless?** Yes. The upstream is HTTP/SSE; a WS - first hop still blocks on the same upstream chunks and would falsely advertise - `supports_websockets`. -4. **Is Codex passthrough broken?** Not for native `gpt-*` (it is used). For routed models - passthrough cannot exist by design; the bridge is the only option and is where the - defects live. - -## Scope & baseline - -- **In scope:** root-cause analysis, transport evaluation, prioritized patch direction. -- **Out of scope (this phase):** the actual code changes (deferred to an approval-gated - implementation phase; see `30_patch-direction.md`). -- **Baseline at authoring time:** `bun test` → 26 pass / 0 fail; `bun x tsc --noEmit` clean; - phase 100.5 error/header fidelity committed as `a0d4ec9`. - -## Documents - -- `10_root-cause-analysis.md` — Codex `ApiError::Stream` trigger table ↔ opencodex RC1–RC5 -- `20_transport-evaluation.md` — SSE multiplexing / WebSocket verdict -- `30_patch-direction.md` — P0/P1/P2 file-level patch direction + verification plan diff --git a/devlog/_fin/110_codex-stream-stability/10_root-cause-analysis.md b/devlog/_fin/110_codex-stream-stability/10_root-cause-analysis.md deleted file mode 100644 index 6c37b5118..000000000 --- a/devlog/_fin/110_codex-stream-stability/10_root-cause-analysis.md +++ /dev/null @@ -1,172 +0,0 @@ -# 110.10 — Root-Cause Analysis: Codex Stream Errors - -All line citations verified against source on the authoring date. opencodex paths are -relative to the repo root; the Codex parser is the vendored upstream at -`/tmp/opencodex-codex-src/codex-rs/codex-api/src/sse/responses.rs`. - -## 1. The Codex stream-error model (consumer side) - -Codex consumes the proxy's SSE in `process_sse` (`responses.rs:434-479`). The poll loop -defines *every* way a stream can fail — each becomes `ApiError::Stream(...)`: - -```rust -let response = timeout(idle_timeout, stream.next()).await; // :446 -match response { - Ok(Some(Ok(sse))) => sse, // event - Ok(Some(Err(e))) => { send Err(ApiError::Stream(e)); return; } // :454 frame decode - Ok(None) => { send Err(response_error // :457-460 stream end - .unwrap_or(ApiError::Stream( - "stream closed before response.completed"))); return; } - Err(_) => { send Err(ApiError::Stream( // :464-468 idle timeout - "idle timeout waiting for SSE")); return; } -} -``` - -Per-event handling (`responses.rs:347-410`) adds three more: - -| Trigger | Line | Condition | -|---------|------|-----------| -| `response.failed` w/o classifiable `error` | `:349`, `:378` | `error` absent / fails to deserialize into `Error` | -| `response.incomplete` | `:391` | any `response.incomplete` event | -| `response.completed` parse failure | `:406` | `ResponseCompleted` fails to deserialize | -| stream closed before completed | `:459` | byte stream ends, no prior error captured | -| idle timeout | `:466` | no SSE within `idle_timeout` | -| SSE frame decode error | `:454` | a malformed frame on the wire | - -Key facts that constrain the proxy: -- The terminal success event is `response.completed` (`:393`). The chat-completions - `data: [DONE]` sentinel is **ignored** by this parser — it keys on `response.completed`. -- `response.failed` reads `response.error`, **never** `last_error` (`:350`). Absent/unparseable - `error` → `ApiError::Stream("response.failed event received")` (`:349`). -- `ResponseCompleted` requires only `id: String`; `usage` and `end_turn` are - `#[serde(default)] Option<…>` (`:102-108`). The bridge always sets `id` - (`bridge.ts:71`), so completed payloads parse — `:406` is not a current bridge defect. -- Recognized error `code`s (`responses.rs:557-580`): `context_length_exceeded`, - `insufficient_quota`, `usage_not_included`, `invalid_prompt`, `cyber_policy`, - `server_is_overloaded`, `slow_down`. **`rate_limit_exceeded` is not in this set** — it - falls through to `ApiError::Retryable { delay }` (`:369-372`), not a dedicated rate-limit - error. (Correction to the original hypothesis; behavior is acceptable but the wording - "matches the parser code-checks" was wrong for that code.) - -## 2. RC1 — Bridge ends the stream with no terminal `response.completed` (Bridge path) - -**Severity: High. Path: bridge/routed only.** - -`bridgeToResponsesSSE` emits a terminal event **only** inside two switch cases: - -```ts -case "done": emit("response.completed", …) // bridge.ts:276 -case "error": emit("response.failed", …) // bridge.ts:286 -// catch (err): emit("response.failed", …) // bridge.ts:298 -… -emitDone(); // bridge.ts:307 → "data: [DONE]\n\n" (ignored by Codex) -controller.close(); // bridge.ts:308 → byte stream ends -``` - -If the adapter generator **returns without yielding `done` or `error`**, the `for await` -loop (`bridge.ts:171`) simply ends, and control falls to `emitDone()` + `close()`. No -`response.completed` is sent. Codex then hits `Ok(None)` → **"stream closed before -response.completed"** (`responses.rs:459`). - -This is reachable today. `anthropic.ts` emits `done` **only** inside `case "message_delta"` -guarded by `if (usage)` (`anthropic.ts:264-271`); `message_stop` is a no-op -(`anthropic.ts:274-276`); the read loop breaks on EOF with no post-loop terminal yield -(`anthropic.ts:207-209`, `finally` at `:287`). So a stream that ends after `message_stop`, -or whose `message_delta` carries no `usage`, yields **no** `done` → RC1 fires. - -> Contrast: `openai-chat.ts` is safe — it handles `[DONE]` (`:185`) **and** has a post-loop -> fallback `yield { type: "done" }` (`:239`). The defect is the missing invariant -> "*the bridge guarantees a terminal Responses event*," not any single adapter. - -## 3. RC2 — No abort on disconnect; bridge re-throws on a closed controller (Both paths) - -**Severity: High (interactive use). Path: both.** - -The upstream fetch passes **no `signal`**: - -```ts -upstreamResponse = await fetch(request.url, { method, headers, body }); // server.ts:179-183 (bridge) -upstreamResponse = await fetch(request.url, { … }); // server.ts:145-149 (passthrough) -``` - -The bridge `ReadableStream` defines **only `start(controller)`** — no `cancel(reason)` -(`bridge.ts:59-60`). Consequences when the Codex client disconnects (interrupt, new turn, -tool-cycle, timeout — frequent in interactive use): - -1. The upstream socket is never aborted → leaked connection, wasted upstream tokens/time. -2. The next `controller.enqueue()` (`bridge.ts:62`) throws on the now-closed stream. That - throw is caught at `bridge.ts:297`, which calls `emit("response.failed")` → `enqueue` - **throws again, now uncaught inside `start()`** → unhandled rejection; `emitDone()`/ - `close()` (`:307-308`) also throw. - -Across a long interactive session this is the most plausible driver of *"엄청 발생"* -(errors *en masse*): every cancel leaks an upstream stream and emits noisy proxy-side -errors. On the passthrough path the leak is the same (no `signal`), though opencodex returns -`upstreamResponse.body` directly so there is no custom `cancel` to add there — the fix is the -`signal`. - -## 4. RC3 — No idle heartbeat; slow routed providers trip the idle timeout (Bridge path) - -**Severity: Medium (provider-dependent). Path: bridge/routed.** - -Codex aborts with **"idle timeout waiting for SSE"** if no event arrives within -`idle_timeout` (`responses.rs:446,464-468`). The bridge emits `response.created` immediately -(`bridge.ts:75`), which covers first-token latency, but it emits **nothing during mid-stream -stalls** — a slow routed provider, a long upstream reasoning gap, or a slow tool round-trip -produces silence on the opencodex→Codex hop. There is no periodic keep-alive in -`bridgeToResponsesSSE`. Native passthrough inherits the ChatGPT backend's own pacing/keep- -alives, so this primarily bites routed models — i.e. the exact configuration in which the -proxy is most often used (e.g. `opencode-go/deepseek-v4-pro`). - -## 5. RC4 — Bridge fidelity: error envelope + dropped frames (Bridge path) - -**Severity: Medium. Path: bridge/routed. Partly fixed by phase 100.5.** - -- **Error envelope (fixed):** pre-100.5 the bridge emitted `response.failed` with only - `last_error`. Codex reads `error` (`responses.rs:350`), so every translated failure became - `ApiError::Stream("response.failed event received")` (`:349`). Phase 100.5 (`a0d4ec9`) - added a classified `error` via `classifyError` (`errors.ts`, `bridge.ts:289-290`). - `context_length_exceeded` and `insufficient_quota` now match the parser's `is_*_error` - checks exactly (`responses.rs:557-580`). **Caveat:** `errors.ts:26` emits - `rate_limit_exceeded`, which the parser does **not** special-case → generic - `ApiError::Retryable` (`:369-372`). The bridge also emits both `error` and `last_error` - (`bridge.ts:289-290`); `last_error` is dead weight (the parser ignores it) but harmless. -- **Silently dropped frames:** all adapters `catch { continue }` on a JSON parse failure - (`openai-chat.ts:191-193`, `anthropic.ts:226-229`, `google.ts:142-143`). A malformed or - chunk-split upstream frame is dropped silently. This does not throw on the Codex side - (it ignores unparseable frames, `responses.rs:476-478`), but it can truncate content and, - combined with RC1, end the stream without a terminal event. -- **Malformed proxy output:** if opencodex ever emits a malformed Responses frame, Codex - surfaces it as `ApiError::Stream` (`responses.rs:454`). Not currently observed, but the - reason to keep `sseEvent` (`bridge.ts:8-9`) strictly well-formed. - -## 6. RC5 — Passthrough header fidelity (Passthrough path) - -**Severity: Medium. Path: native `gpt-*`. Mitigated by phase 100.5; verify.** - -On passthrough, opencodex relays `upstreamResponse.body` with `sanitizePassthroughHeaders` -(`server.ts:153-155`). Bun's `fetch` auto-decompresses the body but leaves the upstream -`content-encoding: gzip` and a stale `content-length`. If those are relayed, the Codex -client double-decodes / truncates → a malformed frame → `ApiError::Stream` -(`responses.rs:454`). Phase 100.5 expanded the drop set to cover -`content-encoding, content-length, transfer-encoding, connection, keep-alive, -proxy-authenticate, proxy-authorization, te, trailer, upgrade` (`server.ts:241-259`), which -mitigates this. **Verification owed:** confirm `content-type: text/event-stream` survives -sanitization and that Bun always auto-decompresses the passthrough body (if it ever relays -raw gzip bytes, dropping `content-encoding` would itself corrupt the stream). - -## 7. Likelihood & impact, mapped to actual usage - -The proxy is most often pointed at **routed models** (chat/completions upstreams) — e.g. the -`opencode-go/deepseek-v4-pro` session in this project's history. That puts the user squarely -on the **bridge path**, where RC1 + RC3 (+ RC2 on disconnect) compound: - -1. **RC1** (missing terminal) and **RC2** (disconnect re-throw) — highest expected frequency - in interactive Codex sessions; directly produce `ApiError::Stream`. -2. **RC3** (idle timeout) — frequency scales with upstream latency/stalls. -3. **RC4 / RC5** — envelope correctness (mostly fixed) and header hygiene (mostly fixed); - residual risk is silent truncation and the `rate_limit_exceeded` classification gap. - -The single highest-leverage invariant to restore: **the proxy must always terminate a -streaming response with exactly one `response.completed` or a classified `response.failed`, -and must abort the upstream when the client goes away.** Patch direction in `30_…`. diff --git a/devlog/_fin/110_codex-stream-stability/20_transport-evaluation.md b/devlog/_fin/110_codex-stream-stability/20_transport-evaluation.md deleted file mode 100644 index e57af5330..000000000 --- a/devlog/_fin/110_codex-stream-stability/20_transport-evaluation.md +++ /dev/null @@ -1,78 +0,0 @@ -# 110.20 — Transport Evaluation: SSE Multiplexing and WebSockets - -## Question - -Would **SSE multiplexing** or **WebSockets** improve performance, or reduce the stream -errors? Is adding WS to the `chat/completions` adapter worthwhile? - -**Verdict: No on all counts.** The stream errors are lifecycle/reliability defects -(`10_root-cause-analysis.md`), not transport-bandwidth defects. No transport change touches -RC1–RC5. The phase 100 "no WebSocket" decision stands and is reaffirmed here. - -> **Cross-link (amended, not reversed):** this verdict is about routed *reliability/performance* -> — still correct. Native transport *parity* (satisfying Codex's `supports_websockets` -> capability so `ocx` is a drop-in for the native OpenAI provider) is a different axis, tracked -> in `devlog/120_codex-websocket-parity/`. See `120/02_transport-decision.md`. - -## Phase 100 already decided this - -> "Decision: no 100.6 websocket spike / routed providers keep `supports_websockets` -> absent/false / Reason: upstream routed providers are HTTP/SSE, so websocket is not -> end-to-end" — `devlog/100_codex-native-parity/00_overview.md:82-84` - -> "enabling websocket only between Codex and opencodex cannot make a provider/model -> websocket-capable when the upstream provider is not websocket-capable end-to-end … a -> websocket first hop would still block on the same upstream SSE chunks, so it is unlikely to -> improve first-token latency or throughput … setting `supports_websockets = true` would -> advertise a native capability opencodex does not actually provide for routed models." -> — `devlog/100_codex-native-parity/40_responses-lite-websockets.md:102-108` - -## Why WebSockets do not help - -opencodex sits **mid-chain**: `Codex CLI → opencodex → upstream provider`. The upstream -(ChatGPT backend for native, or a chat/completions provider for routed) speaks **HTTP/SSE**. - -- A WS first hop (Codex↔opencodex) still **blocks on the upstream SSE** opencodex is reading. - First-token latency and throughput are bounded by the upstream, unchanged by the hop's - framing. -- WS would add a second protocol surface (frame assembly, ping/pong, close codes) — *more* - ways to produce a "stream error," not fewer. -- For routed models, advertising `supports_websockets` would be **false advertising**: - there is no end-to-end WS. -- **Adding WS to the `chat/completions` adapter is pointless**: the upstream is HTTP/SSE - chat/completions. Wrapping the proxy hop in WS changes nothing upstream and adds a - translation surface. (Matches the reporter's own conclusion.) - -## Why SSE multiplexing does not help the errors - -"SSE multiplexing" usually means carrying many SSE streams over one HTTP/2 connection. - -- The Codex CLI opens **one Responses stream per request**; there is no fan-out to multiplex. -- HTTP/2 multiplexing is a **transport-layer** concern handled by the server/runtime and the - client below the application — it does not change whether a stream is terminated correctly, - aborted on disconnect, or kept alive during a stall. None of RC1–RC5 are connection-count - problems. -- opencodex already disables proxy buffering (`X-Accel-Buffering: no`, `server.ts:211`) and - sets `Cache-Control: no-cache` (`:209`), which are the SSE-relevant transport knobs. - -## Where the real performance wins are - -| Lever | Effect | Touches an RC? | -|-------|--------|----------------| -| Keep **native passthrough** for `gpt-*` | Zero re-encode CPU; backend pacing/keep-alives inherited | avoids RC1/RC3 by construction | -| **Abort upstream on disconnect** (RC2) | No leaked sockets, no wasted upstream tokens/time | RC2 | -| **Idle heartbeat** (RC3) | Avoids false idle-timeout aborts that force a full, expensive retry | RC3 | -| **Guaranteed terminal event** (RC1) | Avoids "stream closed" aborts → full retries | RC1 | -| WebSockets | none | none | -| SSE/HTTP-2 multiplexing | none (one stream per request) | none | - -The dominant cost of the errors is not transport overhead — it is the **full re-run** Codex -performs after an `ApiError::Stream`. Eliminating the aborts (RC1–RC3) removes those re-runs. -That is the performance win, and it comes from reliability fixes, not a new transport. - -## Conclusion - -Do not invest in WebSockets or SSE multiplexing for this problem. Invest in the SSE -lifecycle fixes in `30_patch-direction.md`. Revisit WS only if a provider ever exposes a -real end-to-end WS endpoint opencodex can bridge without converting back to HTTP/SSE -internally (the phase 100 condition, `00_overview.md:87-89`). diff --git a/devlog/_fin/110_codex-stream-stability/30_patch-direction.md b/devlog/_fin/110_codex-stream-stability/30_patch-direction.md deleted file mode 100644 index a996bcb4a..000000000 --- a/devlog/_fin/110_codex-stream-stability/30_patch-direction.md +++ /dev/null @@ -1,179 +0,0 @@ -# 110.30 — Patch Direction - -Prioritized, file-level direction for the fixes implied by `10_root-cause-analysis.md`. -This is **direction, not applied code** — implementation is an approval-gated follow-up -phase. Sketches are illustrative; exact line offsets shift as the files evolve. - -**Non-goals (explicit):** no WebSockets; no attempt to "force passthrough" for routed models -(structurally impossible — see `10_…` §7 and `20_…`). The invariant to restore: *every -streaming response terminates with exactly one `response.completed` or a classified -`response.failed`, and the upstream is aborted when the client disconnects.* - ---- - -## P0 — Stream lifecycle correctness (highest leverage) - -### P0a · RC1 — Guarantee a terminal Responses event (`src/bridge.ts`) - -Track whether a terminal event was emitted; if the adapter generator ends without one, -synthesize `response.completed` before closing. - -```ts -// bridge.ts — inside start(controller) -let terminated = false; -// set `terminated = true` in case "done" (after emit), case "error" (after emit), -// and in the catch block (after emit). - -try { - for await (const event of events) { /* … existing switch … */ } -} catch (err) { - /* existing emit("response.failed", …) */ // bridge.ts:298 - terminated = true; -} - -if (!terminated) { // NEW — RC1 fix - if (currentMsg) closeCurrentMessage(); - if (currentReasoning) closeCurrentReasoning(); - if (currentRawReasoning) closeCurrentRawReasoning(); - if (currentToolCall) closeCurrentToolCall(); - emit("response.completed", { - response: { ...responseSnapshot("completed", finishedItems), usage: responsesUsage(undefined) }, - }); -} - -emitDone(); // bridge.ts:307 (kept; harmless for Codex) -controller.close(); // bridge.ts:308 -``` - -Defense in depth at the adapter layer — make `anthropic.ts` always yield a terminal `done` -on EOF, mirroring `openai-chat.ts:239`: - -```ts -// anthropic.ts — after the read loop, before `finally { reader.releaseLock() }` -yield { type: "done", usage: pendingUsage }; // pendingUsage may be undefined; bridge handles it -``` - -**Test:** feed `bridgeToResponsesSSE` an event sequence ending **without** `done`/`error` -(e.g. `[{type:"text_delta",text:"hi"}]`) and assert the SSE contains exactly one -`response.completed`. - -### P0b · RC2 — Abort upstream on disconnect + never throw on a closed controller - -`src/server.ts` — own an `AbortController`, pass its signal to **both** fetches, and let the -returned stream's cancel abort it. - -```ts -const ac = new AbortController(); -upstreamResponse = await fetch(request.url, { method, headers, body, signal: ac.signal }); // server.ts:179-183 (bridge) -// passthrough fetch (server.ts:145-149) likewise gets `signal: ac.signal` -``` - -`src/bridge.ts` — accept the controller (or an `onCancel` callback) and add `cancel()`; -guard every enqueue so a closed controller is a no-op, not a throw: - -```ts -return new ReadableStream({ - async start(controller) { - const emit = (name, data) => { - try { controller.enqueue(encoder.encode(sseEvent(name, { type: name, sequence_number: seq++, ...data }))); } - catch { /* client gone — stop emitting */ } // NEW — stops the RC2 double-throw - }; - /* … */ - }, - cancel() { onAbort?.(); }, // NEW — aborts the upstream fetch -}); -``` - -For the **passthrough** path opencodex returns `upstreamResponse.body` directly; to abort the -upstream on client cancel, pipe it through a pass-through `TransformStream` whose `cancel()` -calls `ac.abort()` (or rely on the runtime propagating cancel to the signalled fetch — verify -in Bun). The minimal, certain win is passing `signal` so an explicit abort is possible. - -**Test:** start consuming the bridge stream, call `reader.cancel()`, assert the provided -abort callback fired and no unhandled rejection occurs. - ---- - -## P1 — Stall and passthrough robustness - -### P1a · RC3 — Idle keep-alive (`src/bridge.ts`) — IMPLEMENTED (`61dcec2`) - -**Correction (see `40_p0-implementation.md`):** a plain SSE comment (`:\n\n`) will NOT work. -Codex's loop is `timeout(idle_timeout, stream.next())` (`responses.rs:446`) over an -`eventsource_stream` (`responses.rs:12,440`), which parses at the **event** level — a -comment-only frame dispatches no event per the SSE spec, so `.next()` stays pending and the -timer is NOT reset. The keep-alive must be a **real SSE event** that deserializes into -`ResponsesStreamEvent` and is then ignored by the parser's catch-all -(`responses.rs:426-427`, `_ => Ok(None)`) — e.g. a benign `{"type":"response.heartbeat"}` -frame emitted WITHOUT consuming the main `sequence_number` counter. - -**Observed `idle_timeout` values** (vendored codex): default -`DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000` (`model-provider-info/src/lib.rs:26`), but provider -overrides go as low as `5_000` (`model-provider/src/provider.rs:366`) and `9_000` -(`config/src/thread_config/remote.rs:472,535`). A safe interval is ~2500 ms (under the 5 s floor). - -```ts -const HEARTBEAT_MS = 2_500; // under the observed 5 s provider floor; ideally configurable -const beat = setInterval(() => { - if (closed) return; - // a REAL event (not a comment) so eventsource_stream yields it and resets Codex's idle timer; - // an unhandled type is ignored by the parser. Do NOT bump `seq` (keep real events contiguous). - try { controller.enqueue(encoder.encode('event: response.heartbeat\ndata: {"type":"response.heartbeat"}\n\n')); } - catch { closed = true; } -}, HEARTBEAT_MS); -// clearInterval(beat) before controller.close() and inside cancel() -``` - -**Implemented** (`61dcec2`): a real, parser-ignored `response.heartbeat` emitted only during -upstream silence (an `activity` flag skips ticks when real events flow), interval 2000 ms (under -the 5 s provider floor), cleared on every terminal path + close + cancel. The earlier "needs the -user's idle_timeout / maintainer sign-off" concern was retired by an independent review: a ~2 s -interval covers the worst floor without the value, and unknown event types are codex's own -forward-compat path (`responses.rs:426-431`, `_ => Ok(None)`), so emitting one is in-contract. -Unit-tested in `tests/bridge-lifecycle.test.ts` (heartbeat appears during silence). - -### P1b · RC5 — Passthrough header regression test (`tests/`) - -`sanitizePassthroughHeaders` (`server.ts:241-259`) already drops the stale encoding/length -and hop-by-hop headers (phase 100.5). Add an explicit regression test that -`content-type: text/event-stream` **survives** sanitization and `content-encoding` / -`content-length` are dropped, and document a one-time manual check that Bun auto-decompresses -the passthrough body (if it ever relays raw gzip, dropping `content-encoding` would corrupt -the stream — that case needs different handling). - ---- - -## P2 — Fidelity hardening (lower urgency) - -- **`src/errors.ts` — rate-limit classification.** `rate_limit_exceeded` (`errors.ts:26`) is - not recognized by the Codex parser and degrades to generic `ApiError::Retryable` - (`responses.rs:369-372`). Acceptable, but consider mapping 503/overload to - `server_is_overloaded` / `slow_down` (parser-recognized, `responses.rs:577-579`) for - faithful backoff. Also consider dropping the redundant `last_error` from the bridge - `response.failed` (`bridge.ts:289-290`) — the parser ignores it. -- **Dropped-frame visibility.** Adapters `catch { continue }` on bad JSON - (`openai-chat.ts:191-193`, `anthropic.ts:226-229`, `google.ts:142-143`). Add debug/telemetry - logging so silent truncation is detectable, rather than swallowing frames silently. - ---- - -## Verification plan (for the implementation phase) - -1. **Unit (`bun test`):** - - RC1 terminal-guarantee test (P0a). - - RC2 cancel/abort test (P0b). - - RC3 heartbeat-interval test (P1a, can use a fake/short interval). - - RC5 header-preservation test (P1b). -2. **Static:** `bun x tsc --noEmit` clean; `git diff --check`. -3. **Regression:** full `bun test` stays green (baseline 26 pass / 0 fail). -4. **Live (user environment):** run the Codex CLI against `ocx` with a **routed** model over - a multi-turn session that includes interrupts; confirm the absence of `ApiError::Stream` - ("stream closed before response.completed" / "idle timeout") and no leaked upstream - connections. This is the acceptance gate — the symptom is only fully reproducible with a - live Codex client. - -## Sequencing - -P0a + P0b together restore the core invariant and address the most frequent errors; ship -them first behind the unit tests above. P1 follows. P2 is opportunistic. None of this -requires or benefits from a transport change (`20_…`). diff --git a/devlog/_fin/110_codex-stream-stability/40_p0-implementation.md b/devlog/_fin/110_codex-stream-stability/40_p0-implementation.md deleted file mode 100644 index 1726d960c..000000000 --- a/devlog/_fin/110_codex-stream-stability/40_p0-implementation.md +++ /dev/null @@ -1,87 +0,0 @@ -# 110.40 — P0 Implementation: Stream Lifecycle Fixes - -Implements the P0 items from `30_patch-direction.md`. P1a (heartbeat) and P2 are deferred -with rationale below. Scope kept to changes verifiable without a live Codex session. - -## What shipped - -### P0a — RC1: guaranteed terminal `response.completed` (`src/bridge.ts`) - -Commit `1528114`. `bridgeToResponsesSSE` now tracks a `terminated` flag set on the -`done`/`error`/`catch` terminals. If the adapter generator returns **without** a terminal -event (e.g. `anthropic.ts` reaching EOF after `message_stop`, `:274-276`), the bridge closes -any open items and synthesizes a `response.completed` before `emitDone()`/`close()`. This -removes the path to the Codex parser's `"stream closed before response.completed"` → -`ApiError::Stream` (`responses.rs:457-460`). - -Enforced at the **bridge** (not per-adapter) because the invariant is "the bridge always -emits exactly one terminal Responses event," independent of any adapter's quirks -(`10_root-cause-analysis.md` §2). All bridge callers inherit it, including -`web-search/loop.ts:186`. - -### P0b — RC2: no-throw emit + upstream abort on disconnect (`src/bridge.ts`, `src/server.ts`) - -Commits `1528114` (bridge) + `e2ae0b8` (server). - -- `emit`/`emitDone` are now wrapped: a `closed` flag short-circuits and a `try/catch` swallows - enqueue-after-teardown, killing the double-throw that previously fired inside `start()` when - the client vanished mid-stream. -- The bridge `ReadableStream` gained a `cancel()` that sets `closed` and calls an `onCancel` - hook. -- `handleResponses` (`server.ts`) creates an `AbortController`, passes `signal` to the routed - upstream `fetch`, and passes `() => upstream.abort()` as the bridge's `onCancel`. A client - disconnect now aborts the upstream instead of leaking the connection / draining tokens. - -### P0c — RC2 (passthrough path): abort the upstream on client disconnect - -Commit `955f3dd`. The passthrough branch now passes `signal` to the upstream fetch and relays -the body through `relayWithAbort` (`src/server.ts`), whose `cancel()` calls `upstream.abort()`. -A directly-relayed body does not propagate the consumer's cancel to a signalled fetch (verified -with a `Bun.serve` probe and `tests/passthrough-abort.test.ts`), so this closes the passthrough -half of RC2 with byte-verbatim fidelity preserved. - -### RC3 — idle keep-alive (`src/bridge.ts`) [patch-direction §P1a] - -Commit `61dcec2`. During upstream silence the bridge emits a real, parser-ignored -`response.heartbeat` (~2 s interval, fired only when no real event occurred since the last tick), -re-arming Codex's `timeout(idle_timeout, stream.next())` so a stalled routed provider never trips -"idle timeout waiting for SSE". Unknown event types are codex's documented forward-compat path -(`responses.rs:426-431`, `_ => Ok(None)`) with zero side-effects; the 2 s interval is under the 5 s -provider floor, so the user's exact `idle_timeout` is not required. Cleared on every terminal path, -close, and cancel. Native passthrough is unaffected. Unit-tested (`tests/bridge-lifecycle.test.ts`). - -## RC status after P0 - -| ID | Cause | Status | -|----|-------|--------| -| RC1 | Missing terminal `response.completed` (bridge) | **Fixed** (`1528114`) + unit-tested | -| RC2 | Double-throw on disconnect | **Fixed** (`1528114`) | -| RC2 | Upstream not aborted on disconnect (both paths) | **Fixed** — bridge (`e2ae0b8`), passthrough `relayWithAbort` (`955f3dd`, Bun.serve-probe verified) | -| RC3 | No idle keep-alive (idle-timeout aborts) | **Fixed** (`61dcec2`) — parser-ignored `response.heartbeat` during silence, unit-tested | -| RC4 | Bridge error envelope | Fixed in 100.5 (`a0d4ec9`); `rate_limit_exceeded` note stands | -| RC5 | Passthrough header fidelity | Mitigated in 100.5; regression already covered by `tests/error-fidelity.test.ts` | - -## Verification - -- New `tests/bridge-lifecycle.test.ts` (4 tests): terminal guarantee for a no-`done` stream, - single `response.completed` on normal `done` (no double terminal), `response.failed` with no - synthetic completed on `error`, and `cancel()` firing the `onCancel`/abort hook. -- `bun test` → **30 pass / 0 fail** (was 26; +4). `bun x tsc --noEmit` clean. - -## Deferred (and why) - -- **P2** (rate-limit code mapping, dropped-frame logging) — opportunistic only; current behavior - is acceptable. The streaming path is deliberately quiet (no `console.*` in any adapter/bridge), - so dropped-frame logging needs a logging-convention decision and risks spam on benign non-JSON - frames; rate_limit → `Retryable` is already correct. Not shipped. - -(RC3 idle keep-alive was initially deferred here, then implemented in `61dcec2` after an -independent review retired the deferral rationale — see the RC3 subsection above and -`30_patch-direction.md` §P1a.) - -## Remaining acceptance gate - -The symptom is only fully reproducible with a live Codex CLI pointed at `ocx` using a routed -model over a multi-turn session that includes interrupts. Unit + regression tests prove the -mechanism-level fixes; the end-to-end confirmation (no `ApiError::Stream`, no leaked upstream -connections) is owed in the user's environment. diff --git a/devlog/_fin/110_codex-stream-stability/50_closure-overview.md b/devlog/_fin/110_codex-stream-stability/50_closure-overview.md deleted file mode 100644 index 8017e8a58..000000000 --- a/devlog/_fin/110_codex-stream-stability/50_closure-overview.md +++ /dev/null @@ -1,88 +0,0 @@ -# 110.50 — Closure Overview: Remaining Fidelity Items + GPT-Pro 100.n Leftovers - -Phase 110 P0 (RC1–RC3) is implemented and unit-tested (`40_p0-implementation.md`). This -closure series turns the **remaining** open items into phase-100-fidelity, implementation-ready -plans: each downstream doc has Objective / Evidence (file:line) / Files (NEW full content, -MODIFY before-after diff) / Verification / Commit, so implementation can begin directly. - -> **Scope of this series is documentation only.** No production code is changed by the goal -> that produced these docs; the docs are the deliverable. Implementation is a separate, -> approval-gated step (each doc carries its own commit line). - -## Why a closure series - -Two streams of open work converge on the **bridge/adapter fidelity** layer that 110 owns: - -1. **110's own deferred items** — P1b (RC5 passthrough regression test owed) and P2 - (rate-limit/overload classification, dropped-frame visibility), plus the **live E2E - acceptance gate** that unit tests cannot satisfy (`30_patch-direction.md` §P1b, §P2, §Verification). -2. **GPT Pro's "Missing Analysis" on phase 100** — fidelity gaps GPT Pro found while - implementing 100.1–100.5 that are **the same SSE/adapter layer** as 110, not catalog/policy. - (Source: `FINAL_REPORT.md` → "Missing Analysis Found".) - -Folding them into one 110 closure keeps the bridge-fidelity invariant in a single owner and -avoids a redundant "phase 100.6". **WebSocket parity is explicitly NOT here** — it is a -different transport axis tracked as phase 120 (see `20_transport-evaluation.md`; the "no WS for -routed reliability" verdict still stands). - -## Citation basis (IMPORTANT — read before implementing) - -The 110 RCA (`10_root-cause-analysis.md`) cited an **ephemeral** Codex snapshot at -`/tmp/opencodex-codex-src/...`. These closure docs re-base every Codex-parser citation to the -**stable local checkout** the user actually runs: - -```text -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs -``` - -The two differ. A concrete consequence, verified during the plan audit: the stable parser -**does recognize `rate_limit_exceeded`** — `try_parse_retry_after` (`responses.rs:487-509`) -gates on `code == "rate_limit_exceeded"` and extracts the delay from the message text (test -fixture `responses.rs:844` carries `"Please try again in 11.054s."`). The 110 RCA's claim that -`rate_limit_exceeded` "is not in the recognized set" was based on the stale `/tmp` snapshot and -is **superseded** by `53_rate-limit-and-overload-classification.md`. - -## Open-item status (post-P0) - -| ID | Item | Path | Status before closure | Closure doc | -|----|------|------|-----------------------|-------------| -| RC1 | terminal `response.completed` guarantee | bridge | Fixed (`1528114`) | — | -| RC2 | upstream abort on disconnect (both paths) | both | Fixed (`e2ae0b8`,`955f3dd`) | — | -| RC3 | idle keep-alive heartbeat | bridge | Fixed (`61dcec2`) | — | -| **F1** | inline `error` envelope inside a **200** success stream is swallowed | bridge/adapters | **Open** | `51_…` | -| **F2** | usage+choices same-chunk content drop; usage lost on EOF-without-`[DONE]` | adapters | **Open** | `52_…` | -| **F3** | 503/overload not mapped to a Codex-recognized code; retry-after message fidelity | bridge/errors | **Open** (was 110 P2) | `53_…` | -| **F4** | RC5 passthrough header regression test owed; dropped-frame visibility | passthrough/adapters | **Open** (was 110 P1b/P2) | `54_…` | -| **F5** | live-Codex acceptance gate | both | **Owed** | `55_…` | - -## GPT-Pro 100.n leftover → closure mapping - -| GPT Pro "Missing Analysis" item | Closure doc | -|---------------------------------|-------------| -| "successful HTTP/SSE streams can contain embedded provider error envelopes" | `51_…` (F1) | -| "terminal usage may be isolated, combined with a content choice, or followed by EOF without `[DONE]` … adapters must retain usage without skipping the rest of the chunk" | `52_…` (F2) | -| "rate-limit delay parser requires both `rate_limit_exceeded` and a parseable 'Please try again in Ns/ms' message, not only Retry-After" | `53_…` (F3) | -| "generic provider 429 'quota exceeded' often means a temporary request/token bucket rather than fatal paid-credit exhaustion" | `53_…` (F3, retryable-vs-fatal note) | -| "routed normalization also needed to strip native `comp_hash`" | closed in 100.4 (`8c3aa60`/`85a4daa`); verify-only, no new doc | -| "explicit mappings must allow intentionally unmapped built-in providers to fall back conservatively" | closed in 100.4; verify-only, no new doc | - -## Document index - -- `51_success-stream-error-envelope.md` — F1: detect inline `{"error"}` in a 200 stream → classified `response.failed` -- `52_combined-usage-choice.md` — F2: stop dropping content/usage in `openai-chat.ts` and `google.ts` -- `53_rate-limit-and-overload-classification.md` — F3: overload→`server_is_overloaded`; retry-after message contract -- `54_passthrough-and-dropped-frame.md` — F4: RC5 regression test + opt-in dropped-frame logging -- `55_e2e-acceptance.md` — F5: the live-Codex acceptance gate that closes 110 - -## Sequencing - -F1 + F2 are the highest-leverage correctness fixes (they convert silent truncation into -faithful completion/failure) — implement first, behind their unit tests. F3 is faithful-backoff -polish. F4 is regression hardening + observability. F5 is the human acceptance gate and runs -last, in the user's environment. None requires a transport change. - -## Non-goals - -- No WebSockets / no transport change (phase 120). -- No "force passthrough" for routed models (structurally impossible — `10_…` §7). -- No catalog/policy changes (phase 100 territory). diff --git a/devlog/_fin/110_codex-stream-stability/51_success-stream-error-envelope.md b/devlog/_fin/110_codex-stream-stability/51_success-stream-error-envelope.md deleted file mode 100644 index 8593f991c..000000000 --- a/devlog/_fin/110_codex-stream-stability/51_success-stream-error-envelope.md +++ /dev/null @@ -1,208 +0,0 @@ -# 110.51 — F1: Inline Error Envelope Inside a 200 Success Stream - -## Objective - -A chat/completions upstream can return **HTTP 200** and then emit an **inline error envelope** -mid-stream — `data: {"error": {"message": "...", "code": "..."}}` — instead of a clean -`[DONE]`. Today `openai-chat.ts` and `google.ts` **silently swallow** that frame (it has no -`choices`, so the loop `continue`s), then fall through to a post-loop `done`. The bridge turns -that into a `response.completed` with **truncated content**: Codex reports success, the user -sees a half-answer, and the real upstream error is lost. - -The fix: detect a top-level `error` field after JSON-parsing each frame and `yield { type: -"error", message }`. The bridge already converts an adapter `error` event into a classified -`response.failed` (`bridge.ts:322-336`), so this reuses the entire existing failure path — -the change is one guard per adapter. `anthropic.ts` already handles this via its -`case "error"` (`anthropic.ts:277-280`); only the OpenAI-chat and Google adapters need it. - -## Evidence - -Codex consumes a classified failure correctly (stable checkout): - -```text -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:318 is_context_window_error / classification flow -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:513-535 is_*_error recognized codes -``` - -opencodex gap (200-stream inline error is dropped): - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts:189-204 JSON.parse → usage → choices; no error branch -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts:142-146 JSON.parse → candidates; no error branch -``` - -opencodex already-correct sink (reused unchanged): - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:322-336 case "error" → response.failed { error, last_error } (classified) -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/anthropic.ts:277-280 case "error" (already present) -``` - -## Files - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts -``` - -Insert an inline-error guard immediately after the JSON parse, before the `usage` check -(current lines 189-196): - -```diff - let chunk: Record; - try { - chunk = JSON.parse(payload) as Record; - } catch { - continue; - } - -+ // A 200/OK chat-completions stream may carry an inline provider error envelope -+ // instead of a clean [DONE]. Surface it as a terminal error so the bridge emits a -+ // classified response.failed (bridge.ts:322) — never a truncated response.completed. -+ if (chunk.error) { -+ const err = chunk.error as { message?: string } | undefined; -+ if (currentToolCallId) yield { type: "tool_call_end" }; -+ yield { type: "error", message: err?.message ?? "upstream error" }; -+ return; -+ } -+ - if (chunk.usage) { - pendingUsage = usageFromOpenAIChat(chunk.usage as Record); - continue; - } -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts -``` - -Insert the same guard after the JSON parse, before the `candidates` check (current lines 142-146). -Gemini surfaces stream errors as a top-level `error` object: - -```diff - let chunk: Record; - try { chunk = JSON.parse(payload); } catch { continue; } - -+ // Inline provider error inside a 200 stream → terminal error (see openai-chat.ts). -+ if (chunk.error) { -+ const err = chunk.error as { message?: string } | undefined; -+ yield { type: "error", message: err?.message ?? "upstream error" }; -+ return; -+ } -+ - const candidates = chunk.candidates as { content?: { parts?: unknown[] }; finishReason?: string }[] | undefined; - if (!candidates?.length) continue; -``` - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/adapter-error-inline.test.ts -``` - -Complete content: - -```ts -import { describe, expect, test } from "bun:test"; -import { openaiChatAdapter } from "../src/adapters/openai-chat"; -import { googleAdapter } from "../src/adapters/google"; -import { bridgeToResponsesSSE } from "../src/bridge"; -import type { AdapterEvent } from "../src/types"; - -function sseResponse(frames: string[]): Response { - const body = new ReadableStream({ - start(controller) { - const enc = new TextEncoder(); - for (const f of frames) controller.enqueue(enc.encode(f)); - controller.close(); - }, - }); - return new Response(body, { status: 200, headers: { "content-type": "text/event-stream" } }); -} - -async function collect(gen: AsyncGenerator): Promise { - const out: AdapterEvent[] = []; - for await (const e of gen) out.push(e); - return out; -} - -async function collectSse(stream: ReadableStream): Promise<{ event?: string; data: Record }[]> { - const reader = stream.getReader(); - const decoder = new TextDecoder(); - let text = ""; - while (true) { - const { done, value } = await reader.read(); - if (done) break; - text += decoder.decode(value, { stream: true }); - } - return text.split("\n\n").map(f => f.trim()).filter(f => f && f !== "data: [DONE]").map(frame => { - const lines = frame.split("\n"); - const event = lines.find(l => l.startsWith("event: "))?.slice(7); - const dataLine = lines.find(l => l.startsWith("data: ")); - return { event, data: JSON.parse(dataLine?.slice(6) ?? "{}") as Record }; - }); -} - -describe("inline error envelope in a 200 stream", () => { - test("openai-chat yields a terminal error, not silent truncation", async () => { - const res = sseResponse([ - 'data: {"choices":[{"delta":{"content":"par"}}]}\n\n', - 'data: {"error":{"message":"Rate limit reached for model","code":"rate_limit_exceeded"}}\n\n', - ]); - const events = await collect(openaiChatAdapter.parseStream(res)); - expect(events.some(e => e.type === "error")).toBe(true); - expect(events.find(e => e.type === "error")).toMatchObject({ message: "Rate limit reached for model" }); - }); - - test("google yields a terminal error on an inline error frame", async () => { - const res = sseResponse([ - 'data: {"error":{"message":"RESOURCE_EXHAUSTED","code":429}}\n\n', - ]); - const events = await collect(googleAdapter.parseStream(res)); - expect(events.find(e => e.type === "error")).toMatchObject({ message: "RESOURCE_EXHAUSTED" }); - }); - - test("bridge converts the adapter error into a classified response.failed", async () => { - async function* gen(): AsyncGenerator { - yield { type: "text_delta", text: "par" }; - yield { type: "error", message: "Rate limit reached for model" }; - } - const frames = await collectSse(bridgeToResponsesSSE(gen(), "routed/model")); - const failed = frames.find(f => f.event === "response.failed"); - expect(failed).toBeDefined(); - expect((failed!.data.response as Record).error).toMatchObject({ code: "rate_limit_exceeded" }); - expect(frames.some(f => f.event === "response.completed")).toBe(false); - }); -}); -``` - -> If the adapter export names differ (`openaiChatAdapter` / `googleAdapter`), align the imports -> with the actual exports in `src/adapters/index.ts` during implementation — the test logic is -> unchanged. - -## Verification - -```bash -bun test tests/adapter-error-inline.test.ts -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Expected: - -```text -inline-error tests pass (openai-chat + google yield error; bridge emits response.failed, no response.completed) -full suite passes -typecheck clean -whitespace check clean -``` - -## Commit - -```text -[agent] fix: surface inline provider error envelope in 200 streams -``` diff --git a/devlog/_fin/110_codex-stream-stability/52_combined-usage-choice.md b/devlog/_fin/110_codex-stream-stability/52_combined-usage-choice.md deleted file mode 100644 index db7abaea6..000000000 --- a/devlog/_fin/110_codex-stream-stability/52_combined-usage-choice.md +++ /dev/null @@ -1,145 +0,0 @@ -# 110.52 — F2: Combined Usage+Choice Drop and EOF-without-`[DONE]` Usage Loss - -## Objective - -GPT Pro: *"terminal usage may be isolated, combined with a content choice, or followed by EOF -without `[DONE]`, so adapters must retain usage without skipping the rest of the chunk."* Two -concrete defects implement that gap today: - -- **F2a — content dropped on a usage+choice chunk.** `openai-chat.ts:196-199` does - `if (chunk.usage) { pendingUsage = …; continue; }`. A provider that sends `usage` **and** a - final `choices[].delta.content` in the **same** chunk loses that content — the `continue` - skips the choices parsing below it. -- **F2b — usage dropped on EOF-without-`[DONE]`.** When a stream ends by socket EOF (no - `[DONE]` sentinel), the post-loop terminal yields `done` **without** usage: - `openai-chat.ts:239` (`yield { type: "done" }`) and `google.ts:172` (same). The - `pendingUsage` accumulated from a prior usage chunk is silently discarded, so Codex shows a - successful turn with **zero token usage**. - -`google.ts` has an additional latent defect: it yields `done` **twice** on a normal finish — -inline at `:164-169` (when `finishReason && usageMeta`) **and** unconditionally post-loop at -`:172`. Consolidating to a single post-loop `done` fixes F2b and the double-terminal at once. - -## Evidence - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts:196-199 if (chunk.usage) { pendingUsage = …; continue; } ← F2a -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts:239 yield { type: "done" }; (post-loop, no usage) ← F2b -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts:163-169 inline done with usage when finishReason && usageMeta ← double-done -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts:172 yield { type: "done" }; (post-loop, no usage) ← F2b -``` - -Bridge usage projection that receives `done.usage` (unchanged): - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:311-321 case "done" → response.completed { usage: responsesUsage(event.usage) } -``` - -## Files - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts -``` - -F2a — record usage but do **not** `continue`; fall through to choices parsing (current 196-199): - -```diff -- if (chunk.usage) { -- pendingUsage = usageFromOpenAIChat(chunk.usage as Record); -- continue; -- } -+ if (chunk.usage) { -+ // Record usage but keep parsing: some providers send usage and the final content -+ // delta in the SAME chunk; a `continue` here would drop that content. -+ pendingUsage = usageFromOpenAIChat(chunk.usage as Record); -+ } -``` - -> **Why removing `continue` is safe (do not skip this):** the line immediately below is -> `const choices = chunk.choices …; if (!choices || choices.length === 0) continue;`. A -> usage-only chunk has no `choices`, so that guard already no-ops it. Removing the early -> `continue` therefore only changes behavior for the usage+content chunk — the case we are -> fixing — and is a no-op for every other chunk. - -F2b — carry `pendingUsage` on the post-loop terminal (current line 239): - -```diff - if (currentToolCallId) { - yield { type: "tool_call_end" }; - } -- yield { type: "done" }; -+ yield { type: "done", usage: pendingUsage }; -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts -``` - -Replace the inline `done` with a `pendingUsage` accumulator and emit a single post-loop `done`. -Declare `pendingUsage` immediately after `let buffer = "";` (`src/adapters/google.ts:126`, with -the `reader`/`decoder`/`buffer` loop locals at `:124-126`): - -```diff - let buffer = ""; -+ let pendingUsage: OcxUsage | undefined; -``` - -Record usage instead of yielding an inline `done` (current 163-169): - -```diff - const usageMeta = chunk.usageMetadata as Record | undefined; -- if (candidates[0].finishReason && usageMeta) { -- yield { -- type: "done", -- usage: usageFromGemini(usageMeta), -- }; -- } -+ if (usageMeta) { -+ pendingUsage = usageFromGemini(usageMeta); -+ } -``` - -Emit one terminal `done` with usage post-loop (current line 172): - -```diff -- yield { type: "done" }; -+ yield { type: "done", usage: pendingUsage }; -``` - -> Add `import type { OcxUsage } from "../types";` if `google.ts` does not already import it -> (verify the existing import block during implementation). - -## Verification - -```bash -bun test tests/adapter-usage.test.ts -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Add to `tests/adapter-usage.test.ts` (or a new `tests/adapter-eof-usage.test.ts`) cases that -assert: - -- openai-chat: a single chunk carrying **both** `usage` and `choices[].delta.content` - yields the `text_delta` **and** a final `done.usage` (content not dropped). -- openai-chat: a stream ending by EOF (no `[DONE]`) after a usage chunk yields - `done.usage` equal to the accumulated usage (not `undefined`). -- google: a normal finish yields **exactly one** `done`, and it carries usage. - -Expected: - -```text -combined-chunk content preserved; EOF usage retained; google emits a single done with usage -full suite passes; typecheck clean; whitespace clean -``` - -## Commit - -```text -[agent] fix: retain usage and content on combined and EOF-terminated streams -``` diff --git a/devlog/_fin/110_codex-stream-stability/53_rate-limit-and-overload-classification.md b/devlog/_fin/110_codex-stream-stability/53_rate-limit-and-overload-classification.md deleted file mode 100644 index 77e3207ff..000000000 --- a/devlog/_fin/110_codex-stream-stability/53_rate-limit-and-overload-classification.md +++ /dev/null @@ -1,148 +0,0 @@ -# 110.53 — F3: Overload Mapping and Retry-After Message Fidelity - -## Objective - -Make translated backoff faithful to what the **stable** Codex parser actually does. This -supersedes the 110 RCA note that "`rate_limit_exceeded` is not recognized" — that was based on -the stale `/tmp/opencodex-codex-src` snapshot. The stable checkout -(`/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs`) recognizes -`rate_limit_exceeded` and extracts the delay from the **message text**. - -Three faithful-backoff items, plus one cleanup: - -- **F3a — map overload to a recognized code.** `errors.ts:31-33` maps every `status >= 500` to - `upstream_server_error`, which Codex does **not** special-case. A 503 / "overloaded" upstream - should map to `server_is_overloaded`, which Codex recognizes (`is_server_overloaded_error`, - `responses.rs:533-535`) and backs off on with retry-after (`responses.rs:332-335`). -- **F3b — preserve the retry-after delay text.** Codex's `try_parse_retry_after` - (`responses.rs:487-509`) reads the delay from the **error message** (e.g. `"Please try again - in 11.054s."`, fixture `responses.rs:844`). `classifyError` already passes `message` through - verbatim, so an upstream phrase survives — the requirement is a **contract**: never normalize - or truncate a rate-limit message in a way that strips `"try again in Ns/ms"`, and never - fabricate a fake delay when the upstream gave none. -- **F3c — keep transient 429 "quota" retryable.** GPT Pro: a generic 429 "quota exceeded" is - often a temporary request/token bucket, not fatal paid-credit exhaustion. `errors.ts:18-24` - matches bare `"quota exceeded"` → `insufficient_quota` (a Codex-recognized **fatal** quota - error, `is_quota_exceeded_error`, `responses.rs:517`). Tighten the match so a bare 429 bucket - falls through to the retryable `rate_limit_exceeded` branch. -- **F3d (optional cleanup) — drop the dead `last_error`.** The bridge emits both `error` and - `last_error` on `response.failed` (`bridge.ts:331,344`); the parser reads only `error` - (`responses.rs` classification reads `response.error`). `last_error` is harmless dead weight. - -## Evidence - -```text -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:318 is_context_window_error(&error) (classification entry) -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:332-335 is_server_overloaded_error → delay = try_parse_retry_after(&error) -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:487-509 try_parse_retry_after gates on code == "rate_limit_exceeded", parses "try again in Ns" -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:517 is_quota_exceeded_error (fatal quota) -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:533-535 is_server_overloaded_error → "server_is_overloaded" | "slow_down" -/Users/jun/Developer/codex/codex-cli/codex-rs/codex-api/src/sse/responses.rs:844 test fixture: rate_limit_exceeded message "Please try again in 11.054s." -``` - -opencodex classifier: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/errors.ts:18-37 -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts:330-331, 343-344 (F3d optional) -``` - -## Files - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/errors.ts -``` - -F3c — drop the over-broad `"quota exceeded"` so transient 429 buckets stay retryable -(current 18-24): - -```diff - if ( - text.includes("insufficient_quota") || -- text.includes("quota exceeded") || - text.includes("exceeded your current quota") - ) { - return { message, type: "insufficient_quota", code: "insufficient_quota" }; - } -``` - -F3a — add an overload branch mapping to a Codex-recognized code. Insert it **between line 30** -(the auth branch's closing `}`, `errors.ts:28-30`) **and line 31** (the `if (status >= 500)` -check) — so 401/403 auth is matched first, then overload, then the generic 5xx fallback: - -```diff - if (status === 401 || status === 403 || type === "authentication_error") { - return { message, type: "authentication_error", code: "invalid_api_key" }; - } -+ if ( -+ status === 503 || -+ text.includes("overloaded") || -+ text.includes("server is busy") || -+ text.includes("temporarily unavailable") -+ ) { -+ // Codex recognizes "server_is_overloaded" and applies retry-after backoff -+ // (responses.rs:332-335,533-535); generic "upstream_server_error" is not recognized. -+ return { message, type: "server_error", code: "server_is_overloaded" }; -+ } - if (status >= 500) { - return { message, type: "server_error", code: "upstream_server_error" }; - } -``` - -> **Contract (F3b) — no code change, enforce in review:** the `message` argument flows into the -> emitted envelope verbatim. Callers must pass the **upstream** error text (which carries -> `"try again in Ns"`) and must not synthesize a rate-limit error with a fabricated delay. When -> no upstream delay text exists, emit the message without one — Codex falls back to its default -> backoff, which is correct. - -### MODIFY (optional — F3d) - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts -``` - -```diff - emit("response.failed", { - response: { - ...responseSnapshot("failed", finishedItems), - error: responseError(502, "upstream_error", event.message), -- last_error: responseError(502, "upstream_error", event.message), - }, - }); -``` - -(and the symmetric drop in the `catch` block at `bridge.ts:343-344`). Defer if any consumer -other than the Codex parser reads `last_error`; the parser does not. - -## Verification - -Extend `tests/error-fidelity.test.ts`: - -```bash -bun test tests/error-fidelity.test.ts -bun test tests -bun x tsc --noEmit -git diff --check -``` - -Assert: - -- `classifyError(503, "upstream_error", "The server is overloaded")` → `code: "server_is_overloaded"`. -- `classifyError(429, "upstream_error", "You have exceeded your quota for requests per min. Please try again in 5s")` - → `code: "rate_limit_exceeded"` (retryable) **and** message still contains `"try again in 5s"`. -- `classifyError(402, "upstream_error", "You exceeded your current quota")` → `code: "insufficient_quota"` (still fatal). - -Expected: - -```text -overload → server_is_overloaded; transient 429 quota → rate_limit_exceeded (delay text preserved); -real exhaustion → insufficient_quota; full suite passes; typecheck clean -``` - -## Commit - -```text -[agent] fix: map overload to server_is_overloaded and keep transient 429 retryable -``` diff --git a/devlog/_fin/110_codex-stream-stability/54_passthrough-and-dropped-frame.md b/devlog/_fin/110_codex-stream-stability/54_passthrough-and-dropped-frame.md deleted file mode 100644 index ab8c26544..000000000 --- a/devlog/_fin/110_codex-stream-stability/54_passthrough-and-dropped-frame.md +++ /dev/null @@ -1,161 +0,0 @@ -# 110.54 — F4: Passthrough Header Regression + Dropped-Frame Visibility - -## Objective - -Close 110's two hardening items (was P1b + P2 in `30_patch-direction.md`): - -- **F4a — RC5 regression test owed.** `sanitizePassthroughHeaders` (`server.ts:283-301`) drops - stale encoding/length + hop-by-hop headers (phase 100.5). No test asserts the **positive** - half: that `content-type: text/event-stream` **survives** sanitization. If a future edit - over-broadens the DROP set and strips `content-type`, native `gpt-*` passthrough breaks - silently. Add a regression test (the existing `error-fidelity.test.ts` only covers - `application/json`). -- **F4b — Bun auto-decompress check.** Document the one-time confirmation that Bun's `fetch` - auto-decompresses the passthrough body, which is the premise that makes dropping - `content-encoding` safe (`server.ts:278-281`). If Bun ever relays raw gzip bytes, dropping - `content-encoding` would corrupt the stream — a different fix. -- **F4c — dropped-frame visibility.** Every adapter `catch { continue }`s on a JSON parse - failure (`openai-chat.ts:192-193`, `google.ts:143`, `anthropic.ts` parse catch). A - chunk-split or malformed upstream frame is dropped **silently**, which can truncate content - and (with RC1) end a stream early. The streaming path is deliberately quiet (no unconditional - `console.*`), so add **opt-in** logging behind an env flag rather than always-on spam. - -## Evidence - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:278-301 sanitizePassthroughHeaders + DROP set -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts:189-194 try { JSON.parse } catch { continue } -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts:143 try { JSON.parse } catch { continue } -/Users/jun/Developer/new/700_projects/opencodex/tests/error-fidelity.test.ts existing sanitize test (json only) -``` - -## Files - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/passthrough-headers.test.ts -``` - -Complete content: - -```ts -import { describe, expect, test } from "bun:test"; -import { sanitizePassthroughHeaders } from "../src/server"; - -describe("passthrough header sanitization (RC5)", () => { - test("content-type: text/event-stream survives sanitization", () => { - const sanitized = sanitizePassthroughHeaders(new Headers({ - "content-type": "text/event-stream; charset=utf-8", - "content-encoding": "gzip", - "content-length": "4096", - "x-request-id": "req_abc", - })); - expect(sanitized.get("content-type")).toBe("text/event-stream; charset=utf-8"); - expect(sanitized.has("content-encoding")).toBe(false); - expect(sanitized.has("content-length")).toBe(false); - expect(sanitized.get("x-request-id")).toBe("req_abc"); - }); - - test("hop-by-hop and stale framing headers are dropped, telemetry preserved", () => { - const sanitized = sanitizePassthroughHeaders(new Headers({ - "transfer-encoding": "chunked", - "connection": "keep-alive", - "te": "trailers", - "upgrade": "websocket", - "openai-processing-ms": "812", - "x-ratelimit-remaining-tokens": "29000", - })); - for (const h of ["transfer-encoding", "connection", "te", "upgrade"]) { - expect(sanitized.has(h)).toBe(false); - } - expect(sanitized.get("openai-processing-ms")).toBe("812"); - expect(sanitized.get("x-ratelimit-remaining-tokens")).toBe("29000"); - }); -}); -``` - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/debug.ts -``` - -Complete content: - -```ts -// Opt-in frame-drop visibility. The streaming path is intentionally quiet (no unconditional -// console output), so this no-ops unless OCX_DEBUG_FRAMES=1. Lets a malformed/chunk-split -// upstream frame be detected instead of silently truncating content. -const DEBUG_FRAMES = process.env.OCX_DEBUG_FRAMES === "1"; - -export function debugDroppedFrame(adapter: string, payload: string): void { - if (!DEBUG_FRAMES) return; - const preview = payload.length > 200 ? `${payload.slice(0, 200)}…` : payload; - console.error(`[ocx:frame-drop] ${adapter}: ${preview}`); -} -``` - -### MODIFY (adapters — wire the opt-in helper into each parse catch) - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-chat.ts -``` - -```diff -+import { debugDroppedFrame } from "../debug"; -@@ - try { - chunk = JSON.parse(payload) as Record; - } catch { -+ debugDroppedFrame("openai-chat", payload); - continue; - } -``` - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/google.ts -``` - -```diff -+import { debugDroppedFrame } from "../debug"; -@@ -- try { chunk = JSON.parse(payload); } catch { continue; } -+ try { chunk = JSON.parse(payload); } catch { debugDroppedFrame("google", payload); continue; } -``` - -Apply the same one-line change to the `anthropic.ts` parse catch (locate its -`} catch { continue; }` during implementation; the import + helper call are identical). - -## Verification - -```bash -bun test tests/passthrough-headers.test.ts -bun test tests -bun x tsc --noEmit -git diff --check -``` - -**F4b manual check (one-time, record the result in the commit body):** - -```bash -# Confirm Bun auto-decompresses a gzip upstream body so dropping content-encoding is safe. -bun -e 'const r = await fetch("https://httpbin.org/gzip"); console.log("content-encoding:", r.headers.get("content-encoding")); const t = await r.text(); console.log("decoded JSON ok:", t.trim().startsWith("{"));' -# Expect: body is already-decoded JSON (auto-decompressed). If it prints raw gzip bytes, -# dropping content-encoding is NOT safe and F4b needs a real decode step instead. -``` - -Expected: - -```text -content-type survives; stale/hop-by-hop dropped; telemetry preserved -OCX_DEBUG_FRAMES default off → no console output in normal runs -Bun auto-decompress confirmed -full suite passes; typecheck clean -``` - -## Commit - -```text -[agent] test: lock passthrough SSE header survival; add opt-in frame-drop logging -``` diff --git a/devlog/_fin/110_codex-stream-stability/55_e2e-acceptance.md b/devlog/_fin/110_codex-stream-stability/55_e2e-acceptance.md deleted file mode 100644 index 2ddb6cb83..000000000 --- a/devlog/_fin/110_codex-stream-stability/55_e2e-acceptance.md +++ /dev/null @@ -1,103 +0,0 @@ -# 110.55 — F5: Live-Codex Acceptance Gate - -## Objective - -RC1–RC3 and F1–F4 are provable by unit tests at the mechanism level, but the original symptom -— *"엄청 발생"* (`ApiError::Stream` en masse) — is only fully reproducible with a **live Codex -CLI** driving a **routed** model through `ocx` over a multi-turn session with interrupts. This -doc defines that acceptance gate as a concrete, runnable checklist. It is a **verification -plan**, not code; it closes 110 once executed in the user's environment. - -## Preconditions - -- `ocx` running and reachable (default `http://localhost:10100`). -- A routed provider configured with real credentials (the historical repro config is - `opencode-go/deepseek-v4-pro`; any chat/completions routed model reproduces the bridge path). -- Codex CLI installed (`command -v codex`). - -## Codex configuration - -`~/.codex/config.toml` — point Codex at `ocx` as an OpenAI-compatible Responses provider and -select a **routed** model so the **bridge** path (not native passthrough) is exercised: - -```toml -model_provider = "opencodex" -model = "opencode-go/deepseek-v4-pro" - -[model_providers.opencodex] -base_url = "http://localhost:10100/v1" -wire_api = "responses" -requires_openai_auth = true -# supports_websockets intentionally absent — WS parity is phase 120, not 110. -``` - -## Scenarios (each must hold) - -1. **Single long answer.** Ask for a multi-paragraph answer. Expect a complete reply, exactly - one terminal, and no `ApiError::Stream` in Codex logs. (RC1 + F1/F2) -2. **Interrupt mid-stream.** Start a long generation, press the interrupt key, immediately send - a new turn. Expect: the new turn answers cleanly; no leaked upstream connection from the - aborted turn; no proxy-side unhandled rejection. (RC2) -3. **Slow / stalling provider.** Use a slow routed model (or a long reasoning prompt). Expect: - no `"idle timeout waiting for SSE"` — the `response.heartbeat` keeps Codex's idle timer - armed through the stall. (RC3) -4. **Tool round-trip.** Trigger a turn with a function/tool call and a follow-up. Expect: tool - call commits, no `JSON.parse("")` 400, conversation continues. (bridge tool-call finalize) -5. **Upstream error mid-stream.** Force a rate-limit/overload (hammer the provider, or use a - key near its limit). Expect: Codex surfaces a **classified** failure (rate-limit/overload), - **not** `"response.failed event received"` or a truncated success. (F1 + F3) - -## Observation methods - -```bash -# 1. Codex client-side stream errors (should be empty across the session): -# watch the Codex CLI log / TUI for "ApiError::Stream", "stream closed before response.completed", -# "idle timeout waiting for SSE", "response.failed event received". - -# 2. Leaked upstream sockets after an interrupt (count should return to baseline): -lsof -p "$(pgrep -f 'ocx|opencodex' | head -1)" 2>/dev/null | grep -c ESTABLISHED - -# 3. ocx-side noise (should be quiet; with OCX_DEBUG_FRAMES=1, inspect any frame drops): -OCX_DEBUG_FRAMES=1 ocx # run ocx with frame-drop visibility during the session -``` - -## Pass criteria - -| # | Criterion | How verified | -|---|-----------|--------------| -| 1 | Zero `ApiError::Stream` across all 5 scenarios | Codex log scan | -| 2 | No `"stream closed before response.completed"` | Codex log scan | -| 3 | No `"idle timeout waiting for SSE"` on the slow scenario | Codex log scan (scenario 3) | -| 4 | ESTABLISHED upstream count returns to baseline after each interrupt | `lsof` before/after (scenario 2) | -| 5 | Mid-stream upstream errors arrive **classified** (rate-limit/overload/context) | Codex error surface (scenario 5) | -| 6 | No proxy-side unhandled rejection in the `ocx` process | `ocx` stderr | - -## Recording - -On completion, append a short results block to this doc (date, codex version, routed model, -pass/fail per criterion) and reference it from `50_closure-overview.md`'s status table (flip F5 -to **Closed**). If any criterion fails, capture the exact log line and open a follow-up doc in -the 56+ range rather than editing the implemented fixes blind. - -## Results — executed 2026-06-20 (F5 PASS) - -Live run against the real `opencode-go` upstream (`opencode.ai/zen/go/v1`) using the saved token, -model `opencode-go/kimi-k2.7-code`, via the **bridge path** (openai-chat adapter). Isolated repo -build on port 10199 (the running 10100 instance and codex config untouched). - -| Criterion | Result | -|-----------|--------| -| Clean lifecycle, single terminal | PASS — `response.created → reasoning_text.delta×20 → output_item.done → message → response.completed` | -| Exactly one `response.completed`; zero `ApiError::Stream` | PASS — `status: completed`; 0 error/`stream closed` frames | -| Usage present in terminal (F2 live) | PASS — `usage {input_tokens:13, output_tokens:25, total_tokens:38}` | -| Answer correctness | PASS — output text `"hello world"` as prompted | -| RC2 client-disconnect mid-stream | PASS — curl `--max-time 1.5` (exit 28); server stayed healthy; a subsequent request returned `response.completed`; no unhandled rejection / crash in server log | - -Error-path fixes (F1 inline error, F3 overload/transient-429) are covered by unit tests -(`adapter-error-inline`, `error-fidelity`) since the live upstream returned success. Stall/idle -(RC3) is covered by `bridge-lifecycle` unit tests. **F5 status: Closed.** - -## Non-goals - -- No native `gpt-*` WS path here (phase 120). -- Not a load/perf benchmark — this gate is correctness/lifecycle only. diff --git a/devlog/_fin/120_codex-websocket-parity/00_overview.md b/devlog/_fin/120_codex-websocket-parity/00_overview.md deleted file mode 100644 index 626fa6c5d..000000000 --- a/devlog/_fin/120_codex-websocket-parity/00_overview.md +++ /dev/null @@ -1,78 +0,0 @@ -# 120.00 — Overview: Responses WebSocket Parity for opencodex - -## What this phase is - -Make the opencodex provider able to speak the Codex **Responses WebSocket** protocol, so that -Codex's native WS-first transport works *through* `ocx` — reaching transport parity with the -native OpenAI provider. Today opencodex serves only HTTP `POST /v1/responses` (SSE); it has no -WebSocket endpoint. This phase designs and plans that endpoint. - -This is a **foundation cycle** (research + decision). It produces three docs (00–02) and does -**not** change production code. The implementation-ready sub-phase plans live in `10_`–`13_`. - -## Why now — and why it is NOT a 110 bug - -Codex chooses the WS path purely on the provider capability flag -`Provider.supports_websockets` (`/Users/jun/Developer/codex/codex-cli/codex-rs/core/src/client.rs:772`). -**The catalog opencodex serves advertises that flag on no entry today** — verified against the -live served catalog `/Users/jun/.codex/opencodex-catalog.json` (zero `supports_websockets` -occurrences, native or routed). The mechanism differs by entry class, which matters for the -rollout in `12_`: - -- **Routed** entries are explicitly stripped: `src/codex-catalog.ts:78` - (`delete entry.supports_websockets`, inside `normalizeRoutedCatalogEntry`). -- **Native** `gpt-*` entries are cloned from the installed Codex template by `deriveEntry` - (`codex-catalog.ts:145-154`) **without** a strip — they simply inherit no flag because the - current installed template carries none. **Latent risk:** if a future Codex template adds - `supports_websockets` to native entries, `deriveEntry` would leak it and Codex would start - attempting WS against `ocx` with no endpoint → a new handshake-failure error ("RC6"). `12_` - must guard the native path, not only manage the routed strip. - -So today Codex **never attempts a WS first-hop** against `ocx` (no current error). Two consequences: - -1. The absence of WS causes **zero** current stream errors — which is exactly why phase - `110/20_transport-evaluation.md` correctly concluded "WebSockets do not help" for routed - **reliability**. That verdict still stands and is **not** reversed by this phase. -2. WS parity is therefore a **capability/feature**, not a bug fix. 120 adds an opt-in transport; - it is a *different axis* (native transport parity) from 110 (SSE lifecycle correctness). - -**Guard (the one real link to 110):** never advertise `supports_websockets = true` until the -endpoint actually exists. A flag flip without an endpoint makes Codex attempt a WS upgrade that -fails the handshake — a *new* stream-error source (call it "RC6") that 110 never had. The flag -deletion at `codex-catalog.ts:78` stays until `12_metadata-enable-fallback.md` ships. - -## Relationship to phases 100 / 110 - -| Phase | Axis | Status | -|-------|------|--------| -| 100 | catalog/policy/error parity (HTTP) | implemented | -| 110 | SSE **lifecycle** reliability (RC1–RC5 + F1–F5 closure) | P0 done; closure planned (`110/50_`) | -| **120** | **WebSocket transport parity** (new) | this phase — foundation only | - -110.20 gets a one-line cross-link to 120 (it is amended, not reversed): "WS adds nothing to -routed *reliability*; native *transport parity* is tracked in phase 120." - -## Scope of the foundation cycle - -- `00_overview.md` — this doc: framing, premise, non-bug status, guard. -- `01_codex-ws-protocol-analysis.md` — the exact Codex WS wire protocol (frames, handshake, - lifecycle, minimum server obligations), cited to the stable codex checkout. -- `02_transport-decision.md` — the two implementation strategies (Codex-facing WS bridge MVP vs - native upstream WS), the recommendation, effort, and the capability-flag rollout. - -## Out of scope (this cycle) - -- Any production code. (Implementation plans are `10_`–`13_`; implementation itself is a later, - approval-gated step.) -- Reversing 110.20's routed-reliability conclusion. -- A routed-provider end-to-end WS (structurally impossible — routed upstreams are HTTP/SSE; the - best routed WS can be is a Codex-facing first hop, covered by the MVP option in `02_`). - -## Implementation sub-phases (planned in 10–13) - -| Doc | Sub-phase | Summary | -|-----|-----------|---------| -| `10_` | 120.2 WS endpoint MVP | Bun WS upgrade on `/v1/responses`; parse `response.create`; reuse the existing route/adapter/bridge pipeline; emit Responses events as WS Text frames; `response.processed` no-op; close→abort upstream | -| `11_` | 120.3 native upstream WS | native `gpt-*` → connect to upstream ChatGPT backend WS (true end-to-end parity) | -| `12_` | 120.4 metadata enable + fallback | selectively re-advertise `supports_websockets`; verify Codex HTTP fallback; terminal/heartbeat parity on WS | -| `13_` | 120.5 live E2E | Codex CLI over WS against `ocx` (native + routed), interrupts/stalls/tools | diff --git a/devlog/_fin/120_codex-websocket-parity/01_codex-ws-protocol-analysis.md b/devlog/_fin/120_codex-websocket-parity/01_codex-ws-protocol-analysis.md deleted file mode 100644 index 741c3c0c7..000000000 --- a/devlog/_fin/120_codex-websocket-parity/01_codex-ws-protocol-analysis.md +++ /dev/null @@ -1,121 +0,0 @@ -# 120.01 — Codex Responses WebSocket Protocol Analysis - -Reference map of the wire protocol Codex speaks on the WS path, derived by reading the **stable -codex checkout**. All citations are relative to: - -```text -/Users/jun/Developer/codex/codex-cli/codex-rs/ -``` - -Files: `codex-api/src/endpoint/responses_websocket.rs` (WS client), `codex-api/src/common.rs` -(request frames), `codex-api/src/provider.rs` (URL), `codex-api/src/sse/responses.rs` (event -schema — shared with SSE), `core/src/client.rs` (selection + headers). - -This is the contract a Responses-WS **server** (opencodex) must satisfy. Implementation plans -(`10_`–`13_`) cite specific rows here. - -## 1. Path selection (when Codex uses WS) - -- Gated solely by `Provider.supports_websockets` (+ a runtime `disable_websockets` kill switch): - `core/src/client.rs:772` → `if !provider.info().supports_websockets || disable_websockets { return false; }`. -- No `wire_api` gate in the WS code. The connection is opened lazily and **reused across turns** - within a session (`core/src/client.rs:13-19`). - -## 2. URL derivation - -- `Provider::websocket_url_for_path()` (`codex-api/src/provider.rs:92-103`): swap scheme - `http→ws`, `https→wss`; **path unchanged**. Called with `"responses"` - (`responses_websocket.rs:378`). -- So a provider `base_url = http://localhost:10100/v1` yields `ws://localhost:10100/v1/responses`. - -## 3. Handshake / upgrade - -- **Auth is HTTP-upgrade-time only** (no per-frame auth): `auth.add_auth_headers(&mut headers)` - on the upgrade request (`responses_websocket.rs:383`). -- Beta header `OpenAI-Beta: ` (`core/src/client.rs:912-914`), - plus session/request headers (`core/src/client.rs:904-910`) and `x-codex-turn-state` when - present (`responses_websocket.rs:525-532`). -- Upgrade must succeed with **HTTP 101 Switching Protocols** (`responses_websocket.rs:501-506`). -- `permessage-deflate` compression is negotiated by default (`responses_websocket.rs:542-549`). -- Server upgrade-**response** headers Codex reads (all optional): `x-reasoning-included` - (`:514`), `x-models-etag` (`:515-519`), `openai-model` (`:520-524`), `x-codex-turn-state` - (`:525-532`). - -## 4. Client → server frames (what Codex SENDS) - -Envelope `ResponsesWsRequest`, serde `#[serde(tag = "type")]` (`codex-api/src/common.rs:269-277`): - -| `type` | Struct | When | -|--------|--------|------| -| `response.create` | `ResponseCreateWsRequest` (`common.rs:215-240`) | each turn | -| `response.processed` | `ResponseProcessedWsRequest { response_id }` (`common.rs:242-245`) | ack after consuming a response | - -- `response.create` fields: `model, instructions, previous_response_id?, input: Vec, - tools, tool_choice, parallel_tool_calls, reasoning?, store, stream, include, service_tier?, - prompt_cache_key?, text?, generate?, client_metadata?` (`common.rs:215-240`). It is the same - semantic payload as the HTTP `POST /v1/responses` body. -- Sent as a single WS **Text** message (`responses_websocket.rs:782,795`). -- `previous_response_id` links multi-turn on the reused connection (`common.rs:197,221`). -- `generate: false` is a **prewarm** (open + warm the connection without an LLM request) - (`core/src/client.rs:15-16`). - -## 5. Server → client frames (what Codex EXPECTS) - -- Same schema as SSE: `ResponsesStreamEvent` (`sse/responses.rs:145-158`), parsed by - `serde_json::from_str` from each Text frame (`responses_websocket.rs:704-718`). -- Event types parsed (`sse/responses.rs:263-397`): `response.created`, - `response.output_item.added/.done`, `response.output_text.delta`, - `response.reasoning_summary_text.delta`, `response.reasoning_text.delta`, - `response.reasoning_summary_part.added`, `response.custom_tool_call_input.delta`, - `codex.rate_limits`, `response.metadata`, and the terminals below. -- **Terminal success: `response.completed`** — the ONLY success terminal; Codex breaks the loop - on it (`responses_websocket.rs:747-750`; payload shape `sse/responses.rs:358-375`, requires - `response.id`). -- **Error terminals:** `response.failed` / `response.incomplete` → classified `ApiError` - (same classifier as SSE, `sse/responses.rs`). A standalone error **frame** - `{"type":"error","status":429,"error":{…}}` is also accepted (`responses_websocket.rs:575-636`); - `code == "websocket_connection_limit_reached"` → `ApiError::Retryable` (`:613-622`). -- **Unknown event types are silently skipped** (forward-compat, `sse/responses.rs:391-394`). - -## 6. Lifecycle: close / cancel / idle / ping / errors - -| Concern | Behavior | Cite | -|---------|----------|------| -| Server closes before completed | `ApiError::Stream("websocket closed by server before response.completed")` | `:762-765` | -| Stream EOF before completed | `ApiError::Stream("stream closed before response.completed")` | `:693-696` | -| Idle (per-message) timeout on recv | `ApiError::Stream("idle timeout waiting for websocket")` | `:682-684` | -| Idle timeout on send | `ApiError::Stream("idle timeout sending websocket request")` | `:793-798` | -| Server `Ping` | Codex auto-replies `Pong` (no app action) | `:93-98` | -| Binary frame | forbidden → `ApiError::Stream("unexpected binary websocket event")` | `:759-760` | -| Transport/WS error | `ApiError::Stream(err)` | `:690-691` | -| Cancel | Codex drops the stream; **no explicit close frame** — server detects socket EOF | (no send-close in source) | - -## 7. Minimum server obligations (opencodex MUST) - -Derived strictly from §3–§6: - -1. Accept a WS **upgrade at `/v1/responses`** returning HTTP 101 (`:500-506`). -2. Parse incoming **Text** frames as `ResponsesWsRequest`; handle `response.create` - (`common.rs:273-274`). -3. Emit at least one event, starting with `response.created` (`sse/responses.rs:307-310`). -4. Emit exactly one **`response.completed`** to terminate success (`:747-750`); emit nothing - after it. -5. On failure, emit a classified `response.failed` (reuse 100.5 / 110 classifier). -6. Accept (may no-op) `response.processed` acks (`:219-241`). -7. Text-only frames; never send Binary (`:759-768`). -8. Tolerate client socket EOF as cancel; abort upstream work. -9. Optional but recommended upgrade-response headers: `openai-model`, `x-reasoning-included`. - -## 8. Gotchas - -- `response.completed` is mandatory + terminal; same invariant as the SSE bridge's RC1 - (`110/40_p0-implementation.md`) — the WS path can reuse the bridge's terminal-guarantee logic. -- Idle timeout is **per-message**, so a WS stall needs the same heartbeat strategy as RC3 — but - note Codex auto-Pongs server **Ping** frames (`:93-98`), so a WS keep-alive can be a real - `Ping` (cheaper than the SSE `response.heartbeat` workaround). -- Auth only at handshake — no per-frame token. opencodex's existing per-request auth must move - to upgrade time. -- `previous_response_id` + connection reuse means the WS endpoint is **stateful per connection** - (multiple `response.create` over one socket), unlike the stateless HTTP handler. -- `permessage-deflate` is negotiated; Bun's `ServerWebSocket` supports compression — verify it - is enabled or that uncompressed is accepted. diff --git a/devlog/_fin/120_codex-websocket-parity/02_transport-decision.md b/devlog/_fin/120_codex-websocket-parity/02_transport-decision.md deleted file mode 100644 index 588ffb121..000000000 --- a/devlog/_fin/120_codex-websocket-parity/02_transport-decision.md +++ /dev/null @@ -1,106 +0,0 @@ -# 120.02 — Transport Decision: Codex-Facing WS Bridge vs Native Upstream WS - -## The decision - -opencodex sits mid-chain: `Codex CLI → opencodex → upstream provider`. WS parity can be -implemented at one or both hops. Two strategies, not mutually exclusive: - -- **Option A — Codex-facing WS bridge (MVP).** Implement the WS *server* hop (Codex ↔ ocx). - Upstream stays HTTP/SSE for every provider. ocx accepts the WS upgrade, parses - `response.create`, runs the **existing** `handleResponses` pipeline, and emits the bridge's - Responses events as WS **Text** frames instead of SSE. -- **Option B — Native upstream WS (true parity).** For native `gpt-*` only, ocx also speaks WS - *upstream* to the ChatGPT backend (a Bun WS **client** re-implementing Codex's - `ResponsesWebsocketClient`), so the native path is end-to-end WS. - -**Recommendation: ship A first; treat B as an optional follow-up for native `gpt-*` only.** - -## Why A first - -- It satisfies Codex's capability requirement: once `supports_websockets` is advertised, Codex - opens WS and A answers it correctly for **every** model (native + routed), because the - server-side obligations (`01_§7`) are transport-agnostic and the *upstream* read is unchanged. -- It reuses everything already built: routing (`router.ts`), adapters, the bridge's terminal - guarantee (RC1), abort-on-disconnect (RC2), and error classification (100.5/110). The WS - endpoint is largely an **alternate emitter** over the same event stream. -- For routed models, A is the *only* possible WS (their upstream is HTTP/SSE — end-to-end WS is - structurally impossible, see `110/20`). A gives them a valid Codex-facing WS first hop. -- Effort: **2.5–4 days** (see breakdown in `10_`). - -## Why B is deferred (and native-only) - -- B reproduces Codex's `ResponsesWebsocketClient` in Bun: upstream WS connect, beta/auth headers - at handshake, `response.create`/`response.processed` framing, `permessage-deflate`, ping/pong, - close/idle semantics (`01_§3–§6`). High surface, real value only for native `gpt-*` (the only - upstream that actually speaks Responses WS). -- It buys end-to-end WS for native passthrough; it does **not** help routed models at all. -- Effort: **4–7 days** (see `11_`). Do it only if native transport parity is independently - wanted. - -## Side-by-side - -| Dimension | A — Codex-facing WS bridge | B — Native upstream WS | -|-----------|----------------------------|------------------------| -| Hop implemented | Codex ↔ ocx (server) | + ocx ↔ ChatGPT backend (client) | -| Covers routed models | yes (first hop only) | no (native only) | -| Reuses existing pipeline | fully | partially | -| New surface | Bun WS server + WS emitter | + Bun WS client, upstream handshake | -| End-to-end WS | no (upstream still SSE) | yes (native only) | -| Effort | 2.5–4 d | 4–7 d | -| Risk | low (alternate emitter) | medium (re-impl client) | - -## Integration point (feasibility) - -opencodex runs on `Bun.serve({ fetch })` (`/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:508-510`), -routing `POST /v1/responses` at `:547`. Bun's native WS support plugs in here without a new -dependency: - -```ts -// server.ts — inside Bun.serve fetch(), before the POST handler: -if (url.pathname === "/v1/responses" && server.upgrade(req, { data: { /* auth, headers */ } })) { - return; // upgraded to WS — handled by the `websocket` handler below -} -// ... -Bun.serve({ - fetch, - websocket: { - async message(ws, raw) { /* parse response.create → run pipeline → ws.send(event frames) */ }, - close(ws) { /* abort upstream (RC2 parity) */ }, - }, -}); -``` - -The same `parseRequest → routeModel → adapter → bridge` chain runs; only the **sink** changes -from an SSE `ReadableStream` to `ws.send(...)`. `10_` specifies extracting the bridge's event -generation so both SSE and WS share one emitter (no logic fork). - -## Capability-flag rollout (the 110 guard, operationalized) - -Today the served catalog advertises WS on no entry (verified, `00_overview.md`): routed is -stripped at `codex-catalog.ts:78`; native inherits none from the current template but *could* -leak it from a future template (no native strip). The rollout must therefore control **both** -paths — advertise intentionally when ready, and never let native leak it early. Sequence: - -1. A shipped (`10_`) + verified (`13_`) → advertise `supports_websockets = true` for routed + - native (A serves both). Done in `12_`. -2. If B is later shipped, native `gpt-*` upgrades to end-to-end WS transparently — the flag is - already on; only the upstream hop changes. -3. Never advertise before the endpoint exists (a failed upgrade = a new "RC6" stream error). - `12_` includes verifying Codex's HTTP fallback when WS is *not* advertised, as the safety net. - -## Reconciliation with 110.20 (amend, do not reverse) - -110.20 concluded WS does not improve routed **reliability or performance** — still true: A's -upstream read is the same SSE, so first-token latency/throughput are unchanged, and A adds a -protocol surface rather than removing an error class. 120 is justified on a **different axis**: -native transport **parity** and satisfying Codex's WS capability so `ocx` is a drop-in for the -native OpenAI provider. Action: add a one-line cross-link at the top of -`110/20_transport-evaluation.md` pointing here; do **not** delete its conclusion. - -## Decision record - -- **Adopt A as 120.2 (`10_`).** Build the Codex-facing WS server as an alternate emitter over - the existing pipeline. -- **Defer B to 120.3 (`11_`), native-only, optional.** -- **Gate the flag (`12_`/120.4)**; **live E2E (`13_`/120.5)**. -- Keep routed end-to-end WS permanently out of scope (impossible). diff --git a/devlog/_fin/120_codex-websocket-parity/10_ws-endpoint-mvp.md b/devlog/_fin/120_codex-websocket-parity/10_ws-endpoint-mvp.md deleted file mode 100644 index 51fa50aa7..000000000 --- a/devlog/_fin/120_codex-websocket-parity/10_ws-endpoint-mvp.md +++ /dev/null @@ -1,249 +0,0 @@ -# 120.10 — 120.2 WS Endpoint MVP (Option A: Codex-Facing WS Bridge) - -## Objective - -Add a Responses **WebSocket server** at `/v1/responses` so Codex's WS-first transport works -through `ocx` for **every** model. Strategy: **re-frame the existing SSE bridge output onto the -socket** — the WS path consumes the same `bridgeToResponsesSSE` / passthrough `ReadableStream` -and sends each frame's JSON as a WS Text message. This reuses RC1 (terminal guarantee), RC2 -(abort on disconnect), RC3 (heartbeat), and the 100.5/110 error classifier **unchanged**, so WS -and HTTP/SSE are guaranteed identical event semantics. No upstream change; the upstream read -stays HTTP/SSE. - -Satisfies the minimum server obligations in `01_codex-ws-protocol-analysis.md §7`. - -## Evidence - -opencodex pipeline + integration point: - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:87-223 handleResponses (parse→route→oauth→vision→web-search→adapter→bridge) -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:141-162 passthrough → Response(relayWithAbort(body)) (SSE bytes) -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:202-223 routed stream → bridgeToResponsesSSE(...) → SSE Response -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:508-510 Bun.serve({ fetch }) — add `websocket` handler here -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:547-559 POST /v1/responses — add WS upgrade beside it -``` - -Codex WS contract this satisfies: - -```text -01_codex-ws-protocol-analysis.md §4 client sends response.create / response.processed (Text) -01_codex-ws-protocol-analysis.md §5 server streams events; terminal = response.completed -01_codex-ws-protocol-analysis.md §6 text-only; client EOF = cancel; server Ping auto-Pong -``` - -> **Implementation note (as shipped):** the actual implementation used a lower-risk variant of -> the extraction below — the WS handler builds a synthetic `Request` and calls `handleResponses` -> **unchanged**, then re-frames its `Response.body` SSE onto the socket (cancelling the reader on -> close = RC2 abort). This avoids the `handleResponsesCore` refactor while achieving identical -> behavior. The `CoreResult` extraction (below) remains a valid alternative if the pipeline ever -> needs a non-Response return shape. Verified live: WS upgrade → `response.create` → real -> opencode-go bridge → `response.completed` (26 frames, content deltas present). - -## Design - -1. **Extract `handleResponsesCore`** from `handleResponses`: everything after the JSON body is - obtained becomes a function that returns a discriminated result instead of an HTTP `Response`. - `handleResponses` (HTTP) wraps it into a `Response` exactly as today; the WS handler consumes - the `stream` variant. -2. **Re-frame SSE → WS** in a new `src/ws-bridge.ts`: read the `ReadableStream`, split on - `\n\n`, send each frame's `data:` JSON as a Text message; drop `data: [DONE]` (WS terminal is - `response.completed`); the bridge's `response.heartbeat` frames re-frame as-is and re-arm - Codex's WS idle timer (`01_§5` unknown-type ignore). -3. **Wire Bun.serve**: upgrade `/v1/responses` WS requests; per-connection state runs the - pipeline per `response.create`; `close` aborts the upstream (RC2 parity). - -## Files - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/ws-bridge.ts -``` - -Complete content: - -```ts -import type { ServerWebSocket } from "bun"; - -export interface WsData { - headers?: Headers; // inbound upgrade headers, captured at upgrade and threaded to the pipeline - abort?: () => void; // set per-turn so close() can abort the upstream (RC2 parity) -} - -// Re-frame the existing SSE bridge/passthrough output onto a WebSocket. The frames' JSON already -// carries { type, sequence_number, ... }, so each is sent verbatim as a Text message. [DONE] is -// dropped (WS terminal is response.completed); response.heartbeat frames pass through and re-arm -// Codex's idle timer (unknown type → ignored, 01_§5). -export async function pumpSseToWebSocket( - ws: ServerWebSocket, - sseStream: ReadableStream, -): Promise { - const reader = sseStream.getReader(); - const decoder = new TextDecoder(); - let buffer = ""; - try { - while (true) { - const { done, value } = await reader.read(); - if (done) break; - buffer += decoder.decode(value, { stream: true }); - let idx: number; - while ((idx = buffer.indexOf("\n\n")) !== -1) { - const frame = buffer.slice(0, idx); - buffer = buffer.slice(idx + 2); - const dataLine = frame.split("\n").find(l => l.startsWith("data: ")); - if (!dataLine) continue; - const payload = dataLine.slice(6); - if (payload === "[DONE]") continue; // WS terminal is response.completed - if (ws.readyState === 1 /* OPEN */) ws.send(payload); - } - } - } finally { - reader.releaseLock(); - } -} -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts -``` - -**(a) Extract `handleResponsesCore`.** Change `handleResponses` to obtain the body, then delegate. -The existing pipeline body (`server.ts:95-223`) moves verbatim into `handleResponsesCore`, with -its `return new Response(...)` / `return formatErrorResponse(...)` sites returning a tagged -result instead: - -```ts -type CoreResult = - | { kind: "stream"; stream: ReadableStream; onAbort: () => void } - | { kind: "passthrough"; response: Response; onAbort: () => void } - | { kind: "json"; json: Record } - | { kind: "error"; status: number; type: string; message: string }; - -async function handleResponsesCore( - body: unknown, headers: Headers, config: OcxConfig, logCtx: { model: string; provider: string }, -): Promise { - // The current handleResponses body moves here verbatim (parse/route/oauth/vision/web-search/ - // adapter/bridge), with each `return` rewritten 1:1 by current line: - // :141-161 passthrough → return { kind: "passthrough", response: , onAbort: () => upstream.abort() }; - // :202-222 routed stream → return { kind: "stream", stream: sseStream, onAbort: () => upstream.abort() }; - // :223+ non-stream → return { kind: "json", json: buildResponseJSON(eventStream, parsed.modelId) }; - // every formatErrorResponse(status,type,msg) site → return { kind: "error", status, type, message: msg }; - // `req.headers` usages become the `headers` param. The web-search branch (:168-178) returns a - // Response today; wrap its result as { kind: "passthrough", response, onAbort } so WS gets it too. -} - -// HTTP keeps today's behavior: -async function handleResponses(req: Request, config: OcxConfig, logCtx: { model: string; provider: string }): Promise { - let body: unknown; - try { body = await req.json(); } catch { return formatErrorResponse(400, "invalid_request_error", "Invalid JSON body"); } - const r = await handleResponsesCore(body, req.headers, config, logCtx); - switch (r.kind) { - case "stream": return new Response(r.stream, { headers: { "Content-Type": "text/event-stream", "Cache-Control": "no-cache", "Connection": "keep-alive", "X-Accel-Buffering": "no" } }); - case "passthrough": return r.response; - case "json": return jsonResponse(r.json); - case "error": return formatErrorResponse(r.status, r.type, r.message); - } -} -``` - -> The HTTP responses are byte-identical to today — this is a pure extraction. Verify the existing -> suite stays green before adding WS. - -**(b) Add the WS upgrade + handler in `Bun.serve`** (`server.ts:508-565`): - -```diff -+import { pumpSseToWebSocket, type WsData } from "./ws-bridge"; -@@ - const server = Bun.serve({ - port: listenPort, - async fetch(req) { - const url = new URL(req.url); -@@ -+ // Responses WebSocket (phase 120). Codex upgrades the same /v1/responses path (01_§2). -+ if (url.pathname === "/v1/responses" && req.headers.get("upgrade")?.toLowerCase() === "websocket") { -+ // Capture inbound headers at upgrade (auth is handshake-time only on the WS path, 01_§3). -+ if (server.upgrade(req, { data: { headers: req.headers } })) return undefined as unknown as Response; -+ return formatErrorResponse(426, "upgrade_required", "WebSocket upgrade failed"); -+ } -@@ - return formatErrorResponse(404, "not_found", `Unknown endpoint: ${req.method} ${url.pathname}`); - }, -+ websocket: { -+ async message(ws: ServerWebSocket, raw: string | Buffer) { -+ let frame: { type?: string } & Record; -+ try { frame = JSON.parse(typeof raw === "string" ? raw : raw.toString()); } -+ catch { return; } // text-only contract; ignore unparseable -+ if (frame.type === "response.processed") return; // ack — no-op (01_§7.6) -+ if (frame.type !== "response.create") return; -+ const { type: _t, ...payload } = frame; // payload == Responses request body -+ const logCtx = { model: "unknown", provider: "unknown" }; -+ const r = await handleResponsesCore(payload, ws.data.headers ?? new Headers(), config, logCtx); -+ if (r.kind === "error") { -+ ws.send(JSON.stringify({ type: "response.failed", response: { status: "failed", error: { message: r.message, type: r.type, code: null } } })); -+ return; -+ } -+ if (r.kind === "json") { ws.send(JSON.stringify({ type: "response.completed", response: r.json })); return; } -+ const stream = r.kind === "stream" ? r.stream : r.response.body!; -+ ws.data.abort = r.onAbort; -+ await pumpSseToWebSocket(ws, stream); -+ }, -+ close(ws: ServerWebSocket) { ws.data.abort?.(); }, // RC2: abort upstream on client disconnect -+ }, - }); -``` - -> **Auth (`01_§3`):** WS auth is handshake-time only — there is no per-frame token. The upgrade -> captures `req.headers` into `ws.data.headers`, and the message handler threads them into -> `handleResponsesCore` (exactly as the HTTP path passes `req.headers`). In practice routed -> providers authenticate via the configured `apiKey`/OAuth **inside** the pipeline, so inbound -> headers are usually unused; capturing them keeps parity with HTTP and covers the rare provider -> that reads an inbound header. Native end-to-end upstream auth is `11_`. - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/tests/ws-endpoint.test.ts -``` - -Integration test: start the server on an ephemeral port, open a `WebSocket` to -`ws://127.0.0.1:/v1/responses`, send a `response.create` for a stub/echo routed model, and -assert the received frames include `response.created` … exactly one `response.completed`, and no -`[DONE]` text frame. (Use a stubbed adapter or a local fake upstream so no paid call is made.) - -## Verification - -```bash -bun test tests/ws-endpoint.test.ts -bun test tests # HTTP suite must stay green (extraction is behavior-preserving) -bun x tsc --noEmit -git diff --check -``` - -Expected: - -```text -WS: response.created → … → exactly one response.completed; no [DONE] frame; close aborts upstream -HTTP suite unchanged (handleResponsesCore extraction is byte-identical) -typecheck clean -``` - -## Effort - -~2.5–4 days: extraction (0.5–1d) + ws-bridge + handler (1–1.5d) + tests + edge cases -(close/cancel, multi-turn over one socket) (1–1.5d). - -## Commit - -```text -[agent] feat: Responses WebSocket endpoint (MVP, re-frames SSE bridge onto WS) -``` - -## Out of scope (this sub-phase) - -- Advertising `supports_websockets` (stays deleted at `codex-catalog.ts:78` until `12_`). -- Native upstream WS (`11_`). -- Real WS `Ping` keep-alive (the re-framed `response.heartbeat` already re-arms the timer; Ping - is an optional optimization noted in `01_§8`). diff --git a/devlog/_fin/120_codex-websocket-parity/11_native-upstream-ws.md b/devlog/_fin/120_codex-websocket-parity/11_native-upstream-ws.md deleted file mode 100644 index e3fa4d72b..000000000 --- a/devlog/_fin/120_codex-websocket-parity/11_native-upstream-ws.md +++ /dev/null @@ -1,89 +0,0 @@ -# 120.11 — 120.3 Native Upstream WebSocket (Option B, deferred / native-only) - -## Status: DEFERRED, OPTIONAL - -This sub-phase delivers **end-to-end** WS for native `gpt-*` only: `Codex ⇄ ocx ⇄ ChatGPT -backend`, all WS. It is **not required** for WS parity — the MVP (`10_`) already answers Codex's -WS with a Codex-facing bridge over an HTTP/SSE upstream read. Do B only if native transport -parity (true upstream WS, no SSE re-encode for native) is independently wanted. Routed models -are permanently excluded (their upstream is HTTP/SSE). - -## Objective - -For native `gpt-*` (the passthrough path), connect to the upstream ChatGPT backend over WS using -the same protocol Codex uses (`01_codex-ws-protocol-analysis.md`), and forward the upstream -Responses event frames straight to the Codex-facing WS from `10_` — no SSE conversion on the -native path. - -## Preconditions (MUST resolve before implementing — not derivable from static source) - -These could **not** be pinned from the codex checkout and require a live capture (run the real -Codex against the native backend with WS on, capture the upgrade + first frames): - -1. **Exact `OpenAI-Beta` WS header value.** `core/src/client.rs:912-914` references - `RESPONSES_WEBSOCKETS_V2_BETA_HEADER_VALUE`; the literal was not resolvable by static grep. - Capture it from a live native WS handshake. -2. **Upstream WS URL.** Confirm the ChatGPT backend WS endpoint reachable with the user's OAuth - (scheme `wss`, path `/responses` per `provider.rs:92-103`), and whether the OAuth bearer is - accepted at WS upgrade. -3. **`permessage-deflate`.** Confirm the backend negotiates it and Bun's WS client supports the - negotiated extension (`01_§8`). - -> Record the captures in a `12_`-adjacent note before coding B. If any precondition fails (e.g. -> the backend rejects third-party WS clients), **B is infeasible** and the MVP (`10_`) stands as -> the final native answer — state that and stop. - -## Design - -A Bun WebSocket **client** mirroring `ResponsesWebsocketClient`: - -1. On a native `response.create` arriving at the Codex-facing WS (`10_`), open (or reuse) an - upstream WS via `new WebSocket(wssUrl, { headers: { Authorization, "OpenAI-Beta": , … } })` - — Bun supports a headers option on the client constructor. -2. Forward the `response.create` payload upstream as a Text frame (`01_§4`). -3. Pipe every upstream event frame **verbatim** to the Codex-facing `ws.send` (native frames are - already Responses events — no translation). -4. Map upstream `response.completed` → terminal; upstream close/idle/error → the same - `ApiError`-equivalent classified failure the bridge emits (reuse `errors.ts`). -5. Reply to upstream `Ping` with `Pong` (Bun client auto-handles; verify). -6. **Fallback:** if the upstream WS upgrade fails, fall back to the existing HTTP/SSE passthrough - (`server.ts:141-162`) for that turn — never hard-fail a native turn on a WS problem. - -## Files - -### NEW - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/adapters/openai-responses-ws.ts -``` - -An upstream WS client: `connect(provider, authHeaders) → { send(frame), events: AsyncIterable }`, -plus close/idle/error → classified terminal. Mirrors `01_§3-§6`. (Full content authored after the -preconditions are captured — the handshake headers are the only unknowns; the framing is fixed by -`01_§4-§5`.) - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts -``` - -In the WS `message` handler (`10_`), branch native vs routed: native + WS-upstream-available → -use `openai-responses-ws`; else → the `10_` re-frame path. Keep the HTTP/SSE fallback. - -## Verification - -- Unit: upstream-WS client framing + terminal/close mapping against recorded fixtures. -- Live: native `gpt-5.5` turn over end-to-end WS; confirm no SSE re-encode on the native path - (instrument the passthrough to assert the WS-upstream branch ran); fallback fires when the - upstream WS is forced to fail. - -## Effort - -~4–7 days, dominated by precondition capture + handshake correctness + fallback design. - -## Commit - -```text -[agent] feat: native upstream WebSocket passthrough (end-to-end WS for gpt-*) -``` diff --git a/devlog/_fin/120_codex-websocket-parity/12_metadata-enable-fallback.md b/devlog/_fin/120_codex-websocket-parity/12_metadata-enable-fallback.md deleted file mode 100644 index 22e496213..000000000 --- a/devlog/_fin/120_codex-websocket-parity/12_metadata-enable-fallback.md +++ /dev/null @@ -1,127 +0,0 @@ -# 120.12 — 120.4 Capability Flag Enable + HTTP Fallback - -## Objective - -Once the MVP endpoint (`10_`) is live and tested, advertise `supports_websockets` **intentionally -and centrally**, so Codex opens WS against `ocx`. Today the flag leaks through two different -paths (`00_overview.md`): routed is stripped at `codex-catalog.ts:78`; native simply inherits -none from the current template but **could** leak it from a future template (no native strip). -This sub-phase puts both under one config-gated switch and verifies Codex's HTTP fallback as the -safety net. - -## Evidence - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:78 routed strip (delete supports_websockets) -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:145-154 deriveEntry — native clone (no strip → latent leak) -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:195 buildCatalogEntries(template, gptSlugs, goModels, featured) -/Users/jun/Developer/codex/codex-cli/codex-rs/core/src/client.rs:772 Codex WS selection on supports_websockets -``` - -Verified current state: served catalog `/Users/jun/.codex/opencodex-catalog.json` has zero -`supports_websockets` (so Codex attempts no WS today). - -## Design - -Add a single config switch and a **central override** in `buildCatalogEntries`, so every emitted -entry's flag is set deterministically (overriding both the routed strip and any native template -leak): - -- `config.websockets?: boolean` (default `false`) in `OcxConfig`. -- `buildCatalogEntries(..., wsEnabled)`: after each entry is derived, `if (wsEnabled) - entry.supports_websockets = true; else delete entry.supports_websockets;` — for **native and - routed alike**. This makes the advertised capability match the actually-implemented endpoint - and closes the native-leak risk in one place. - -## Files - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/types.ts (or wherever OcxConfig is defined) -``` - -```diff - export interface OcxConfig { - // … existing fields … -+ /** Advertise supports_websockets so Codex opens the WS endpoint (120). Default false until 120.2 ships. */ -+ websockets?: boolean; - } -``` - -### MODIFY - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts -``` - -Thread a `wsEnabled` flag into `buildCatalogEntries` and apply the central override: - -```diff --export function buildCatalogEntries(template: RawEntry | null, gptSlugs: string[], goModels: CatalogModel[], featured?: string[]): RawEntry[] { -+export function buildCatalogEntries(template: RawEntry | null, gptSlugs: string[], goModels: CatalogModel[], featured?: string[], wsEnabled = false): RawEntry[] { - // … existing derivation of native + routed entries into `entries` … -+ // Central capability override: the advertised flag must match the implemented endpoint (120). -+ // Overrides both the routed strip (:78) and any native template leak (:145-154). -+ for (const entry of entries) { -+ if (wsEnabled) entry.supports_websockets = true; -+ else delete entry.supports_websockets; -+ } - return entries; - } -``` - -(`delete entry.supports_websockets` at `:78` may stay as defense-in-depth; the central override -is authoritative.) - -### MODIFY — call sites pass the flag - -Both `buildCatalogEntries` call sites (verified): - -```text -/Users/jun/Developer/new/700_projects/opencodex/src/server.ts:537 (/v1/models codex catalog response) -/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts:330 (on-disk catalog injection, goEntries) -``` - -```diff -- return jsonResponse({ models: buildCatalogEntries(loadCatalogTemplate(), nativeSlugs, goOrdered, config.subagentModels) }); -+ return jsonResponse({ models: buildCatalogEntries(loadCatalogTemplate(), nativeSlugs, goOrdered, config.subagentModels, config.websockets ?? false) }); -``` - -The `codex-catalog.ts:330` injection call (`goEntries = buildCatalogEntries(...)`) needs the same -5th-arg addition; thread `config.websockets` into that function's caller. - -## Verification - -```bash -bun test tests/codex-catalog.test.ts -bun test tests -bun x tsc --noEmit -``` - -Add catalog tests: - -- `websockets: false` (default) → no entry has `supports_websockets` (native **or** routed), - even if the template carries it (inject a template with the flag and assert it is removed). -- `websockets: true` → every served entry has `supports_websockets: true`. - -**Live fallback check (the safety net):** - -1. `websockets: false`, regenerate the served catalog → confirm `/Users/jun/.codex/opencodex-catalog.json` - has no flag and Codex uses HTTP (no WS attempt). -2. `websockets: true` with the `10_` endpoint running → Codex opens WS and a turn completes. -3. `websockets: true` with the endpoint **stopped** → confirm Codex falls back to HTTP (or fails - cleanly) rather than hanging; record the behavior. This is the RC6 guard rail. - -Expected: - -```text -flag off → zero advertisement (native+routed); flag on → uniform advertisement -endpoint up + flag on → WS turn completes; endpoint down → HTTP fallback (no hang) -``` - -## Commit - -```text -[agent] feat: gate supports_websockets advertisement behind config.websockets -``` diff --git a/devlog/_fin/120_codex-websocket-parity/13_e2e-live-verification.md b/devlog/_fin/120_codex-websocket-parity/13_e2e-live-verification.md deleted file mode 100644 index 3f3239f5d..000000000 --- a/devlog/_fin/120_codex-websocket-parity/13_e2e-live-verification.md +++ /dev/null @@ -1,106 +0,0 @@ -# 120.13 — 120.5 Live End-to-End WebSocket Verification - -## Objective - -The acceptance gate for phase 120: a real Codex CLI driving `ocx` over **WebSocket** for both a -native `gpt-*` and a routed model, across a multi-turn session with interrupts, stalls, and tool -calls — confirming WS parity with the HTTP/SSE path and no new error class (RC6). Verification -plan, not code. Runs after `10_` (+ optionally `11_`) and `12_` are implemented. - -## Preconditions - -- `10_` WS endpoint implemented; `12_` flag enable implemented. -- `config.websockets = true`; served catalog regenerated (verify `supports_websockets: true` in - `/Users/jun/.codex/opencodex-catalog.json`). -- `ocx` running; Codex CLI installed. - -## Codex configuration - -```toml -# native gpt over WS: -model_provider = "opencodex" -model = "gpt-5.5" - -[model_providers.opencodex] -base_url = "http://localhost:10100/v1" # Codex derives ws://localhost:10100/v1/responses (01_§2) -wire_api = "responses" -requires_openai_auth = true -# supports_websockets now advertised by the catalog (12_), so Codex opens WS. -``` - -Run the routed scenarios by switching `model = "opencode-go/deepseek-v4-pro"`. - -## Scenarios (each must hold over WS) - -1. **Native long answer.** `gpt-5.5`, multi-paragraph reply. Expect: WS upgrade succeeds (101), - one `response.completed`, no `ApiError::Stream`. -2. **Routed long answer.** `opencode-go/deepseek-v4-pro` over WS (MVP re-frame path). Same - expectations — proves Option A serves routed over WS. -3. **Interrupt mid-stream.** Interrupt a long generation, send a new turn. Expect: new turn - answers; the WS `close`/cancel aborts the upstream (no leaked connection); same socket may be - reused for the next turn (`01_§4` multi-turn). -4. **Stall.** Slow routed model / long reasoning gap. Expect: no `"idle timeout waiting for - websocket"` — the re-framed `response.heartbeat` (or a real Ping, if added) re-arms Codex's - idle timer. -5. **Tool round-trip.** A turn with a function/tool call + follow-up over WS. Expect: tool call - commits, conversation continues. -6. **Parity diff.** Run the same prompt over HTTP (`base_url` HTTP, WS off) and WS; expect - semantically identical event sequences (created → items → completed) — confirms the - re-frame is faithful. - -## Observation methods - -```bash -# WS upgrade actually happened (not silent HTTP fallback): -# confirm a 101 upgrade in ocx logs, or capture with: lsof -iTCP -sTCP:ESTABLISHED | grep 10100 -# Codex client-side stream errors (must be empty): -# scan Codex log/TUI for: ApiError::Stream, "websocket closed by server before response.completed", -# "idle timeout waiting for websocket", "unexpected binary websocket event". -# Leaked upstream after interrupt (return to baseline): -lsof -p "$(pgrep -f 'ocx|opencodex' | head -1)" 2>/dev/null | grep -c ESTABLISHED -``` - -## Pass criteria - -| # | Criterion | Verified by | -|---|-----------|-------------| -| 1 | WS upgrade succeeds (HTTP 101) for native + routed | ocx log / lsof | -| 2 | Exactly one `response.completed` per turn; zero `ApiError::Stream` | Codex log scan | -| 3 | Interrupt aborts upstream; ESTABLISHED returns to baseline | `lsof` before/after (scenario 3) | -| 4 | No `"idle timeout waiting for websocket"` on the stall scenario | Codex log scan (scenario 4) | -| 5 | Tool round-trip completes over WS | Codex session | -| 6 | WS and HTTP event sequences are semantically identical | side-by-side (scenario 6) | -| 7 | No binary frame / no `"unexpected binary websocket event"` | Codex log scan | - -## Recording - -Append a results block (date, codex version, models, pass/fail per criterion) and flip phase 120 -status to **verified** in `00_overview.md`. If WS underperforms or regresses vs HTTP, the MVP can -be shipped with the flag **off** by default (HTTP remains the supported path) until issues are -resolved — `12_`'s flag makes that a config toggle, not a code revert. - -## Results — executed 2026-06-20 (120.2 + 120.4 PASS) - -Live runs against the real `opencode-go` upstream (saved token), isolated repo build on port -10199; codex catalog backed up and restored around the flag tests. - -| Check | Result | -|-------|--------| -| WS upgrade + full data plane (120.2) | PASS — `ws://…/v1/responses` → `response.create` → real opencode-go bridge → `response.completed` (26 frames, content deltas present), no server crash | -| WS client-disconnect (RC2 over WS) | PASS — covered by `ws-endpoint` unit test (cancel hook aborts the reader); HTTP-path RC2 validated live in `110/55` | -| Flag OFF → on-disk catalog (the file Codex reads) | PASS — native `gpt-5.5` **and** routed `opencode-go/kimi-k2.7-code` both `supports_websockets` ABSENT → Codex uses HTTP (native-leak closed) | -| Flag ON → on-disk catalog | PASS — native **and** routed both `supports_websockets=true` → Codex opens WS | -| HTTP path still served (coexistence/fallback) | PASS — HTTP `/v1/responses` bridge validated live in `110/55`; WS is additive | - -Note: the flag is applied to the **on-disk** `~/.codex/opencodex-catalog.json` written by -`syncCatalogModels` (the file Codex reads via `model_catalog_json`), not only the `/v1/models` -HTTP response — both paths now honor `config.websockets`. **120.2 + 120.4 status: verified.** - -Deferred (by plan): driving the actual Codex CLI binary end-to-end over WS (gold-standard manual -check — enable `config.websockets`, restart `ocx`, run a routed turn) and native upstream WS -(`11_`, optional). - -## Non-goals - -- Not a throughput benchmark (parity/correctness only). -- Native end-to-end upstream WS (`11_`) is verified separately if that sub-phase ships. diff --git a/devlog/_fin/120_desktop_3p_alias_spec.md b/devlog/_fin/120_desktop_3p_alias_spec.md deleted file mode 100644 index e929ddf42..000000000 --- a/devlog/_fin/120_desktop_3p_alias_spec.md +++ /dev/null @@ -1,246 +0,0 @@ -# Claude Desktop 3P short alias specification - -## Decision - -Use a three-character, letter-first base36 code derived from SHA-256 of the -canonical route key, and keep a small persisted reverse registry. Expose routed -models as: - -```text -claude-{tier}-4-{code} -``` - -For example, `native/gpt-5.6-sol` is exposed as -`claude-opus-4-bji`. The discovery `display_name` remains honest, for example -`GPT-5.6-Sol (native)`. - -This replaces the long `claude-ocx-{provider}--{model}` form only on the Claude -Desktop 3P / Claude Code discovery surface. The old form should remain accepted -inbound during migration because it may already be stored in Claude Code -settings. - -Why this choice: - -- A two-character base36 space has only 1,296 values and about a 61% chance of - at least one collision at 50 models. It is not adequate. -- A three-character letter-first base36 space has 33,696 values and about a - 3.6% collision chance at 50 models. It still looks like a plausible Claude - revision token and is small enough for the filter-sensitive surface. -- Hashing is stable under unrelated additions and removals. Alphabet and - phonetic compression are either longer or unstable when model naming - conventions change. Sorted index plus salt directly violates stability. -- A lookup table is required anyway: no 2-3 character code can reversibly - contain an arbitrary `provider/id` string. - -## Types and function signatures - -```ts -export type ClaudeAliasTier = "opus" | "sonnet" | "haiku"; - -export interface ClaudeAliasRecord { - route: string; // canonical provider/model key, including native/... - code: string; // /^[a-z][0-9a-z]{2}$/ - tier: ClaudeAliasTier; - aliases?: string[]; // retained aliases after an explicit tier migration -} - -export interface ClaudeAliasRegistry { - version: 1; - records: Record; // keyed by canonical route -} - -export function deriveClaudeAliasCode(route: string): string; -export function resolveClaudeAliasTier( - route: string, - metadata: ModelCapabilityMetadata | undefined, - config: OcxConfig, -): ClaudeAliasTier; -export function aliasForClaude3p( - route: string, - registry: ClaudeAliasRegistry, -): string | null; -export function resolveClaude3pAlias( - alias: string, - registry: ClaudeAliasRegistry, -): string | null; -``` - -`route` is always the internal canonical route key. Native OpenAI models use the -pseudo-provider form `native/{slug}` in the registry, even though decoding them -returns the bare slug expected by `routeModel`. Routed model IDs may themselves -contain `/`; split a route only at its first `/` when provider and model need to -be separated. - -Do not trim, lowercase, Unicode-normalize, or otherwise rewrite a route before -hashing. Route construction must already have produced the exact canonical -provider and model IDs. Reject empty providers, empty model IDs, control -characters, and non-canonical duplicate spellings before alias generation. - -## Code derivation algorithm - -`deriveClaudeAliasCode(route)` is defined exactly as follows: - -1. Encode the canonical `route` as UTF-8. -2. Compute SHA-256. -3. Interpret the complete 32-byte digest as one unsigned big-endian integer - `h`. -4. Compute `n = h mod 33_696` (`26 * 36 * 36`). -5. The first character is ASCII `a + floor(n / 1_296)`. -6. Encode `n mod 1_296` as two lowercase base36 characters, left-padded with - `0`. - -The letter-first restriction makes codes such as `bji`, `k4u`, and `soa` look -more like model revisions than bare numeric suffixes. SHA-256 is available from -`node:crypto`, requires no dependency, and the full algorithm must have golden -tests so runtime or language changes cannot alter it. - -`aliasForClaude3p` behavior: - -1. If the route is `anthropic/{model}` and `{model}` starts with `claude-`, - return `{model}` unchanged. Real Anthropic Claude IDs already satisfy both - clients and must not be hidden behind an alias. -2. Otherwise find the pinned record for the route and return - `claude-${record.tier}-4-${record.code}`. -3. Return `null` if the route is invalid, unregistered, or has an unresolved - collision. Discovery omits such an entry and emits one actionable warning. - -A Claude model reached through a non-Anthropic route, such as -`kiro/claude-opus-4.6`, is encoded. Passing it through would lose the provider -identity on inbound routing. - -## Example encodings - -These examples use the exact hash algorithm above. Tier values illustrate the -policy in the next section. - -| Canonical route | Desktop / Code model ID | Inbound result | -| --- | --- | --- | -| `native/gpt-5.6-sol` | `claude-opus-4-bji` | `gpt-5.6-sol` | -| `native/gpt-5.6-terra` | `claude-sonnet-4-k4u` | `gpt-5.6-terra` | -| `native/gpt-5.6-luna` | `claude-haiku-4-f7l` | `gpt-5.6-luna` | -| `opencode-go/glm-5.2` | `claude-sonnet-4-kgz` | `opencode-go/glm-5.2` | -| `zai/glm-5.2` | `claude-sonnet-4-soa` | `zai/glm-5.2` | -| `openrouter/openai/gpt-5.6-sol` | `claude-opus-4-zuw` | `openrouter/openai/gpt-5.6-sol` | -| `kiro/claude-opus-4.6` | `claude-opus-4-zzn` | `kiro/claude-opus-4.6` | -| `anthropic/claude-opus-4-6` | `claude-opus-4-6` (pass-through) | `anthropic/claude-opus-4-6` via the normal Anthropic route | - -The fixed `4` is a compatibility marker, not an encoded model generation. It -should remain fixed for registry version 1; changing it would invalidate saved -client selections. - -## Tier assignment - -Tier is capability metadata, not part of the hash. Resolve it once when a -record is first created, then pin it in the registry so metadata updates do not -silently rename an alias saved by a client. - -Precedence, highest first: - -1. **Per-model config override:** `claudeCode.aliasTiers[route]`. This is the - authoritative escape hatch for operators and custom providers. -2. **Declared provider/model capability:** a curated `claudeAliasTier` on model - metadata. Built-in routes should use this rather than heuristics. Providers - with a documented capability class may supply a provider default, overridden - by model metadata. -3. **Name heuristic fallback:** exact hyphen/dot/underscore-delimited tokens - `opus`, `pro`, `max`, `ultra`, `frontier`, or `sol` imply `opus`; tokens - `haiku`, `mini`, `nano`, `lite`, `flash`, or `luna` imply `haiku`; all other - models use `sonnet`. If both groups match, choose `sonnet` and warn because - the name is ambiguous. - -Reasoning-effort ladders, context-window size, and vision support must not alone -determine tier; they are not reliable proxies for overall capability. The -heuristic is only a discovery fallback. The generated display name always shows -the real model and provider, so the tier is not presented as vendor identity. - -An intentional tier change is a migration: add the newly tiered alias to the -record and retain the prior full alias in `record.aliases` for inbound decoding. -Only the new alias is advertised. Do not recycle old aliases. - -## Registry and collision handling - -Keep built-in records in a checked-in versioned manifest so every installation -uses the same assignments. Merge custom/live-model records into a runtime file -under the opencodex config directory, written atomically. The runtime registry -is portable configuration and should survive upgrades; removals become -tombstones rather than freeing an alias for reuse. - -At registry build time, construct both `route -> record` and `full alias -> -route` indexes and reject duplicates. Codes should be globally unique across -all tiers, even though differing tiers would technically make full aliases -different. Global uniqueness prevents a later tier migration from exposing a -latent collision. - -If two routes derive the same primary code: - -1. Never renumber or re-tier an existing record. -2. Built-in collisions receive an explicit checked-in code override reviewed - with the manifest change. -3. A new custom/live route is not advertised until it has a unique explicit - `claudeCode.aliasCodes[route]` override matching - `/^[a-z][0-9a-z]{2}$/`; persist that override in the runtime registry. -4. Reject an override already owned by any active record or tombstone. - -This explicit handling is deliberate. Automatically selecting “the next free -hash” would make a model's result depend on discovery order or on which other -models happen to be installed, violating determinism. With only 33,696 possible -codes, universal collision-free encoding is mathematically impossible; the -registry is the contract that resolves the rare exception without moving old -aliases. - -## Inbound decoding - -Build the reverse index at startup from active records, tombstones, and each -record's legacy `aliases`. For a Messages API request model: - -1. Try exact lookup of the complete incoming ID in the reverse index. Do not - parse tier or code independently and do not accept near matches. -2. For `native/...`, return the model portion after the first `/` as a bare - native slug. For every other record, return the canonical route unchanged. -3. If no short alias matches, run the existing `resolveAlias` decoder for the - legacy `claude-ocx-...` format. -4. Then retain the current `claudeCode.modelMap` exact and date-suffix-stripped - fallbacks. -5. Otherwise pass the model through unchanged, preserving real Claude IDs. - -Exact full-ID lookup prevents a request such as `claude-haiku-4-bji` from being -accepted when the registered alias is `claude-opus-4-bji`. It also makes tier -migrations and tombstones explicit instead of relying on lossy suffix parsing. - -## Edge cases and operational rules - -- **Removed/disabled model:** stop advertising it. Keep its reverse record so a - saved selection decodes deterministically; normal routing then returns the - existing unavailable/disabled-model error. Never assign its alias to another - route. -- **Provider changes:** provider is part of the hash and route identity, so the - same model ID through two providers intentionally receives different codes. -- **Nested model IDs:** `openrouter/openai/gpt-5.6-sol` is valid; provider is - `openrouter`, model ID is `openai/gpt-5.6-sol`, split on the first slash only. -- **Case:** route IDs are case-sensitive. `MiniMax-M3` and `minimax-m3` are - distinct unless the owning provider canonicalizes them before this layer. -- **Malformed input:** reject whitespace-only components, control characters, - invalid code overrides, duplicate full aliases, and registry versions newer - than the implementation understands. -- **Concurrent writes:** update the runtime registry with lock + temporary file - + atomic rename. Re-read and revalidate under the lock before committing. -- **Corrupt/missing registry:** built-ins can be rebuilt from the checked-in - manifest. Do not regenerate custom collision overrides silently; warn and - omit affected aliases until the portable registry/config is restored. -- **Client compatibility:** return the same short IDs from both the Desktop 3P - and Claude Code Anthropic-flavored `/v1/models` response. Keep truthful - `display_name` values and do not use short aliases on OpenAI/Codex catalogs. - -## Required verification - -- Golden vectors for every example above, including UTF-8 hashing semantics. -- Round trips for native, routed, nested-model-ID, and non-Anthropic Claude - routes. -- Anthropic pass-through tests. -- Corpus test asserting unique codes and full aliases for every built-in route. -- Addition/removal test proving existing records and aliases are byte-identical. -- Collision fixture proving the newcomer is omitted until an explicit unique - override is supplied and that the incumbent never changes. -- Tier migration test proving only the new alias is advertised while both old - and new aliases decode. -- Legacy `claude-ocx-...` and `modelMap` precedence regression tests. diff --git a/devlog/_fin/120_desktop_3p_aliases.md b/devlog/_fin/120_desktop_3p_aliases.md deleted file mode 100644 index 2c8257407..000000000 --- a/devlog/_fin/120_desktop_3p_aliases.md +++ /dev/null @@ -1,48 +0,0 @@ -# 120 — Desktop 3P Compatible Aliases - -## Problem - -Claude Desktop 3P mode filters `/v1/models` to only show "recognizably Claude" models. -The existing `claude-ocx-{provider}--{model}` aliases are filtered out because Desktop -recognizes the underlying model name (gpt, glm, grok) as non-Claude. - -## Goal - -Generate model IDs that: -- Look like Claude version strings → pass Desktop's model guard -- Encode the actual routed model deterministically -- Are stable across model additions/removals -- Work for both Desktop 3P auto-discovery AND Claude Code CLI -- Coexist with existing `claude-ocx-*` aliases (CLI keeps working) - -## Discovery (2026-07-12) - -- Desktop 3P GA: July 9, 2026 -- Desktop uses `GET /v1/models` for auto-discovery -- Filters by "recognizably Claude" (stricter than `claude-` prefix) -- `inferenceModels` config overrides discovery (manual, user tested working) -- Manual entries accept opaque IDs with `anthropicFamilyTier` metadata -- User confirmed: `opus-1` with display name "GPT 5.6 Sol" works end-to-end - -## Encoding Spec - -*(Pending: sol agent Goodall writing detailed spec at 120_desktop_3p_alias_spec.md)* - -Key requirements: -- Input: `provider/modelId` (e.g., `native/gpt-5.6-sol`) -- Output: `claude-{tier}-4-{code}` (e.g., `claude-opus-4-s0`) -- Deterministic, stable, collision-resistant for ~50 models -- Tier (opus/sonnet/haiku) assigned by model capability -- Real Claude models pass through without encoding - -## Implementation Plan - -| File | Change | -|------|--------| -| `src/claude/alias.ts` | Add Desktop alias generation + decode | -| `src/claude/inbound.ts` | `resolveInboundModel` handles new format | -| `src/server/index.ts` | `/v1/models` returns Desktop aliases for 3P clients | -| `src/server/management-api.ts` | GUI export endpoint for Desktop 3P config | -| `gui/src/pages/ClaudeCode.tsx` | "Desktop 3P 설정 내보내기" button | -| `tests/` | Alias generation + decode tests | -| `docs/` | Desktop 3P integration guide | diff --git a/devlog/_fin/130_provider-catalog-single-source/00_overview.md b/devlog/_fin/130_provider-catalog-single-source/00_overview.md deleted file mode 100644 index 792bc5dfc..000000000 --- a/devlog/_fin/130_provider-catalog-single-source/00_overview.md +++ /dev/null @@ -1,92 +0,0 @@ -# 130.00 — Overview: Provider Catalog Single-Sourcing - -## What this phase is - -opencodex advertises and configures LLM providers through **three hand-maintained catalogs** -that are supposed to stay aligned but are edited independently. When they drift, users see a -provider in the GUI that `ocx init` cannot offer, or a configured provider that never receives -bundled jawcode model metadata in the Codex catalog. - -Docs 00–02 inventory every catalog entry across the three surfaces, document the two hotfixed -drift bugs from this work cycle as motivating symptoms, and design a single-source-of-truth -registry with an incremental migration path and a CI drift guard. Docs 10–30 record the Phase -130 implementation that followed. - -## Symptom (motivating bugs — already hotfixed) - -Two independent drift bugs were discovered and patched in the same work cycle. They illustrate -why manual triple-maintenance fails: - -| ID | Symptom | Root cause | Hotfix (this cycle) | -|----|---------|------------|---------------------| -| **BUG A** | `opencode-go` appeared in the GUI quick-pick (`AddProviderModal.tsx:35`) and had bundled metadata (`PROVIDER_ALIASES` → `opencode-go`), but **`ocx init` could not offer it** and `enrichProviderFromCatalog("opencode-go", …)` was a no-op | GUI `PRESETS` + metadata alias existed; `KEY_LOGIN_PROVIDERS` entry was missing | Added `opencode-go` to `KEY_LOGIN_PROVIDERS` (`key-providers.ts:75-77`) | -| **BUG B** | `minimax` / `minimax-cn` were in `KEY_LOGIN_PROVIDERS` (`key-providers.ts:71-72`) but **had no metadata alias** → no bundled jawcode rows; jawcode also uses different provider id naming (`minimax` / `minimax-code` / `minimax-code-cn` vs opencodex `minimax` / `minimax-cn`) | Metadata generator never mapped opencodex ids to jawcode's `minimax` bundle | Added `minimax` / `minimax-cn` → `minimax` in `PROVIDER_ALIASES` (`generate-jawcode-metadata.ts:15-16`) and regenerated `jawcode-model-metadata.ts` | - -BUG A violated the explicit **GUI↔CLI parity contract** in `init.ts`. BUG B broke the metadata -enrichment path in `codex-catalog.ts:90-99`. - -## The parity contract (and how drift violates it) - -`buildInitProviders()` documents that the CLI init menu is assembled from the **same registries -the GUI uses**, plus a small hardcoded set: - -```33:36:src/init.ts -/** - * The full CLI provider menu, built from the SAME registries the GUI uses (OAUTH_PROVIDERS + - * KEY_LOGIN_PROVIDERS) plus the ChatGPT-forward, a few non-catalog key providers, and local servers — - * so `ocx init` reaches provider parity with the GUI. Exported for verification. - */ -``` - -In practice the GUI also has a **second surface**: static `PRESETS` in -`AddProviderModal.tsx:29-43` (14 quick-pick rows including `custom`) that are **not** all -re-exported from `KEY_LOGIN_PROVIDERS` or `OAUTH_PROVIDERS`. The modal merges static presets -with `/api/key-providers` at runtime (`AddProviderModal.tsx:86-100`). Metadata aliases form a -**third** surface (`generate-jawcode-metadata.ts:4-17` → `codex-catalog.ts:93-99`). None of -the three is derived from a shared canonical list today. - -## TL;DR - -1. **Three independent catalogs** today: GUI static `PRESETS` + runtime key-catalog fetch; - CLI/registry (`OAUTH_PROVIDERS`, hardcoded init rows, `KEY_LOGIN_PROVIDERS`, - `buildInitProviders()`); bundled metadata (`PROVIDER_ALIASES` → generated - `jawcode-model-metadata.ts`). -2. **BUG A / BUG B** (hotfixed) prove the contract in `init.ts:33-36` is not enforceable without - automation — manual edits to one surface silently break the others. -3. **Full audit** (`01_catalog-source-audit.md`) finds **additional live mismatches** beyond those - bugs: GUI vs OAuth/init field disagreements for `kimi`, `anthropic`, `azure-openai`; GUI - offline degradation (static presets only); 28/31 key-login ids with no metadata alias; - model-id **casing fragility** for minimax (`MiniMax-M2.5` in metadata vs lowercase routed ids). -4. **Design** (`02_single-source-design.md`): one canonical provider registry; derive GUI - presets, init menu, key-login export, and metadata alias map; incremental migration in four - steps; CI test that fails when derived surfaces diverge. - -## The three surfaces (at a glance) - -| Surface | Primary location | Consumed by | Count (authoring time) | -|---------|------------------|-------------|------------------------| -| **A — GUI** | Static `PRESETS` (`AddProviderModal.tsx:29-43`) + `/api/key-providers` → `KEY_LOGIN_PROVIDERS` | Add-provider modal search/quick-pick | 13 featured static (excl. `custom`) + 31 key-login deduped → **43** selectable (when proxy up) | -| **B — CLI / registry** | `OAUTH_PROVIDERS` (`oauth/index.ts:19-65`), `buildInitProviders()` (`init.ts:38-61`), `KEY_LOGIN_PROVIDERS` (`key-providers.ts:26-91`) | `ocx init`, `enrichProviderFromCatalog`, login CLI | 43 init rows (1 forward + 3 oauth + 5 hardcoded key + 31 key-login + 3 local) | -| **C — Metadata** | `PROVIDER_ALIASES` (`generate-jawcode-metadata.ts:4-17`) → `src/generated/jawcode-model-metadata.ts` | `resolveJawcodeProvider` / `getJawcodeModelMetadata` → `applyJawcodeCatalogMetadata` (`codex-catalog.ts:90-99`) | 10 alias keys → 7 jawcode bundles | - -## Scope & baseline - -- **In scope:** cross-surface audit, divergence matrix, canonical registry design, migration - plan, drift-guard test spec, casing-risk documentation. -- **Out of scope (this cycle):** implementing the registry, deleting legacy catalogs, changing - provider wire behavior, jawcode `models.json` edits, regenerating metadata beyond what the - hotfix already did. -- **Baseline at authoring time:** BUG A/B hotfixes present on working tree; `KEY_LOGIN_PROVIDERS` - has 31 entries including `opencode-go`; `PROVIDER_ALIASES` includes `minimax` / `minimax-cn`. - -## Documents - -| Doc | Contents | -|-----|----------| -| `00_overview.md` | This file — framing, hotfix symptoms, parity contract, scope | -| `01_catalog-source-audit.md` | Surface-by-surface map + **complete divergence matrix** | -| `02_single-source-design.md` | Canonical registry shape, per-consumer derivation, migration, CI guard, risks | -| `03_implementation-plan.md` | Confirmed implementation decisions and file-level plan | -| `10_registry-scaffold.md` | Registry and projection scaffold implementation record | -| `20_wiring-and-compat.md` | Consumer wiring, compatibility aliases, and GUI/runtime integration record | -| `30_verification.md` | Final verification evidence and residual risk notes | diff --git a/devlog/_fin/130_provider-catalog-single-source/01_catalog-source-audit.md b/devlog/_fin/130_provider-catalog-single-source/01_catalog-source-audit.md deleted file mode 100644 index cfbe17ec4..000000000 --- a/devlog/_fin/130_provider-catalog-single-source/01_catalog-source-audit.md +++ /dev/null @@ -1,236 +0,0 @@ -# 130.01 — Provider Catalog Source Audit - -Independent read of all three catalogs at authoring time. Every row in the divergence matrix -below is backed by a file:line citation from the repo (not from training data). - -## Surface A — GUI (`AddProviderModal.tsx`) - -### Static `PRESETS` (quick-pick, always present) - -Defined at `gui/src/components/AddProviderModal.tsx:29-43`: - -| id | label | adapter | baseUrl | defaultModel | auth | -|----|-------|---------|---------|--------------|------| -| `openai` | OpenAI (ChatGPT login) | `openai-responses` | `https://chatgpt.com/backend-api/codex` | — | forward | -| `xai` | xAI Grok | `openai-chat` | `https://api.x.ai/v1` | `grok-4.3` | oauth (`oauthProvider: xai`) | -| `anthropic` | Anthropic Claude | `anthropic` | `https://api.anthropic.com` | `claude-sonnet-4-5` | oauth | -| `kimi` | Kimi | `openai-chat` | `https://api.moonshot.ai/v1` | `kimi-k2.6` | oauth | -| `openai-apikey` | OpenAI (API key) | `openai-responses` | `https://api.openai.com/v1` | `gpt-5.5` | key | -| `opencode-go` | opencode go | `openai-chat` | `https://opencode.ai/zen/go/v1` | `kimi-k2.6` | key | -| `openrouter` | OpenRouter | `openai-chat` | `https://openrouter.ai/api/v1` | — | key | -| `groq` | Groq | `openai-chat` | `https://api.groq.com/openai/v1` | — | key | -| `google` | Google Gemini | `google` | `https://generativelanguage.googleapis.com` | `gemini-3-pro` | key | -| `azure-openai` | Azure OpenAI | **`azure-openai`** | Azure deployment URL template | — | key | -| `ollama` | Ollama (local) | `openai-chat` | `http://localhost:11434/v1` | — | key | -| `vllm` | vLLM (local) | `openai-chat` | `http://localhost:8000/v1` | — | key | -| `lm-studio` | LM Studio (local) | `openai-chat` | `http://localhost:1234/v1` | — | key | -| `custom` | Custom provider | `openai-chat` | `""` | — | key | - -### Runtime merge (`/api/key-providers`) - -`AddProviderModal.tsx:86-100` fetches `GET /api/key-providers` (served from -`listKeyLoginProviders()` → `KEY_LOGIN_PROVIDERS`) and appends any id **not** already in static -`PRESETS`, keeping `custom` last. Effective GUI list when proxy is running: **13 static (excl. `custom`) + 30 key-login rows not -already in static** (`opencode-go` deduped at `AddProviderModal.tsx:95-96`) → **43 selectable -presets** — same cardinality as `buildInitProviders()`, plus a `custom` escape hatch. - -When the proxy is **not** running, the fetch fails silently (`AddProviderModal.tsx:90`) → GUI -shows only the **13 non-custom static presets**. CLI `ocx init` still lists all 43 built-in rows. - -## Surface B — CLI / registry - -### `KEY_LOGIN_PROVIDERS` (`key-providers.ts:26-91`) - -31 API-key catalog entries (post–BUG A hotfix). Full id list: - -`deepseek`, `cerebras`, `together`, `fireworks`, `firepass`, `moonshot`, `huggingface`, -`nvidia`, `venice`, `zai`, `nanogpt`, `synthetic`, `qwen-portal`, `qianfan`, `alibaba`, -`parallel`, `zenmux`, `litellm`, `ollama-cloud`, `mistral`, `minimax`, `minimax-cn`, -`kimi-code`, `opencode-zen`, `opencode-go`, `vercel-ai-gateway`, `xiaomi`, `kilo`, -`cloudflare-ai-gateway`, `github-copilot`, `gitlab-duo`. - -`enrichProviderFromCatalog` (`key-providers.ts:99-106`) only consults this map — **no OAuth -ids**, no hardcoded init ids. - -### `OAUTH_PROVIDERS` (`oauth/index.ts:19-65`) - -| id | adapter | baseUrl | defaultModel (registry) | defaultModel (providerConfig) | -|----|---------|---------|-------------------------|-------------------------------| -| `xai` | `openai-chat` | `https://api.x.ai/v1` | `grok-4.3` | `grok-4.3` (`:36`) | -| `anthropic` | `anthropic` | `https://api.anthropic.com` | `claude-sonnet-4-6` | `claude-sonnet-4-6` (`:49`) | -| `kimi` | `openai-chat` | **`https://api.kimi.com/coding/v1`** | `kimi-k2.6` | `kimi-k2.6` (`:61`) | - -### `buildInitProviders()` assembly (`init.ts:38-61`) - -| Segment | ids | Source lines | -|---------|-----|--------------| -| Forward | `openai` | `init.ts:41` | -| OAuth | `xai`, `anthropic`, `kimi` | `init.ts:43-45` ← `OAUTH_PROVIDERS` | -| Hardcoded key | `openai-apikey`, `openrouter`, `groq`, `google`, `azure-openai` | `init.ts:48-52` | -| Key catalog | all `KEY_LOGIN_PROVIDERS` keys | `init.ts:54-55` | -| Local | `ollama`, `vllm`, `lm-studio` | `init.ts:58-60` | - -**Total: 43 init rows.** Hardcoded key block duplicates static GUI presets for the same five ids -instead of importing a shared constant. - -## Surface C — Bundled metadata - -### `PROVIDER_ALIASES` (`generate-jawcode-metadata.ts:4-17`) - -| opencodex id (alias key) | jawcode bundle id | -|--------------------------|-------------------| -| `xai` | `xai` | -| `anthropic` | `anthropic` | -| `google` | `google` | -| `gemini` | `google` | -| `moonshot` | `moonshot` | -| `kimi` | `moonshot` | -| `openrouter` | `openrouter` | -| `opencode-go` | `opencode-go` | -| `minimax` | `minimax` | -| `minimax-cn` | `minimax` | - -`allowedProviders` = unique alias targets (`generate-jawcode-metadata.ts:34`). Generated -`DATA` keys in `jawcode-model-metadata.ts:28-35`: `anthropic`, `google`, `minimax`, `moonshot`, -`opencode-go`, `openrouter`, `xai`. - -### Consumption path (`codex-catalog.ts:90-99`) - -```90:99:src/codex-catalog.ts -function applyJawcodeCatalogMetadata(entry: RawEntry, slug: string): void { - const slash = slug.indexOf("/"); - if (slash < 0) return; - const provider = slug.slice(0, slash); - const modelId = slug.slice(slash + 1); - const jawcodeProvider = resolveJawcodeProvider(provider); - if (!jawcodeProvider) return; - const meta = getJawcodeModelMetadata(jawcodeProvider, modelId); - if (!meta) return; -``` - -Lookup is **exact string match** on `modelId` (`jawcode-model-metadata.ts:42-43`: -`DATA[provider]?.find(r => r[0] === modelId)`). - -**Casing example (minimax):** `KEY_LOGIN_PROVIDERS.minimax.defaultModel` is `MiniMax-M2.5` -(`key-providers.ts:71`). Generated minimax rows use CamelCase ids (`jawcode-model-metadata.ts:31`: -`MiniMax-M2.5`, …). The `opencode-go` bundle also lists lowercase routed ids such as -`minimax-m2.5` (`jawcode-model-metadata.ts:33`). A catalog slug `minimax/minimax-m2.5` resolves -the provider alias but **misses** metadata because `modelId` casing differs. - -**jawcode naming (external):** jawcode `models.json` defines separate provider keys -`minimax`, `minimax-code`, `minimax-code-cn` (e.g. `models.json:37381`, `:37821`). opencodex -maps only `minimax` / `minimax-cn` → jawcode `minimax` (not `minimax-code*`). - ---- - -## Complete divergence matrix - -Legend: **✓** = present / aligned · **—** = intentionally absent · **✗** = mismatch · **△** = -structural asymmetry (not necessarily a bug). - -### 1 — ID presence across surfaces - -| id | GUI static PRESETS | GUI (proxy up) | `KEY_LOGIN` | `buildInit` | metadata alias | Notes | -|----|-------------------|----------------|-------------|-------------|----------------|-------| -| `openai` | ✓ `:30` | ✓ | — | ✓ forward `:41` | — | forward; no metadata needed | -| `xai` | ✓ oauth `:31` | ✓ | — | ✓ oauth `:43-45` | ✓ alias `:5` | OAuth registry | -| `anthropic` | ✓ oauth `:32` | ✓ | — | ✓ oauth | ✓ alias `:6` | OAuth registry | -| `kimi` | ✓ oauth `:33` | ✓ | — | ✓ oauth | ✓ `kimi`→`moonshot` `:10-11` | OAuth id; see field mismatches | -| `openai-apikey` | ✓ `:34` | ✓ | — | ✓ hardcoded `:48` | — | Duplicated hardcoded | -| `opencode-go` | ✓ `:35` | ✓ | ✓ `:75-77` | ✓ key-login | ✓ alias `:12` | **BUG A** was ✗ in `KEY_LOGIN` (fixed) | -| `openrouter` | ✓ `:36` | ✓ | — | ✓ hardcoded `:49` | ✓ alias `:11` | Hardcoded + alias | -| `groq` | ✓ `:37` | ✓ | — | ✓ hardcoded `:50` | — | Hardcoded only | -| `google` | ✓ `:38` | ✓ | — | ✓ hardcoded `:51` | ✓ `google` alias `:7` | id `google` not `gemini` | -| `azure-openai` | ✓ `:39` | ✓ | — | ✓ hardcoded `:52` | — | **adapter mismatch** (below) | -| `ollama` | ✓ local `:40` | ✓ | — | ✓ local `:58` | — | Local server | -| `vllm` | ✓ `:41` | ✓ | — | ✓ local `:59` | — | Local server | -| `lm-studio` | ✓ `:42` | ✓ | — | ✓ local `:60` | — | Local server | -| `custom` | ✓ `:43` | ✓ | — | — | — | GUI-only manual entry | -| `deepseek` | — | ✓ via API | ✓ `:27` | ✓ | — | No metadata alias | -| `cerebras` | — | ✓ | ✓ `:28` | ✓ | — | | -| `together` | — | ✓ | ✓ `:29` | ✓ | — | | -| `fireworks` | — | ✓ | ✓ `:30` | ✓ | — | | -| `firepass` | — | ✓ | ✓ `:31` | ✓ | — | Same baseUrl as fireworks | -| `moonshot` | — | ✓ | ✓ `:32` | ✓ | ✓ `moonshot` alias `:9` | **Different endpoint from oauth `kimi`** | -| `huggingface` | — | ✓ | ✓ `:33` | ✓ | — | | -| `nvidia` | — | ✓ | ✓ `:34` | ✓ | — | | -| `venice` | — | ✓ | ✓ `:35` | ✓ | — | | -| `zai` | — | ✓ | ✓ `:36` | ✓ | — | | -| `nanogpt` | — | ✓ | ✓ `:37` | ✓ | — | | -| `synthetic` | — | ✓ | ✓ `:38` | ✓ | — | | -| `qwen-portal` | — | ✓ | ✓ `:39` | ✓ | — | | -| `qianfan` | — | ✓ | ✓ `:40` | ✓ | — | | -| `alibaba` | — | ✓ | ✓ `:41` | ✓ | — | | -| `parallel` | — | ✓ | ✓ `:42` | ✓ | — | | -| `zenmux` | — | ✓ | ✓ `:43` | ✓ | — | | -| `litellm` | — | ✓ | ✓ `:44` | ✓ | — | | -| `ollama-cloud` | — | ✓ | ✓ `:51-65` | ✓ | — | Rich `models` / `noVisionModels` seed | -| `mistral` | — | ✓ | ✓ `:70` | ✓ | — | | -| `minimax` | — | ✓ | ✓ `:71` | ✓ | ✓ alias `:15` | **BUG B** was ✗ alias (fixed); casing risk | -| `minimax-cn` | — | ✓ | ✓ `:72` | ✓ | ✓ alias `:16` | **BUG B** was ✗ alias (fixed) | -| `kimi-code` | — | ✓ | ✓ `:73` | ✓ | — | Same host as oauth `kimi`; diff defaultModel | -| `opencode-zen` | — | ✓ | ✓ `:74` | ✓ | — | Sibling endpoint to `opencode-go` | -| `vercel-ai-gateway` | — | ✓ | ✓ `:78` | ✓ | — | | -| `xiaomi` | — | ✓ | ✓ `:80` | ✓ | — | `anthropic` adapter | -| `kilo` | — | ✓ | ✓ `:87` | ✓ | — | | -| `cloudflare-ai-gateway` | — | ✓ | ✓ `:88` | ✓ | — | `anthropic` adapter; URL template | -| `github-copilot` | — | ✓ | ✓ `:89` | ✓ | — | | -| `gitlab-duo` | — | ✓ | ✓ `:90` | ✓ | — | | -| `gemini` | — | — | — | — | ✓ alias only `:8` | **No catalog id `gemini`** — alias unused unless user names provider `gemini` | - -**Summary counts** - -| Set | Count | -|-----|-------| -| GUI static preset ids (excl. `custom`) | 13 | -| `KEY_LOGIN_PROVIDERS` ids | 31 | -| `buildInitProviders` rows | 43 | -| `PROVIDER_ALIASES` keys | 10 | -| Key-login ids **without** metadata alias | 28 | -| Init/hardcoded ids with metadata alias but not in `KEY_LOGIN` | 5 (`xai`, `anthropic`, `kimi`, `google`, `openrouter`) | - -### 2 — Field mismatches (same id, different surfaces) - -| id | Field | GUI static (`AddProviderModal.tsx`) | CLI / registry | Severity | -|----|-------|-------------------------------------|----------------|----------| -| `anthropic` | `defaultModel` | `claude-sonnet-4-5` (`:32`) | `claude-sonnet-4-6` (`oauth/index.ts:49`, `init` via oauth `:45`) | ✗ User picks older default in GUI OAuth pane | -| `kimi` | `baseUrl` | `https://api.moonshot.ai/v1` (`:33`) | `https://api.kimi.com/coding/v1` (`oauth/index.ts:58`) | ✗ **Different API hosts** for same OAuth id | -| `kimi` | vs `moonshot` KEY_LOGIN | moonshot.ai in GUI preset | `moonshot` key-login uses moonshot.ai (`key-providers.ts:32`) but oauth `kimi` uses kimi.com (`oauth/index.ts:58`) | △ Two "Kimi" entry points | -| `azure-openai` | `adapter` | `azure-openai` (`:39`) | `azure` (`init.ts:52`) | ✗ Saved config adapter string differs by surface | -| `kimi-code` | `defaultModel` | — | `kimi-k2.5` (`key-providers.ts:73`) vs oauth `kimi-k2.6` (`oauth/index.ts:61`) | △ Same coding API host, different seed default | -| `minimax` | `defaultModel` vs metadata | `MiniMax-M2.5` (`key-providers.ts:71`) | metadata row id `MiniMax-M2.5` (`jawcode-model-metadata.ts:31`) | ✓ aligned, but routed lowercase ids miss lookup | -| `opencode-go` | all fields | `:35` | `key-providers.ts:75-77` | ✓ aligned post–BUG A | - -### 3 — Structural asymmetries (by design today, still drift vectors) - -| ID | Issue | Evidence | -|----|-------|----------| -| **GUI offline** | 28 key-login providers invisible without running proxy | `AddProviderModal.tsx:86-90` silent catch; init always complete (`init.ts:54-55`) | -| **Triple duplication** | `openai-apikey`, `openrouter`, `groq`, `google`, `azure-openai` exist in GUI `PRESETS` **and** `init.ts:48-52` **outside** `KEY_LOGIN_PROVIDERS` | Two edit sites for the same five ids | -| **`opencode-go` double seed** | Listed in static GUI `PRESETS` **and** `KEY_LOGIN_PROVIDERS` | `:35` + `key-providers.ts:75-77` — deduped at runtime (`:95-96`) but two authoring locations | -| **Metadata optional** | 28/31 key-login ids have no jawcode alias — catalog enrichment silently skipped | `applyJawcodeCatalogMetadata` no-op when `resolveJawcodeProvider` returns undefined (`codex-catalog.ts:95-96`) | -| **`enrichProviderFromCatalog` scope** | Only `KEY_LOGIN_PROVIDERS`; OAuth providers never enriched | `key-providers.ts:99-106` | -| **Parity contract** | Claims GUI parity via shared registries but GUI also relies on separate static `PRESETS` | `init.ts:33-36` vs `AddProviderModal.tsx:29-43` | - -### 4 — Hotfixed drift (documented for regression guard) - -| Bug | Before hotfix | After hotfix | Evidence | -|-----|---------------|--------------|----------| -| **A — `opencode-go`** | In GUI `:35` + metadata `:12`; missing `KEY_LOGIN` → init/enrich no-op | Entry at `key-providers.ts:75-77` | Comment at `:75-76` cites GUI parity | -| **B — `minimax`** | In `KEY_LOGIN` `:71-72`; no metadata alias | Aliases `:15-16` in generator; rows in `jawcode-model-metadata.ts:31` | Comment at `generate-jawcode-metadata.ts:13-14` | - ---- - -## Audit conclusions - -1. **ID parity** between `buildInitProviders()` and GUI-with-proxy is now **43 vs 43** (excluding - GUI `custom`), but **static GUI alone** matches only **13** ids — a user adding providers from - a cold GUI sees a much smaller list than `ocx init`. -2. **Field parity** is broken for **`kimi` (baseUrl)**, **`anthropic` (defaultModel)**, and - **`azure-openai` (adapter)** even when ids align. -3. **Metadata coverage** is intentionally sparse (7 jawcode bundles) but **alias maintenance is - manual** and orthogonal to catalog membership — BUG B proved that. -4. **Exact `modelId` matching** makes metadata enrichment fragile for providers whose live `/models` - ids use different casing than jawcode `models.json` (minimax is the concrete example today). - -These findings drive the single-source design in `02_single-source-design.md`. diff --git a/devlog/_fin/130_provider-catalog-single-source/02_single-source-design.md b/devlog/_fin/130_provider-catalog-single-source/02_single-source-design.md deleted file mode 100644 index e1b73faa4..000000000 --- a/devlog/_fin/130_provider-catalog-single-source/02_single-source-design.md +++ /dev/null @@ -1,262 +0,0 @@ -# 130.02 — Single-Source Provider Registry Design - -Planning spec for replacing the three hand-maintained catalogs with **one canonical registry** -and derived views. Implements the parity contract in `init.ts:33-36` mechanically instead of -by comment. - -## Goals - -| Goal | Metric | -|------|--------| -| **Single authoring surface** | Adding a provider = one registry row (+ optional jawcode bundle pointer) | -| **Derived consumers** | GUI presets, `KEY_LOGIN_PROVIDERS`, `buildInitProviders`, metadata aliases generated or asserted — not copied | -| **Drift impossible in CI** | Test fails if any consumer set ≠ registry projection | -| **Incremental migration** | No big-bang; legacy exports remain during transition | -| **Metadata correctness** | Alias map + **normalized model-id lookup** (casing) | - -Non-goals: changing jawcode `models.json`; auto-adding metadata for all 31 key-login providers -(initial bundle set stays curated). - -## Canonical registry - -### Location - -``` -src/providers/registry.ts ← authoring surface (TypeScript for types + comments) -src/providers/derive.ts ← runtime projections for CLI/API/OAuth/metadata -``` - -Keep the registry **under `src/providers/`** (not `gui/`) so CLI, server, OAuth, and -`scripts/generate-jawcode-metadata.ts` can import it directly. The GUI is a standalone Vite -package scoped to `gui/src`, so it must consume a server projection (`GET /api/provider-presets`) -instead of importing repo-root `src/providers/*` at build time. - -Alternative considered: JSON (`providers.json`) — rejected for phase 1 because entries need -inline comments (dashboard URLs, exclusion rationale mirroring `key-providers.ts:66-69`) and -typed `authKind` unions. JSON + JSONC is a follow-up if non-TS editors need access. - -### Row shape (`ProviderRegistryEntry`) - -```ts -/** One logical provider id across all surfaces. */ -export interface ProviderRegistryEntry { - id: string; // config key, e.g. "opencode-go" - label: string; - adapter: string; // canonical adapter string (resolve azure vs azure-openai here) - baseUrl: string; - authKind: "forward" | "oauth" | "key" | "local"; - - /** GUI quick-pick: show in static preset list (not only search catalog). */ - featured?: boolean; - - /** API-key flow: dashboard link. */ - dashboardUrl?: string; - - defaultModel?: string; - models?: string[]; - noVisionModels?: string[]; - noReasoningModels?: string[]; - - /** OAuth registry key when authKind === "oauth" (usually same as id). */ - oauthId?: string; - - /** Bundled jawcode metadata: jawcode bundle id in models.json (e.g. minimax-cn → "minimax"). */ - jawcodeBundle?: string; - - /** - * Optional normalizer for metadata lookup when live /models ids differ from jawcode ids - * (minimax CamelCase vs lowercase). Applied in applyJawcodeCatalogMetadata before getJawcodeModelMetadata. - */ - metadataModelIdNormalize?: "case-insensitive"; -} -``` - -**Design rules** - -1. **`id` is unique** — the config `providers.` key everywhere. -2. **`adapter` is canonical** — pick one string per provider (`azure` vs `azure-openai` decided - once; GUI adapter dropdown may map display labels separately). -3. **`authKind` drives init menu grouping** (`init.ts:64-69` `KIND_HEADING`) and GUI badges - (`AddProviderModal.tsx:220-224`). -4. **`featured: true`** replaces duplicate static `PRESETS` rows — quick-pick is a **filter**, not - a second list. -5. **`jawcodeBundle`** replaces `PROVIDER_ALIASES` manual duplication; unset = no bundled metadata. - -### Registry segments (replaces today's assembly) - -| Segment | authKind | Example ids | Replaces | -|---------|----------|-------------|----------| -| Forward | `forward` | `openai` | `init.ts:41`, `PRESETS :30` | -| OAuth | `oauth` | `xai`, `anthropic`, `kimi` | `OAUTH_PROVIDERS` providerConfig seeds + GUI oauth presets | -| Featured key | `key`, `featured: true` | `openai-apikey`, `opencode-go`, `openrouter`, … | `init.ts:48-52`, static `PRESETS` | -| Key catalog | `key` | `deepseek`, `mistral`, … | `KEY_LOGIN_PROVIDERS` | -| Local | `local` | `ollama`, `vllm`, `lm-studio` | `init.ts:58-60`, `PRESETS :40-42` | - -OAuth **login/refresh implementations** stay in `src/oauth/*.ts` — the registry only holds -static providerConfig seeds; `OAUTH_PROVIDERS` becomes a thin map of handlers keyed by `oauthId`. - -## Per-consumer derivation - -### 1 — GUI (`AddProviderModal.tsx`) - -**Target behavior:** the server exposes a registry-derived `GET /api/provider-presets` endpoint, -and the GUI uses that runtime shape for the add-provider picker. This keeps the standalone GUI -package isolated from repo-root TypeScript while still removing the hardcoded `PRESETS` list. - -Merge logic (`allPresets` `:94-100`) becomes unnecessary because the endpoint returns the final -selectable list: `featured + key catalog − duplicates + custom`. - -**OAuth presets:** derive from `authKind === "oauth"` rows; `oauthProvider` = `oauthId ?? id`. -**Fixes live mismatches** by reading single `baseUrl` / `defaultModel` for `kimi` and `anthropic`. - -### 2 — CLI init (`buildInitProviders`) - -Replace manual assembly (`init.ts:38-61`) with: - -```ts -export function buildInitProviders(): InitProvider[] { - return REGISTRY.map(row => ({ - id: row.id, - label: formatInitLabel(row), // centralizes "— account login" / "— API key" suffixes - adapter: row.adapter, - baseUrl: row.baseUrl, - kind: row.authKind, - dashboardUrl: row.dashboardUrl, - defaultModel: row.defaultModel, - })); -} -``` - -`enrichProviderFromCatalog` reads the same registry slice as key-login rows (not a separate map). - -### 3 — `KEY_LOGIN_PROVIDERS` export - -During migration, keep export shape: - -```ts -export const KEY_LOGIN_PROVIDERS = deriveKeyLoginMap(REGISTRY); -``` - -`listKeyLoginProviders()` unchanged for `/api/key-providers`. - -### 4 — Metadata aliases (`generate-jawcode-metadata.ts`) - -Replace hand-written `PROVIDER_ALIASES` with registry-driven generation: - -```ts -const PROVIDER_ALIASES = deriveJawcodeAliases(REGISTRY); -// { [opencodexId]: jawcodeBundle } for rows where jawcodeBundle is set -``` - -`allowedProviders = unique(jawcodeBundle values)` — same as today (`:34`). - -Generation script imports `REGISTRY` from `src/providers/registry.ts` (Bun/Node compatible). - -### 5 — Catalog metadata application (`codex-catalog.ts`) - -Extend `applyJawcodeCatalogMetadata` (`:90-99`): - -1. `resolveJawcodeProvider(provider)` — still alias map (generated from registry). -2. **New:** `normalizeModelId(provider, modelId)` using registry row's - `metadataModelIdNormalize` before `getJawcodeModelMetadata`. -3. For minimax: try exact id, then case-insensitive match against bundle rows, or map known - lowercase patterns (`minimax-m2.5` → `MiniMax-M2.5`). - -This addresses the casing fragility independent of jawcode id renames. - -## Incremental migration (no big-bang) - -| Step | Work | Risk | Rollback | -|------|------|------|----------| -| **M1 — Registry scaffold** | Add `registry.ts` with all 43+ rows transcribed from current sources; **no consumer changes** | Low | Delete new files | -| **M2 — Drift guard (read-only)** | Test compares registry projections to legacy exports; **fails on current mismatches** until M3 fixes fields | None (test-only) | Skip test in CI temporarily | -| **M3 — Wire CLI + API** | `KEY_LOGIN_PROVIDERS`, `buildInitProviders`, `/api/key-providers` import derived maps; fix `kimi`/`anthropic`/`azure-openai` fields in registry | Medium | Revert imports | -| **M4 — Wire GUI** | Replace static `PRESETS` with `/api/provider-presets`; keep a minimal custom fallback if the proxy request fails | Medium UI | Revert component | -| **M5 — Metadata pipeline** | Generator reads `jawcodeBundle` from registry; add model-id normalizer in `codex-catalog.ts` | Medium catalog | Regenerate old metadata | -| **M6 — Delete duplicates** | Remove hardcoded `init.ts:48-52` block comments; strip legacy alias object from generator | Low | — | - -**Order rationale:** M2 runs early so transcribing the registry **forces** resolving known field -mismatches before wiring consumers. BUG A/B hotfixes are already in legacy sources — M1 copies -them into registry as the golden row values. - -## CI / drift-guard test - -**File:** `tests/provider-registry-parity.test.ts` (or extend existing bun test suite). - -```ts -import { REGISTRY } from "../src/providers/registry"; -import { KEY_LOGIN_PROVIDERS } from "../src/oauth/key-providers"; -import { buildInitProviders } from "../src/init"; -import { deriveFeaturedIds, deriveJawcodeAliases } from "../src/providers/derive"; -import PRESETS from "../gui/..."; // or import derived featured after M4 - -describe("provider registry parity", () => { - it("KEY_LOGIN matches registry key rows", () => { - const fromRegistry = deriveKeyLoginMap(REGISTRY); - expect(fromRegistry).toEqual(KEY_LOGIN_PROVIDERS); - }); - - it("buildInitProviders matches registry init projection", () => { - const fromRegistry = deriveInitProviders(REGISTRY); - const legacy = buildInitProviders(); - expect(fromRegistry.map(p => p.id)).toEqual(legacy.map(p => p.id)); - // field-deep equal after M3 field fixes - }); - - it("metadata aliases match registry jawcodeBundle fields", () => { - expect(deriveJawcodeAliases(REGISTRY)).toEqual(readGeneratedAliases()); - }); - - it("featured GUI ids are a registry projection", () => { - expect(deriveProviderPresets().map(p => p.id).at(-1)).toBe("custom"); - expect(new Set(deriveFeaturedIds(REGISTRY))).toEqual(new Set(EXPECTED_FEATURED_IDS)); - }); -}); -``` - -**CI gate:** `bun test tests/provider-registry-parity.test.ts` in default `bun test` — zero -drift tolerance once M3 lands. During M1–M2, test may be `describe.skip` with a tracking issue -or assert only id sets until field fixes land. - -**Optional stricter guard:** codegen `src/providers/registry.snapshot.json` from REGISTRY in CI -and fail if registry changes without snapshot update (prevents drive-by edits). - -## Risks & mitigations - -| Risk | Impact | Mitigation | -|------|--------|------------| -| **Model-id casing** (`MiniMax-M2.5` vs `minimax-m2.5`) | `getJawcodeModelMetadata` no-op (`jawcode-model-metadata.ts:42-43`); missing `context_window` / modalities in Codex catalog | Registry `metadataModelIdNormalize` + fallback lookup in `applyJawcodeCatalogMetadata` (`codex-catalog.ts:90-99`) | -| **jawcode provider naming** (`minimax-code*` vs `minimax-cn`) | Wrong bundle if alias points to mismatched jawcode key | `jawcodeBundle` explicit per row; document mapping in registry comment | -| **OAuth vs API-key endpoint splits** (`kimi` oauth vs `moonshot` key-login) | User confusion if merged incorrectly | Keep **separate registry rows** with distinct ids; single-source does not mean single endpoint | -| **GUI bundle size** | Importing full registry + 31 rows into GUI | Tree-shake derive functions; registry is small (<50 rows) | -| **Azure adapter string** (`azure` vs `azure-openai`) | Existing configs use one spelling | Pick canonical in registry; migration note + adapter alias in router if needed | -| **Third-party docs** cite `key-providers.ts` | Contributor docs stale | Update `docs-site/.../contributing.md` in M3 (out of scope for 130 planning cycle) | -| **Runtime `/api/key-providers`** consumers | External tools parsing API | Keep endpoint; implementation reads registry | - -## Success criteria (implementation phase, post-130) - -1. Adding a provider = **one registry row** + `bun test` green + regenerate metadata if - `jawcodeBundle` set. -2. `buildInitProviders().map(p => p.id)` equals GUI selectable ids (proxy off) for all non-custom - providers. -3. BUG A/B scenarios covered by parity test: id in `featured` or key catalog ⇒ present in init, - key-login export, and metadata aliases when `jawcodeBundle` set. -4. `minimax/minimax-m2.5` catalog slug receives metadata (context window) after normalizer ships. - -## Decisions resolved before implementation - -1. **Canonical Azure adapter:** `azure-openai`; `azure` remains accepted as a legacy compatibility - alias in the server adapter resolver. -2. **Kimi OAuth baseUrl:** `https://api.kimi.com/coding/v1`; `moonshot` remains a separate API-key - provider row. -3. **Featured set:** preserve the exact 13 non-custom static presets from the pre-130 GUI picker. - ---- - -## Related phases - -| Phase | Link | -|-------|------| -| 110 | Stream reliability — orthogonal | -| 120 | WS parity — catalog `supports_websockets` still separate policy | -| jawcode | `models.json` remains upstream for bundled metadata content | diff --git a/devlog/_fin/130_provider-catalog-single-source/10_registry-scaffold.md b/devlog/_fin/130_provider-catalog-single-source/10_registry-scaffold.md deleted file mode 100644 index 69a58620d..000000000 --- a/devlog/_fin/130_provider-catalog-single-source/10_registry-scaffold.md +++ /dev/null @@ -1,68 +0,0 @@ -# 130.10 — Registry Scaffold - -## Purpose - -Phase 130 replaces provider information copied across the GUI, CLI init menu, OAuth seeds, -key-login catalog, and jawcode metadata generator with a single canonical registry. - -The scaffold work introduced two modules: - -| File | Role | -|------|------| -| `src/providers/registry.ts` | Canonical provider rows and provider metadata types. | -| `src/providers/derive.ts` | Projection helpers used by existing consumers. | - -## Registry shape - -`ProviderRegistryEntry` captures the fields that were previously spread across several modules: - -| Field | Why it exists | -|-------|---------------| -| `id` | Stable provider config key and GUI/CLI identifier. | -| `label` | Human-facing provider name. | -| `adapter` | Canonical adapter string. | -| `baseUrl` | Default endpoint seed. | -| `authKind` | One of `forward`, `oauth`, `key`, or `local`. | -| `featured` | Marks the current 13 GUI quick-pick providers. | -| `dashboardUrl` | API-key login destination. | -| `defaultModel`, `models`, `noVisionModels`, `noReasoningModels` | Provider seed and routing metadata. | -| `oauthId` | OAuth handler key when it differs from `id`. | -| `jawcodeBundle`, `extraMetadataAliases` | Bundled metadata alias source. | -| `metadataModelIdNormalize` | Per-provider metadata lookup normalization policy. | - -## Canonical decisions encoded - -| Topic | Registry value | -|-------|----------------| -| Azure | `azure-openai` is the canonical adapter string. | -| Legacy Azure | `azure` is not authored in the registry; it is handled as a server compatibility alias. | -| Kimi OAuth | `https://api.kimi.com/coding/v1`. | -| Moonshot API key | Separate `moonshot` row using `https://api.moonshot.ai/v1`. | -| Anthropic default | `claude-sonnet-4-6`. | -| Local providers | `authKind: "local"` for `ollama`, `vllm`, and `lm-studio`. | -| MiniMax metadata | `minimax` and `minimax-cn` share `jawcodeBundle: "minimax"` and use case-insensitive metadata fallback. | -| Google alias | `google` owns the row; `gemini` remains a metadata alias only. | - -## Projection helpers - -`src/providers/derive.ts` exposes small, typed projections instead of one large omniscient API: - -| Helper | Consumer | -|--------|----------| -| `deriveKeyLoginMap()` | `KEY_LOGIN_PROVIDERS` compatibility export and `/api/key-providers`. | -| `deriveInitProviders()` | `ocx init` provider menu. | -| `deriveOAuthProviderConfig()` | OAuth provider config seeds. | -| `deriveOAuthDefaultModel()` | OAuth default model fields. | -| `deriveProviderPresets()` | GUI add-provider picker via `/api/provider-presets`. | -| `deriveFeaturedProviderIds()` | Parity test guard for the current featured set. | -| `deriveJawcodeAliases()` | Metadata generator alias map. | -| `shouldCaseFoldMetadataModelId()` | Catalog metadata fallback for MiniMax casing. | - -## Boundary notes - -The GUI does not import `src/providers/*` directly. The plan audit found that the GUI is a -standalone Vite package scoped to `gui/src`, so the server owns the projection and the GUI reads -it at runtime through `/api/provider-presets`. - -The registry is TypeScript rather than JSON because the current provider rows benefit from typed -unions and comments. A later codegen step can export JSON if external tooling needs it. diff --git a/devlog/_fin/130_provider-catalog-single-source/20_wiring-and-compat.md b/devlog/_fin/130_provider-catalog-single-source/20_wiring-and-compat.md deleted file mode 100644 index afd470f19..000000000 --- a/devlog/_fin/130_provider-catalog-single-source/20_wiring-and-compat.md +++ /dev/null @@ -1,96 +0,0 @@ -# 130.20 — Wiring and Compatibility - -## Consumer rewiring - -The implementation keeps existing public surfaces while replacing their authoring source. - -| Consumer | Before | After | -|----------|--------|-------| -| `src/oauth/key-providers.ts` | Hand-maintained `KEY_LOGIN_PROVIDERS` map. | `KEY_LOGIN_PROVIDERS = deriveKeyLoginMap()`. | -| `src/init.ts` | Manual assembly from OAuth rows, hardcoded key rows, key-login rows, and local rows. | `buildInitProviders()` returns `deriveInitProviders()`. | -| `src/oauth/index.ts` | OAuth provider configs copied inline. | Login/refresh handlers stay local; `providerConfig` and `defaultModel` derive from registry. | -| `src/server.ts` | `/api/key-providers` only exposed key-login rows. | Existing endpoint remains; new `/api/provider-presets` returns GUI-ready registry projection. | -| `gui/src/components/AddProviderModal.tsx` | Static `PRESETS` plus `/api/key-providers` merge. | Fetches `/api/provider-presets`; keeps a minimal `custom` fallback. | -| `scripts/generate-jawcode-metadata.ts` | Hand-written `PROVIDER_ALIASES`. | Uses `deriveJawcodeAliases()`. | -| `src/codex-catalog.ts` | Exact metadata lookup only. | Exact lookup first; registry-gated case-insensitive fallback for MiniMax-style providers. | -| `README.md` | Azure adapter documented as `azure`. | Azure adapter documented as canonical `azure-openai`. | - -## Compatibility contracts preserved - -| Contract | Result | -|----------|--------| -| Saved provider config JSON shape | Unchanged. | -| `/api/key-providers` response shape | Still `{ providers: [...] }`. | -| GUI provider creation request shape | Unchanged; only preset source changed. | -| `KEY_LOGIN_PROVIDERS` export | Still a record keyed by provider id. | -| `OAUTH_PROVIDERS` export | Still keyed by OAuth provider id with login/refresh handlers. | -| `resolveJawcodeProvider()` | Still generated and exported from `src/generated/jawcode-model-metadata.ts`. | -| `getJawcodeModelMetadata()` | Still generated and exported with exact lookup behavior. | - -## `/api/key-providers` set expansion - -The `/api/key-providers` response shape is preserved, but its id set intentionally expands from -the old dedicated key-login map to every registry row with `authKind: "key"`. That brings the -featured key providers that were previously GUI-static/init-hardcoded only into the key-provider -catalog too: - -| Newly included key ids | Why | -|------------------------|-----| -| `openai-apikey`, `openrouter`, `groq`, `google`, `azure-openai` | These are real API-key providers and now share the same registry projection as the rest of the key catalog. | - -The parity test freezes the full key-provider id set so this public endpoint cannot expand or -shrink silently after Phase 130. - -## Legacy Azure alias - -The canonical registry value is `adapter: "azure-openai"`. Existing saved configs may still contain -`adapter: "azure"`, so `resolveAdapter()` now accepts both: - -| Adapter string | Behavior | -|----------------|----------| -| `azure-openai` | Canonical Azure adapter path. | -| `azure` | Legacy compatibility alias routed to the same Azure adapter. | - -This is a compatibility fix: before Phase 130, `azure-openai` worked in the GUI path while `azure` -could be emitted by `ocx init`, creating an inconsistent saved config depending on setup path. - -## GUI runtime endpoint - -`GET /api/provider-presets` returns the final picker list: - -```json -{ - "providers": [ - { "id": "openai", "label": "OpenAI (ChatGPT login)", "auth": "forward" }, - { "id": "custom", "label": "Custom provider", "auth": "key" } - ] -} -``` - -The actual response includes adapter, base URL, default model, OAuth provider key, dashboard URL, -and notes when present. `custom` stays last. - -The GUI uses this endpoint because importing repo-root `src/providers/*` into the standalone Vite -package would cross its configured TypeScript boundary. - -If `/api/provider-presets` is unavailable, the modal falls back only to `custom`. That is a -deliberate tradeoff: keeping the old 13 static presets in the GUI would preserve a second authored -catalog and reintroduce the drift Phase 130 removes. - -## Metadata normalization - -MiniMax illustrates a concrete metadata drift bug: - -| Source | Example model id | -|--------|------------------| -| jawcode bundled metadata | `MiniMax-M2.5` | -| routed catalog slug | `minimax/minimax-m2.5` | - -Phase 130 keeps exact lookup as the default and adds a registry-gated fallback: - -1. Resolve provider alias through generated metadata aliases. -2. Try `getJawcodeModelMetadata()` exact lookup. -3. If the provider registry row has `metadataModelIdNormalize: "case-insensitive"`, try - `getJawcodeModelMetadataCaseInsensitive()`. - -This keeps normalization narrow and avoids guessing metadata for unrelated providers. diff --git a/devlog/_fin/130_provider-catalog-single-source/30_verification.md b/devlog/_fin/130_provider-catalog-single-source/30_verification.md deleted file mode 100644 index 1c38b4494..000000000 --- a/devlog/_fin/130_provider-catalog-single-source/30_verification.md +++ /dev/null @@ -1,78 +0,0 @@ -# 130.30 — Verification - -## New drift guard - -`tests/provider-registry-parity.test.ts` covers the Phase 130 invariants: - -| Test area | Assertion | -|-----------|-----------| -| Registry uniqueness | Provider ids are unique. | -| Key-login projection | `KEY_LOGIN_PROVIDERS` equals `deriveKeyLoginMap()` and matches a frozen 36-id endpoint set. | -| CLI init projection | `buildInitProviders()` equals `deriveInitProviders()`. | -| OAuth canonical fields | Kimi URL, Anthropic default, and xAI default derive from registry values. | -| GUI featured set | The current 13 non-custom featured providers are preserved and `custom` remains last. | -| Metadata aliases | Registry-derived aliases match generated alias behavior for `gemini` and `minimax-cn`. | -| Legacy Azure | `adapter: "azure"` still resolves to the Azure adapter. | -| MiniMax casing | `minimax/minimax-m2.5` receives context metadata through the catalog path. | - -## Verification commands - -### Targeted registry guard - -```bash -bun test tests/provider-registry-parity.test.ts -``` - -Result: pass, 8 tests. - -### Full test suite - -```bash -bun test tests -``` - -Result: pass, 60 tests across 13 files. - -### TypeScript - -```bash -bun x tsc --noEmit -``` - -Result: pass. - -### GUI build - -```bash -bun run build:gui -``` - -Result: pass. Vite built `gui/dist` successfully. - -## Line-limit check - -All new Phase 130 implementation files are under the 500-line project limit: - -| File | Lines | -|------|-------| -| `src/providers/registry.ts` | 140 | -| `src/providers/derive.ts` | 163 | -| `tests/provider-registry-parity.test.ts` | 102 | -| `devlog/130_provider-catalog-single-source/10_registry-scaffold.md` | 69 | -| `devlog/130_provider-catalog-single-source/20_wiring-and-compat.md` | 83 | -| `devlog/130_provider-catalog-single-source/30_verification.md` | 60 | - -## Residual risks - -| Risk | Status | -|------|--------| -| GUI endpoint unavailable | The modal keeps a minimal `custom` fallback. This is more degraded than the pre-130 static-preset fallback, but it avoids preserving a second authored provider catalog in the standalone GUI package. | -| Unknown provider metadata | Still intentionally sparse; registry `jawcodeBundle` opt-in controls bundled metadata coverage. | -| Existing third-party docs mentioning `azure` | Runtime remains compatible through alias; README now documents `azure-openai`. | -| Generated metadata file long lines | Existing generated style preserved; file remains under the line-count limit. | -| `/api/key-providers` id set | Shape is unchanged, but the set intentionally expands from 31 dedicated key-login rows to 36 `authKind: "key"` registry rows so featured API-key providers are no longer maintained outside the key projection. The parity test now freezes this set. | - -## Done assessment - -Phase 130 now has one authored provider registry, derived consumers, drift/parity tests, preserved -legacy compatibility, and passing full verification gates. diff --git a/devlog/_fin/131_opencode-go-metadata-drift/00_plan.md b/devlog/_fin/131_opencode-go-metadata-drift/00_plan.md deleted file mode 100644 index fe90cceca..000000000 --- a/devlog/_fin/131_opencode-go-metadata-drift/00_plan.md +++ /dev/null @@ -1,134 +0,0 @@ -# 131.00 — Plan: OpenCode Go Metadata Drift Closure - -## Goal - -Close OpenCode Go model metadata drift across the three places that now matter: - -1. GJC upstream clone on `dev` at `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc`. -2. jawcode at `/Users/jun/Developer/new/700_projects/jawcode`. -3. opencodex generated jawcode metadata at `/Users/jun/Developer/new/700_projects/opencodex`. - -The user-facing bug is that Codex receives wrong context/output limits for routed OpenCode Go -models. The root cause is that OpenCode Go's `/v1/models` endpoint only exposes -`id/object/created/owned_by`, while the exact `context/output/modalities` values live in -OpenCode's official `/data/...` catalog pages. - -## Retry Update: Web SOT Split - -The first upstream GJC PR (#914) was closed after review because it treated every tracked -OpenCode Go id as `/v1/chat/completions` and reused generic data-page prices for rows where -the Go product page publishes a different contract. The retry uses a split source of truth: - -- `https://opencode.ai/docs/go/#endpoints` is authoritative for the OpenCode Go gateway - endpoint/API SDK path. -- `https://opencode.ai/docs/go/#usage-limits` is authoritative for current Go product - prices when the row appears in that table. -- `https://opencode.ai/data/...` pages remain authoritative for context/output/modalities. -- `https://opencode.ai/zen/go/v1/models` is existence-only for this work because it returns - model ids without context/output/pricing metadata. - -This means MiniMax M2.5/M2.7/M3 and Qwen3.6/3.7 Plus/Max must route to -`anthropic-messages` on `https://opencode.ai/zen/go`, while GLM/Kimi/DeepSeek/MiMo rows in -the endpoint table route to `openai-completions` on `https://opencode.ai/zen/go/v1`. -Qwen Plus rows have tiered prices in the Go usage table; generated rows advertise a 1M -context window, so the retry encodes the `> 256K tokens` tier. - -## Official Source Values - -These are the Phase 131 source-of-truth values verified from official OpenCode data pages on -2026-06-20. The jawcode/GJC model type can only represent `text` and `image`, so `video`, -`audio`, and `pdf` are recorded here as source facts but cannot be emitted until the upstream -`Model.input` contract expands. - -| Model | Context | Output | Official input | Represented input | -|---|---:|---:|---|---| -| `deepseek-v4-flash` | 1000000 | 384000 | text | text | -| `deepseek-v4-pro` | 1000000 | 384000 | text | text | -| `glm-5` | 204800 | 131072 | text | text | -| `glm-5.1` | 200000 | 131072 | text | text | -| `glm-5.2` | 1000000 | 131072 | text | text | -| `kimi-k2.5` | 262144 | 262144 | text, image, video | text, image | -| `kimi-k2.6` | 262144 | 262144 | text, image, video | text, image | -| `kimi-k2.7-code` | 262144 | 262144 | text, image, video | text, image | -| `minimax-m2.5` | 204800 | 131072 | text | text | -| `minimax-m2.7` | 204800 | 131072 | text | text | -| `minimax-m3` | 512000 | 128000 | text, image, video | text, image | -| `qwen3.5-plus` | 1000000 | 65536 | text, image, video | text, image | -| `qwen3.6-plus` | 1000000 | 65536 | text, image, video | text, image | -| `qwen3.7-max` | 1000000 | 65536 | text | text | -| `qwen3.7-plus` | 1000000 | 64000 | text, image | text, image | -| `mimo-v2-omni` | 262144 | 131072 | text, image, audio, video, pdf | text, image | -| `mimo-v2-pro` | 1048576 | 131072 | text | text | -| `mimo-v2.5` | 1048576 | 131072 | text, image, audio, video | text, image | -| `mimo-v2.5-pro` | 1048576 | 131072 | text | text | -| `hy3-preview` | 256000 | 64000 | text | text | - -Official evidence pages: - -- `https://opencode.ai/docs/go/` -- `https://opencode.ai/data/deepseek/deepseek-v4-flash` -- `https://opencode.ai/data/zhipu/glm-5-2` -- `https://opencode.ai/data/moonshot/kimi-k2-7-code` -- `https://opencode.ai/data/minimax/minimax-m3` -- `https://opencode.ai/data/qwen/qwen3-5-plus` -- `https://opencode.ai/data/xiaomi/mimo-v2-5` -- `https://opencode.ai/data/tencent/hy3-preview` -- `https://opencode.ai/zen/go/v1/models` - -## PABCD Cycle Map - -### Cycle 1 — Documentation - -Create this `131` devlog folder and record the root cause, source values, and exact patch path. - -### Cycle 2 — GJC `dev` - -Modify: - -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc/packages/ai/src/provider-models/openai-compat.ts` -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc/packages/ai/test/issue-887-repro.test.ts` -- generated: `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc/packages/ai/src/models.json` - -Plan: - -- Add an OpenCode Go official metadata override table in `openai-compat.ts`. -- Make OpenCode Go dynamic discovery map `/v1/models` ids through that table. -- Add descriptor-level appended official rows so models absent from `models.dev` still enter generated `models.json`. -- Expand the issue 887 test from routing-only to routing + metadata + missing-row coverage. -- Run `bun --cwd=packages/ai run generate-models`, targeted test, and package check. - -### Cycle 3 — jawcode - -Apply the same generator-safe patch to: - -- `/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/provider-models/openai-compat.ts` -- `/Users/jun/Developer/new/700_projects/jawcode/packages/ai/test/issue-887-repro.test.ts` -- generated: `/Users/jun/Developer/new/700_projects/jawcode/packages/ai/src/models.json` - -Run the same targeted and package gates. - -### Cycle 4 — opencodex - -Modify: - -- `/Users/jun/Developer/new/700_projects/opencodex/src/generated/jawcode-model-metadata.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts` -- this devlog folder's verification record - -Plan: - -- Regenerate opencodex jawcode metadata from patched jawcode using - `bun run generate:jawcode-metadata`. -- Add/adjust catalog tests for high-risk OpenCode Go entries: - `glm-5.2`, `qwen3.5-plus`, `kimi-k2.7-code`, `minimax-m3`, `hy3-preview`. -- Run opencodex targeted tests, full test suite, typecheck, and a local catalog smoke. - -## Acceptance Criteria - -- GJC `opencode-go` generated rows match official context/output for the 20 tracked models. -- jawcode `opencode-go` generated rows match official context/output for the 20 tracked models. -- opencodex generated metadata exposes the corrected rows. -- Codex catalog entries built by opencodex carry the corrected `context_window`, - `max_context_window`, and `auto_compact_token_limit`. -- No token values are printed. -- No unrelated dirty worktree changes are reverted. diff --git a/devlog/_fin/131_opencode-go-metadata-drift/10_verification.md b/devlog/_fin/131_opencode-go-metadata-drift/10_verification.md deleted file mode 100644 index 468306f4c..000000000 --- a/devlog/_fin/131_opencode-go-metadata-drift/10_verification.md +++ /dev/null @@ -1,152 +0,0 @@ -# 131.10 — Verification: OpenCode Go Metadata Drift Retry - -## Scope - -Phase 131 retry replaces the closed GJC PR #914 assumptions with the current official -OpenCode Go web contract. - -Source-of-truth split: - -- `https://opencode.ai/docs/go/#endpoints`: OpenCode Go endpoint/API SDK routing. -- `https://opencode.ai/docs/go/#usage-limits`: current Go product prices for rows present - in the usage table. -- `https://opencode.ai/data/...`: context window, output limit, and modality facts. -- `https://opencode.ai/zen/go/v1/models`: existence-only; the endpoint does not expose - context/output/pricing metadata. - -## Official Routing Contract - -Routes encoded in GJC and jawcode: - -- `openai-completions` on `https://opencode.ai/zen/go/v1`: `deepseek-v4-flash`, - `deepseek-v4-pro`, `glm-5.1`, `glm-5.2`, `kimi-k2.6`, `kimi-k2.7-code`, `mimo-v2.5`, - `mimo-v2.5-pro`. -- `anthropic-messages` on `https://opencode.ai/zen/go`: `minimax-m2.5`, `minimax-m2.7`, - `minimax-m3`, `qwen3.6-plus`, `qwen3.7-max`, `qwen3.7-plus`. - -Rows present in the data catalog but absent from the current Go endpoint table retain data-page -metadata and are not used to infer undocumented endpoint overrides. - -## GJC - -Repository: - -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc` - -Branch: - -- `codex/opencode-go-contract`, based on `origin/dev`. - -Modified: - -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc/packages/ai/src/provider-models/openai-compat.ts` -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc/packages/ai/test/issue-887-repro.test.ts` -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_upstream_gjc/packages/ai/src/models.json` - -Verification: - -- `bun test packages/ai/test/issue-887-repro.test.ts` passed: 17 tests, 0 failures, - 32 assertions. -- `bun --cwd=packages/ai run generate-models` produced `opencode-go: 20 models`. -- Generated contract check passed: `checked=8 bad=0` for routing, base URL, context, - output, and price samples. -- `bun --cwd=packages/ai run check` passed. - -## jawcode - -Repository: - -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_worktrees/opencode-go-contract` - -Branch: - -- `codex/opencode-go-contract`, based on `origin/dev` in a separate worktree to preserve - unrelated dirty files in `/Users/jun/Developer/new/700_projects/jawcode`. - -Modified: - -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_worktrees/opencode-go-contract/packages/ai/src/provider-models/openai-compat.ts` -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_worktrees/opencode-go-contract/packages/ai/test/issue-887-repro.test.ts` -- `/Users/jun/Developer/new/700_projects/jawcode/devlog/_worktrees/opencode-go-contract/packages/ai/src/models.json` - -Verification: - -- `bun test packages/ai/test/issue-887-repro.test.ts` passed: 17 tests, 0 failures, - 32 assertions. -- `bun --cwd=packages/ai run generate-models` produced `opencode-go: 20 models`. -- Generated contract check passed: `checked=8 bad=0` for routing, base URL, context, - output, and price samples. -- `bun --cwd=packages/ai run check` passed after temporarily linking the existing parent - repo `node_modules` into the worktree for type resolution; the symlink was removed and not - staged. - -Existing dirty files preserved outside the worktree: - -- `/Users/jun/Developer/new/700_projects/jawcode/AGENTS.md` -- `/Users/jun/Developer/new/700_projects/jawcode/.agents/` -- `/Users/jun/Developer/new/700_projects/jawcode/.claude/` - -## opencodex - -Repository: - -- `/Users/jun/Developer/new/700_projects/opencodex` - -Branch: - -- `codex/opencode-go-contract` - -Implementation result: - -- `JAWCODE_MODELS_JSON=/Users/jun/Developer/new/700_projects/jawcode/devlog/_worktrees/opencode-go-contract/packages/ai/src/models.json bun run generate:jawcode-metadata` - was executed and verified. It introduced no `opencode-go` metadata delta because opencodex - stores context/output/modalities only; endpoint and price changes live in GJC/jawcode. -- The generated-file diff was intentionally reduced back to zero to avoid unrelated dynamic - `openrouter` metadata churn from the jawcode worktree. -- The no-code-runtime rationale is recorded in `20_opencodex-integration.md`. - -Verification: - -- `bun test tests/codex-catalog.test.ts` passed: 13 tests, 0 failures, 98 assertions. -- `bun test tests/provider-registry-parity.test.ts` passed: 8 tests, 0 failures, - 23 assertions. -- OpenCode Go generated metadata sample check passed: `metadata_checked=5 bad=0`. -- `bun test tests` passed: 88 tests, 0 failures, 287 assertions. -- `bun x tsc --noEmit` passed. -- local `ocx` catalog smoke passed and the proxy was stopped afterward. - -Runtime `ocx` smoke: - -- `ocx start` started the proxy on `http://localhost:10100`. -- `GET http://127.0.0.1:10100/healthz` returned `health ok`. -- `GET http://127.0.0.1:10100/v1/models?client_version=0.141.0` returned Codex catalog - rows with corrected OpenCode Go limits: - - `opencode-go/glm-5.2`: `context_window=1000000`, - `max_context_window=1000000`, `auto_compact_token_limit=900000`, - `input_modalities=["text"]`. - - `opencode-go/kimi-k2.7-code`: `context_window=262144`, - `max_context_window=262144`, `auto_compact_token_limit=235929`, - `input_modalities=["text","image"]`. - - `opencode-go/minimax-m3`: `context_window=512000`, - `max_context_window=512000`, `auto_compact_token_limit=460800`, - `input_modalities=["text","image"]`. - - `opencode-go/qwen3.7-plus`: `context_window=1000000`, - `max_context_window=1000000`, `auto_compact_token_limit=900000`, - `input_modalities=["text","image"]`. -- Final state confirmed with `ocx stop`: no running proxy found and opencodex was removed - from Codex config. - -## PR / CI - -Opened PRs: - -- GJC upstream: `https://github.com/Yeachan-Heo/gajae-code/pull/915` -- jawcode: `https://github.com/lidge-jun/jawcode/pull/1` -- opencodex: `https://github.com/lidge-jun/opencodex/pull/1` - -Initial CI status after PR creation: - -- GJC PR #915: checks pending (`Affected path validation / plan`, `gjc-state-gates / integrity`, - `gjc-state-gates / read`, `gjc-state-gates / runtime`, `gjc-state-gates / static`). -- jawcode PR #1: checks pending (`Affected path validation`, `jwc-state-gates`). -- opencodex PR #1: no checks reported on the branch at creation time. diff --git a/devlog/_fin/131_opencode-go-metadata-drift/20_opencodex-integration.md b/devlog/_fin/131_opencode-go-metadata-drift/20_opencodex-integration.md deleted file mode 100644 index 79c785bed..000000000 --- a/devlog/_fin/131_opencode-go-metadata-drift/20_opencodex-integration.md +++ /dev/null @@ -1,48 +0,0 @@ -# 131.20 — opencodex Integration Note - -## Why There Is No opencodex Runtime Code Patch - -Phase 131 retry changes two upstream jawcode/GJC surfaces: - -- OpenCode Go endpoint routing: `/v1/chat/completions` versus `/v1/messages`. -- OpenCode Go product pricing from `https://opencode.ai/docs/go/#usage-limits`. - -opencodex does not consume either field. Its Codex catalog integration consumes only the -generated jawcode metadata fields represented in -`src/generated/jawcode-model-metadata.ts`: - -- `contextWindow` -- `maxTokens` -- `input` -- `reasoning` -- optional `wireModelId` - -Those are the fields Codex needs for `context_window`, `max_context_window`, -`auto_compact_token_limit`, and `input_modalities`. - -## Existing Guards - -The opencodex-side regression surface is already covered by -`tests/codex-catalog.test.ts`: - -- `opencode-go high-risk models use official jawcode metadata in the Codex catalog` - locks `glm-5.2`, `qwen3.5-plus`, `kimi-k2.7-code`, `minimax-m3`, and `hy3-preview`. -- `opencode-go catalog sync appends official rows missing from /v1/models` verifies that - generated jawcode rows are appended for configured `opencode-go`, even when the live - provider `/v1/models` endpoint omits them. - -The runtime smoke in `10_verification.md` then verifies the same path through real `ocx start` -and `GET /v1/models?client_version=0.141.0`. - -## Generated Metadata Result - -Regenerating opencodex metadata from the patched jawcode worktree was tested with: - -`JAWCODE_MODELS_JSON=/Users/jun/Developer/new/700_projects/jawcode/devlog/_worktrees/opencode-go-contract/packages/ai/src/models.json bun run generate:jawcode-metadata` - -The relevant `opencode-go` context/output/modalities rows did not change compared with the -existing committed snapshot. The retry's meaningful payload therefore lives in jawcode/GJC, -while opencodex records the source-of-truth split and verifies the live catalog behavior. - -To avoid unrelated dynamic provider churn, no generated metadata diff is committed in -opencodex for this retry. diff --git a/devlog/_fin/132_websocket-correctness-release/00_plan.md b/devlog/_fin/132_websocket-correctness-release/00_plan.md deleted file mode 100644 index 16ef16c3d..000000000 --- a/devlog/_fin/132_websocket-correctness-release/00_plan.md +++ /dev/null @@ -1,185 +0,0 @@ -# 132.00 — Plan: WebSocket Correctness and 1.9.0 Release - -## Goal - -Fix the Phase 120 WebSocket transport findings against current Codex RS, then ship opencodex as -version 1.9.0. - -This phase is C4 because it changes a long-lived transport, Codex provider capability -advertisement, push/main merge state, and npm release behavior. - -## Sources Checked - -- OpenAI WebSocket mode docs: `generate=false` warmup returns a response ID that may be chained - with `previous_response_id`. -- Current Codex RS local checkout: - `/Users/jun/Developer/codex/openai-codex/codex-rs/core/src/client.rs` - - `responses_websocket_enabled()` gates on provider `supports_websockets`. - - `prepare_websocket_request()` sends incremental `input` with `previous_response_id` only when - the prior response id is nonempty; empty id forces a full request. -- Current Codex RS local checkout: - `/Users/jun/Developer/codex/openai-codex/codex-rs/codex-api/src/endpoint/responses_websocket.rs` - - EOF or close before `response.completed` becomes a stream error. - - Standalone `{ "type": "error", "status": ..., "headers": ... }` maps to HTTP-style transport - errors. -- Bun WebSocket docs: `ServerWebSocket.send()` returns `-1`, `0`, or positive byte count; `0` - means the message was dropped. - -## Findings to Close - -1. Routed follow-up requests cannot resolve `previous_response_id`. -2. Any 2xx body is treated as SSE, causing JSON/HTML/empty success stalls. -3. WS pumping does not enforce exactly one terminal event. -4. Native response headers are not preserved or represented on the WS path. -5. Interrupt parity is only socket-close parity. -6. HTTP error status and retry headers are wrapped as in-band `response.failed`. -7. SSE framing is too narrow and ignores WebSocket send drops/backpressure. -8. WebSockets are advertised by default before correctness is complete. -9. The socket stores all inbound headers instead of an allowlisted subset. - -## PABCD Cycle Map - -### Cycle 1 — Research and Safety Defaults - -Modify: - -- `/Users/jun/Developer/new/700_projects/opencodex/.gitignore` -- `/Users/jun/Developer/new/700_projects/opencodex/src/config.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/types.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/codex-inject.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/codex-inject.test.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts` - -Plan: - -- Change `websocketsEnabled(config)` from absent-is-on to explicit opt-in: - `config.websockets === true`. -- Change fresh default config from `websockets: true` to `websockets: false`. -- Update comments/tests to say WebSocket support is opt-in until 132 protocol gates pass. -- Keep catalog and provider table synchronized when the flag is explicitly true. -- Add `.tmp/` to `.gitignore` so repo-local scratch output does not block the clean-tree release - helper. - -Acceptance: - -- Default injected provider table has no `supports_websockets`. -- Explicit `{websockets:true}` advertises provider/catalog WebSocket support. -- Explicit `{websockets:false}` suppresses both. - -### Cycle 2 — WS Protocol Core - -Modify: - -- `/Users/jun/Developer/new/700_projects/opencodex/src/ws-bridge.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/server.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/types.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/ws-endpoint.test.ts` - -Plan: - -- Store only allowlisted inbound WS headers in `WsData`. - - Inbound allowlist: reuse `FORWARD_HEADERS`; do not retain cookies or unrelated upgrade headers. - - Outbound safe header allowlist: `retry-after`, `x-request-id`, `openai-request-id`, - `x-codex-turn-state`, `openai-model`, `x-models-etag`, `x-reasoning-included`, and - `x-ratelimit-*`. -- Add routed continuation safety: - - preserve parsed `previous_response_id` in `OcxParsedRequest`; - - for routed/non-passthrough WS responses, emit empty response ids so Codex sends a full next - request rather than an unresolved incremental suffix; - - native passthrough keeps upstream ids. -- Replace `pumpSseToWebSocket()` with a protocol pump that: - - decodes CRLF/LF SSE, multiline `data:`, arbitrary chunks, and unterminated final events; - - validates JSON payloads; - - detects terminal `response.completed`, `response.failed`, `response.incomplete`; - - sends exactly one terminal and cancels upstream after terminal; - - converts EOF/read failure before terminal into standalone `type:error`; - - checks `ws.send()` results; treat `0` as failed/dropped, and treat `-1` as accepted with - backpressure so the socket remains usable unless a later send drops. -- Change WS request execution to track one in-flight turn per socket: - - a new `response.create` cancels the previous in-flight reader before starting the new turn; - - a per-turn generation id prevents stale frames from the cancelled turn reaching the socket. -- Classify response bodies before pumping: - - actual SSE via content-type or bounded sniff; - - Responses JSON converted into a valid event sequence; - - empty/unexpected success converted to protocol error. -- Emit standalone WS error envelopes for transport failures with safe headers and status. -- Add safe response header allowlist for WebSocket error/metadata paths. - -Acceptance: - -- Unit tests fail before/fix after for continuation empty id, JSON 200 conversion, HTML/empty - protocol errors, duplicate terminal isolation, EOF-before-terminal error, CRLF/multiline/final - unterminated SSE, dropped terminal send, `-1` backpressure tolerance, inbound header minimization, - outbound safe-header filtering, and same-socket new-turn cancellation. - -### Cycle 3 — Codex-Facing Verification - -Modify: - -- `/Users/jun/Developer/new/700_projects/opencodex/tests/ws-endpoint.test.ts` - -New: - -- `/Users/jun/Developer/new/700_projects/opencodex/devlog/132_websocket-correctness-release/10_verification.md` - -Plan: - -- Add targeted tests for: - - routed two-turn socket where second frame has `previous_response_id` and suffix only; - - routed tool-result follow-up preserving context through empty response ids; - - native passthrough preserving upstream ids and safe headers; - - non-2xx error envelope with `status`, `error`, and safe headers; - - same-socket new-turn cancellation proving stale frames are not delivered after a logical - interrupt-like replacement turn. -- Run: - - `bun test tests/ws-endpoint.test.ts` - - `bun test tests/codex-inject.test.ts` - - `bun test tests/codex-catalog.test.ts` - - `bun test tests` - - `bun x tsc --noEmit` -- Run live `ocx` smoke with `websockets:false` default to confirm Codex no longer selects WS by - default; explicit `websockets:true` can be tested with direct WS script. - -Acceptance: - -- All local gates pass. -- `ocx` ends stopped. -- Verification doc records exact commands and residual limitation: socket-close cancellation and - same-socket new-turn cancellation are proven locally; a human-visible TUI Ctrl-C interrupt may - still require a Codex-driven manual/live transcript if automation cannot inject it reliably. - -### Cycle 4 — Push, Main Merge, and 1.9.0 Release - -Modify: - -- `/Users/jun/Developer/new/700_projects/opencodex/package.json` - -Plan: - -- Commit Phase 132 implementation and docs on `dev`. -- Push `dev` to `origin`. -- Merge `dev` into `main` after local gates pass and `git status --porcelain` is clean. -- Run release flow for `1.9.0 --publish` from clean `main`. -- Watch GitHub Release workflow to completion. -- Verify npm registry: - - `npm view @bitkyc08/opencodex@1.9.0 version` - - `npm dist-tag ls @bitkyc08/opencodex` - -Acceptance: - -- `origin/main` contains the 1.9.0 release commit. -- GitHub Release workflow passes. -- npm shows `@bitkyc08/opencodex@1.9.0`. -- Local `ocx status` is not running. - -## Non-Goals - -- Full state reconstruction for routed providers is not required in this phase if the empty-id - conservative behavior is implemented and tested. -- Native upstream WebSocket-to-WebSocket bridging is not required; native passthrough may continue - HTTP Responses upstream as long as Codex-facing WS protocol behavior is correct. -- Non-representable provider modalities outside the current opencodex/jawcode type model are out of - scope. diff --git a/devlog/_fin/132_websocket-correctness-release/10_verification.md b/devlog/_fin/132_websocket-correctness-release/10_verification.md deleted file mode 100644 index 0305e4ef0..000000000 --- a/devlog/_fin/132_websocket-correctness-release/10_verification.md +++ /dev/null @@ -1,142 +0,0 @@ -# 132.10 — Verification: WebSocket Correctness - -## Scope - -Phase 132 hardens the Codex-facing Responses WebSocket path and changes WebSocket advertisement -from default-on to explicit opt-in. - -## Implementation Evidence - -Commits: - -- `334f7c2 fix: gate websocket transport by explicit opt-in` -- `e27f44c fix: harden responses websocket protocol` -- `2e4733d fix: preserve websocket turn cancellation hooks` -- `df8c047 fix: abort websocket turns before upstream headers` -- `91b3671 fix: propagate websocket aborts to sidecars` -- `78ca706 fix: abort web-search loop provider fetches` - -Modified: - -- `/Users/jun/Developer/new/700_projects/opencodex/.gitignore` -- `/Users/jun/Developer/new/700_projects/opencodex/src/config.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/types.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/codex-inject.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/codex-catalog.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/responses/parser.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/server.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/ws-bridge.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/codex-inject.test.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/ws-endpoint.test.ts` - -## Closed Findings - -- Routed continuation safety: routed bridged WS responses now emit empty response ids, so Codex - falls back to a full next request instead of sending an unresolved incremental suffix. -- Body type handling: successful WS bodies are classified as SSE, mislabelled SSE, JSON, or - protocol error. -- Terminal enforcement: the WS pump stops at the first terminal, cancels the reader, and reports EOF - before terminal as standalone `type:error`. -- Header/error fidelity: non-2xx responses use standalone `type:error` with HTTP status and safe - headers, including Codex rate-limit header families; inbound WS data stores only forwarded - allowlist headers. -- Cancellation: socket close still cancels upstream, and same-socket replacement turns cancel the - previous in-flight reader/fetch before upstream headers arrive, including vision/web-search - sidecar fetches, and suppress stale frames. -- Framing/backpressure: CRLF, multiline data, split chunks, unterminated final event, dropped send, - `-1` backpressure, and bounded sniff replay are covered by tests. -- Advertisement: absent `websockets` is now false; only explicit `websockets: true` advertises - provider/catalog `supports_websockets`. - -## Automated Verification - -- `bun test tests/codex-inject.test.ts tests/codex-catalog.test.ts` - - 16 pass, 0 fail, 104 assertions. -- `bun test tests/ws-endpoint.test.ts` - - 19 pass, 0 fail, 39 assertions. -- `bun test tests/passthrough-abort.test.ts` - - 4 pass, 0 fail, 8 assertions. -- `bun test tests/sidecar-abort.test.ts` - - 3 pass, 0 fail, 10 assertions. -- `bun test tests/sidecar-abort.test.ts tests/ws-endpoint.test.ts tests/passthrough-abort.test.ts` - - 26 pass, 0 fail, 57 assertions. -- `bun test tests` - - 84 pass, 0 fail, 274 assertions. -- `bun x tsc --noEmit` - - passed with exit 0. - -## Live `ocx` Smoke - -Initial state: - -- `ocx status` returned `Proxy not running`. -- `~/.opencodex/config.json` had `websockets` absent, with providers `openai`, `opencode-go`, - `anthropic`. - -Default advertisement smoke: - -- `ocx start` started the proxy on `http://localhost:10100`. -- `GET /healthz` returned HTTP 200. -- `GET /v1/models?client_version=0.141.0` returned catalog rows with `supports_websockets` absent - for: - - `gpt-5.5` - - `opencode-go/kimi-k2.7-code` - - `opencode-go/minimax-m3` - -Direct WebSocket smoke: - -- `ws://localhost:10100/v1/responses` with `generate:false` returned - `response.created -> response.completed` and empty response id `""`. -- Direct routed WebSocket one-shot with `opencode-go/kimi-k2.6` returned: - - `result completed` - - 8 frames - - first frame `response.created` - - last frame `response.completed` - - text `OK` - -Shutdown: - -- `ocx stop` stopped PID 23557. -- Final `ocx status` must remain `Proxy not running` before release. - -## Residual Note - -Codex RS exposes no standalone `response.cancel` client frame in the checked source. Phase 132 -therefore implements the server-side safe behavior available to opencodex: socket close cancellation -and same-socket replacement-turn cancellation. A human-visible TUI Ctrl-C transcript remains useful -release evidence if automation can drive it reliably, but it is no longer the only proof of stale -frame isolation. - -## Independent Review Follow-up - -The first read-only Phase 132 release review failed on two remaining release-blocking points: - -- Cancellation was installed only after an upstream `Response` existed. Fix: a turn-level - `AbortController` is now installed immediately on `response.create`, plumbed into - `handleResponses`, and linked to both passthrough and bridged upstream fetches. -- The safe WebSocket error header allowlist did not retain Codex rate-limit headers. Fix: - `safeResponseHeaders()` now preserves the `x-codex--primary/secondary-*` and - `x-codex--limit-name` families parsed by Codex RS. - -Regression coverage: - -- `tests/passthrough-abort.test.ts` asserts that a turn-level abort signal aborts the upstream - controller before response headers arrive. -- `tests/ws-endpoint.test.ts` asserts stale successful response bodies are cancelled before pumping - and Codex rate-limit headers survive WebSocket error sanitization. - -The second read-only Phase 132 release review found a remaining sidecar cancellation gap. Fix: - -- `options.abortSignal` now threads through `describeImagesInPlace`, `describeImage`, - `runWithWebSearch`, and `runWebSearch`. -- `src/abort.ts` composes the per-turn abort signal with each sidecar timeout signal. -- `tests/sidecar-abort.test.ts` asserts both web-search and vision sidecar fetches observe the - WebSocket turn abort signal. - -The third read-only Phase 132 release review found one last web-search loop gap. Fix: - -- The routed-provider fetch inside `runWithWebSearch` now receives the WebSocket turn abort signal. -- `tests/sidecar-abort.test.ts` asserts the loop's routed-provider fetch receives and observes the - same abort signal. diff --git a/devlog/_fin/133_websocket-default-on/00_verification.md b/devlog/_fin/133_websocket-default-on/00_verification.md deleted file mode 100644 index 72e1e63a5..000000000 --- a/devlog/_fin/133_websocket-default-on/00_verification.md +++ /dev/null @@ -1,50 +0,0 @@ -# 133.00 — WebSocket Default-On Release - -> **Superseded by e804ba5 (v1.9.2)**: WebSocket default was reverted to **off** (`config.websockets === true` required). Fresh config writes `"websockets": false`. The decision below was the original v1.9.1 intent but was reversed before v2.0.0. - -## Decision (original, now reversed) - -After Phase 132 hardened the Responses WebSocket bridge and the user verified local Codex behavior, -WebSocket advertisement is restored to default-on for `1.9.1`. - -## Behavior - -- Missing `websockets` now means enabled. -- Fresh `~/.opencodex/config.json` writes `"websockets": true`. -- Explicit `"websockets": false` still suppresses provider/catalog `supports_websockets`. - -## Changed - -- `/Users/jun/Developer/new/700_projects/opencodex/src/config.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/src/types.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/codex-inject.test.ts` -- `/Users/jun/Developer/new/700_projects/opencodex/tests/codex-catalog.test.ts` - -## Verification Plan - -- `bun test tests/codex-inject.test.ts tests/codex-catalog.test.ts` -- `bun test tests` -- `bun x tsc --noEmit` -- `ocx start` smoke: - - `/healthz` returns HTTP 200. - - `/v1/models?client_version=0.141.0` advertises `supports_websockets` by default. - - `ocx stop` leaves the proxy stopped before release. - -## Verification Results - -- `bun test tests/codex-inject.test.ts tests/codex-catalog.test.ts` - - 17 pass, 0 fail, 107 assertions. -- `bun test tests` - - 85 pass, 0 fail, 277 assertions. -- `bun x tsc --noEmit` - - passed with exit 0. -- Local source smoke via `bun src/cli.ts start` - - `/healthz` returned HTTP 200. - - `/v1/models?client_version=0.141.0` returned `supports_websockets: true` for: - - `gpt-5.5` - - `opencode-go/kimi-k2.7-code` - - `opencode-go/minimax-m3` - - Context values stayed correct for the checked routed models: - - `opencode-go/kimi-k2.7-code`: `262144`, auto compact `235929` - - `opencode-go/minimax-m3`: `512000`, auto compact `460800` - - `ocx stop` stopped the proxy and final `ocx status` returned `Proxy not running`. diff --git a/devlog/_fin/140_remaining-provider-ports/00_overview.md b/devlog/_fin/140_remaining-provider-ports/00_overview.md deleted file mode 100644 index 7ea29915f..000000000 --- a/devlog/_fin/140_remaining-provider-ports/00_overview.md +++ /dev/null @@ -1,94 +0,0 @@ -# 140.00 — Overview: Remaining jawcode Provider Ports into opencodex - -## What this phase is - -After prior opencodex port cycles, jawcode still ships **five live providers** that opencodex -does not route today. This phase surveys jawcode's wire/auth/streaming implementations, maps -each provider onto opencodex's **five-adapter** architecture, and produces a sequenced port -plan. It is a **foundation cycle** (research + decision) — three docs (00–02) only. - -**No production code changes in this cycle.** Implementation is a later, approval-gated step -(mirror phases `110/00_overview.md:13-14` and `120/00_overview.md:10-11`). - -## TL;DR - -1. opencodex resolves upstream calls through **exactly five adapters** today - (`server.ts:72-86`: `openai-chat`, `anthropic`, `openai-responses`, `google`, `azure-openai`). - Every remaining port either **extends** one of those adapters (auth/URL/body variants) or - requires a **new adapter id** plus `resolveAdapter()` wiring — there is no sixth slot yet. -2. Of the five un-ported providers, **two are Gemini-family with reuse paths**: - `google-vertex` (Vertex `streamGenerateContent` + GCP ADC) and `google-antigravity` (Cloud - Code Assist OAuth envelope over Gemini-shaped SSE). Both are **MEDIUM** effort if the - existing `google` adapter gains pluggable auth/URL/body hooks (`adapters/google.ts:89-117`). -3. **Three are structurally non-OpenAI-compatible**: `cursor` (HTTP/2 Connect+protobuf agent - protocol), `amazon-bedrock` (SigV4 + `bedrock-converse-stream` eventstream), `kiro` - (CodeWhisperer streaming + Bearer + eventstream). Each needs a **new adapter** — **HARD**. -4. Recommended sequencing: **vertex → antigravity → bedrock → kiro → cursor** (easiest auth/wire - reuse first; cursor last because of bidirectional exec/MCP and conversation state). -5. Model surface (jawcode `models.json` + static lists): **119** bedrock, **145** cursor, - **15** antigravity, **13** vertex; **kiro** has **8** static models in - `special.ts:82-91` (none in `models.json`). - -## The five-adapter constraint - -| Adapter | jawcode APIs it can cover (partially) | Auth modes opencodex supports today | -|---------|--------------------------------------|-------------------------------------| -| `openai-chat` | OpenAI-compatible chat/completions | `key`, `oauth` (`oauth/index.ts:19-64`) | -| `anthropic` | Anthropic Messages | `key`, `oauth` | -| `openai-responses` | Codex/Responses passthrough | `forward`, `key` | -| `google` | `google-generative-ai` (AI Studio) | `key` (`x-goog-api-key`, `google.ts:113`) | -| `azure-openai` | Azure Responses passthrough | `key` | - -The `ProviderAdapter` contract (`adapters/base.ts:8-20`) is small: `buildRequest`, -`parseStream`, optional `parseResponse`. Ports must implement that surface (then the existing -bridge in `server.ts` re-encodes to Responses SSE). - -## The five un-ported providers (scope of 140) - -| Provider | jawcode `Api` | Alive? | Port axis | -|----------|---------------|--------|-----------| -| `google-antigravity` | `google-gemini-cli` (shared wire) | **Yes** — active product | Extend `google` + OAuth | -| `google-vertex` | `google-vertex` | Yes | Extend `google` + GCP auth | -| `cursor` | `cursor-agent` | Yes | **New adapter** + Cursor OAuth | -| `amazon-bedrock` | `bedrock-converse-stream` | Yes | **New adapter** + AWS SigV4 | -| `kiro` | `kiro-streaming` | Yes | **New adapter** + Kiro token import | - -**Explicitly out of this list:** `openai-codex` — already covered by opencodex native `openai` -forward + `openai-apikey` options. **`google-gemini-cli`** — excluded as dead per product -direction (Antigravity supersedes it for Cloud Code Assist OAuth). - -## Why some ports are hard - -| Factor | vertex / antigravity | bedrock / kiro | cursor | -|--------|---------------------|----------------|--------| -| Wire shape | Gemini SSE (`alt=sse`) | Amazon eventstream JSON | Connect+proto over HTTP/2 | -| OpenAI-compat | No (Gemini body) | No | No — agent RPC, not chat | -| Auth | GCP OAuth / ADC | AWS keys / Bearer | Cursor OAuth poll | -| Streaming quirks | CCA wrapper vs Vertex URL | Block-indexed converse events | Bidirectional exec + KV blobs | -| Tool loop | functionCall in SSE | toolUse blocks | Server-driven exec messages | - -Cursor is the outlier: jawcode's `cursor.ts` is ~2.6k lines because the upstream is an **agent -runtime** (`AgentService/Run` at `cursor.ts:357-368`), not a completion endpoint. opencodex -would need either a minimal exec stub or a deliberate "text-only, no server tools" subset. - -## Relationship to prior phases - -| Phase | Axis | Relevance to 140 | -|-------|------|------------------| -| 100 | catalog/policy/error parity (HTTP) | Routed catalog entries need provider config + models | -| 110 | SSE lifecycle reliability | New adapters inherit bridge RC1–RC3 obligations | -| 120 | WebSocket transport parity | Independent; new providers start HTTP/SSE only | -| **140** | **Last jawcode provider ports** | this phase — planning only | - -## Scope & baseline - -- **In scope:** per-provider survey (`01_`), port design + sequencing + risk table (`02_`). -- **Out of scope (this cycle):** adapter implementation, OAuth CLI/GUI flows, catalog sync - changes, tests, or config defaults. Those ship only after explicit approval. -- **Evidence baseline:** jawcode at `packages/ai/src/providers/*.ts`, `types.ts:35-61`, - `descriptors.ts:287-304`; opencodex at `src/adapters/`, `src/oauth/`, `src/server.ts:72-86`. - -## Documents - -- `01_provider-survey.md` — wire/auth/streaming/quirks per provider with jawcode file:line evidence -- `02_port-plan.md` — adapter reuse vs new, auth plan, difficulty table, sequencing, risks diff --git a/devlog/_fin/140_remaining-provider-ports/01_provider-survey.md b/devlog/_fin/140_remaining-provider-ports/01_provider-survey.md deleted file mode 100644 index 371d8b3ef..000000000 --- a/devlog/_fin/140_remaining-provider-ports/01_provider-survey.md +++ /dev/null @@ -1,261 +0,0 @@ -# 140.01 — Provider Survey (jawcode source of truth) - -Evidence from jawcode implementations and model catalogs. Model counts are rows in -`packages/ai/src/models.json` unless noted as static. - -## Summary table - -| Provider | Models | jawcode `Api` | Primary implementation | Auth mechanism | -|----------|--------|---------------|------------------------|----------------| -| `google-antigravity` | **15** | `google-gemini-cli` | `google-gemini-cli.ts` (shared) | OAuth JSON credential + Bearer | -| `google-vertex` | **13** | `google-vertex` | `google-vertex.ts` + `google-shared` | API key **or** GCP ADC Bearer | -| `amazon-bedrock` | **119** | `bedrock-converse-stream` | `amazon-bedrock.ts` | AWS SigV4 (profile/env/ADC chain) | -| `kiro` | **8** static | `kiro-streaming` | `kiro.ts` | Bearer (+ kiro-cli SQLite / refresh) | -| `cursor` | **145** | `cursor-agent` | `cursor.ts` (+ `cursor/gen/agent_pb`) | Cursor OAuth access token | - -Descriptor defaults (`descriptors.ts:301-304`, `291`): bedrock -`us.anthropic.claude-opus-4-6-v1`, antigravity `gemini-3-pro-high`, vertex -`gemini-3-pro-preview`, cursor `claude-sonnet-4-6`, kiro `kiro-auto`. - ---- - -## 1. `google-antigravity` - -### Status - -**Alive.** Shares the Cloud Code Assist stream implementation with `google-gemini-cli`; provider -slug is switched at runtime (`google-gemini-cli.ts:1-5`, `291`). - -### Wire protocol - -- **Endpoint:** Antigravity daily/sandbox fallbacks - (`google-gemini-cli.ts:69-72`, `310`): `daily-cloudcode-pa.googleapis.com`, - `daily-cloudcode-pa.sandbox.googleapis.com`. -- **Path:** `POST …/v1internal:streamGenerateContent?alt=sse` (`338`). -- **Body envelope:** `CloudCodeAssistRequest` — top-level `{ project, model, request, requestType?, userAgent?, requestId? }` - (`195-222`, `785-796`). Antigravity adds `requestType: "agent"`, `userAgent: "antigravity"`, - `requestId: "agent-{uuid}"` (`789-794`). -- **Inner request:** Gemini-shaped `contents`, `systemInstruction`, `generationConfig`, - `tools`, `toolConfig` (`727-756`). -- **Streaming:** SSE JSON chunks with nested `response.candidates[].content.parts` - (`394-501`) — text, `thought`/`thoughtSignature`, `functionCall`. - -### Auth - -- Requires OAuth credential JSON in `options.apiKey` (`286-288`). -- Parsed via `parseGeminiCliCredentials` → `{ accessToken, projectId, refreshToken?, expiresAt? }` - (`143-179`). -- `Authorization: Bearer ${accessToken}` (`320`). -- Login/onboard: `utils/oauth/google-antigravity.ts` — Google OAuth + `loadCodeAssist` / - `onboardUser` project provisioning (`116-172`), stores project id in credential blob. -- Refresh skew: 60s for antigravity (`89`, `182-192`). - -### Streaming / fidelity quirks - -| Quirk | Evidence | -|-------|----------| -| Antigravity session id derived from first user text hash | `653-659`, `731-733` | -| Claude on Antigravity: `anthropic-beta` header, `VALIDATED` tool mode | `95-97`, `324`, `765-770` | -| System instruction injection for Claude/Gemini-3 | `773-783` | -| Tool schema: `parametersJsonSchema` → `parameters` via `normalizeSchemaForCCA` | `662-678` | -| Empty-stream retry (up to 2) | `513-557` | -| Endpoint failover on retry attempts | `337-338` | -| Deletes `maxOutputTokens` for non-Claude antigravity models | `758-763` | - -### Model discovery - -Dynamic via `fetchAntigravityDiscoveryModels` when OAuth token present -(`google.ts:47-66`, `markUnlistedOutsideDynamic: true`). - -### opencodex gap - -Existing `google` adapter targets AI Studio URL -(`google.ts:108-110`: `/v1beta/models/{id}:streamGenerateContent`) with flat Gemini body and -`x-goog-api-key` — **not** the CCA envelope or Bearer auth. - ---- - -## 2. `google-vertex` - -### Wire protocol - -- Delegates to `streamGoogleGenAI` with `api: "google-vertex"` (`google-vertex.ts:24-28`). -- **Two URL modes:** - - API key: `https://aiplatform.googleapis.com/v1/publishers/google/models/{id}:streamGenerateContent?alt=sse` - with `x-goog-api-key` (`37-44`). - - ADC: `https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/google/models/{id}:streamGenerateContent?alt=sse` - with `Authorization: Bearer` (`47-57`). -- **Body:** standard `buildGoogleGenerateContentParams` (Gemini generateContent shape via - `google-shared`). -- **Streaming:** same SSE JSON as AI Studio path inside `streamGoogleGenAI`. - -### Auth - -- API key: `GOOGLE_CLOUD_API_KEY` or real-looking `options.apiKey` (`61-67`). -- ADC: `getVertexAccessToken` from `google-auth.ts:1-13` — service account JWT, authorized_user - refresh, or GCE metadata (`6-9`). -- Project/location required for ADC path (`69-87`). - -### Quirks - -- `retainTextSignature: true` for vertex (`28`) — thought signatures preserved for tool/thinking - continuity. -- No dynamic model discovery yet (`google.ts:36-44` comment). - -### opencodex gap - -`google` adapter hardcodes AI Studio base URL and API key header. Vertex needs configurable -base host, path prefix (`projects/…/locations/…`), and Bearer injection from ADC refresh. - ---- - -## 3. `amazon-bedrock` - -### Wire protocol - -- **API:** Bedrock Runtime `ConverseStream` — `POST /model/{modelId}/converse-stream` - (`amazon-bedrock.ts:217-218`). -- **Accept:** `application/vnd.amazon.eventstream` (`244`). -- **Body:** `ConverseStreamRequest` — `messages`, optional `system`, `inferenceConfig`, - `toolConfig`, `additionalModelRequestFields` for thinking (`204-214`, `763-810`). -- **Streaming:** AWS eventstream frames decoded by `decodeEventStream` (`276`); event types - `messageStart`, `contentBlockStart/Delta/Stop`, `messageStop`, `metadata` (`298-334`). - -### Auth - -- `resolveAwsCredentials` with optional profile (`233-237`); test bypass - `AWS_BEDROCK_SKIP_AUTH` (`230-231`). -- Request signed via custom `signRequest` (SigV4, no AWS SDK) (`246-255`). - -### Message / tool mapping - -- Messages converted to Bedrock wire roles with cache points for Anthropic models - (`570-714`). -- Tool results batched into single user message (`662-696`). -- Thinking: `additionalModelRequestFields.thinking` with adaptive vs budget modes - (`763-810`); `thinkingDisplay` default `"summarized"` (`65`, `779`). - -### Model catalog - -- **119** models in `models.json`. -- Discovery via models.dev key `amazon-bedrock` (`openai-compat.ts:2170-2204`) with cross-region - id transform and EU variant duplication for Claude (`2184-2202`). - -### opencodex gap - -No Bedrock adapter, no SigV4 signer, no eventstream parser in opencodex. Cannot reuse -`anthropic` adapter — request/response shapes differ despite similar message semantics. - ---- - -## 4. `kiro` - -### Wire protocol - -- **Host:** `https://runtime.{region}.kiro.dev/` (`kiro.ts:49-50`, `545-546`). -- **Headers:** AWS-style `x-amz-target: - AmazonCodeWhispererStreamingService.GenerateAssistantResponse` (`50`, `92`). -- **Content-Type:** `application/x-amz-json-1.0` (`90`). -- **Accept:** `application/vnd.amazon.eventstream` (`91`). -- **Body:** `conversationState` with `history`, `currentMessage.userInputMessage`, optional - `tools` / `toolResults` in context (`145-240`). -- **Streaming:** shared `decodeEventStream` (`591`); payload JSON heuristics in - `parseKiroPayload` (`256-305`) — `content`, tool `{name,input,toolUseId}`, `{stop:true}`, - `{usage}`. - -### Auth - -- Multi-source resolver `resolveKiroAuth` (`392-487`): explicit token, `aoa*` apiKey prefix, - `KIRO_ACCESS_TOKEN`, cache + refresh, prokiro `auth.json`, kiro-cli SQLite - (`328-368`, `462-484`). -- Refresh: `POST https://prod.{region}.auth.desktop.kiro.dev/refreshToken` (`312`, `371-389`). -- **Anti-detection:** machine fingerprint in User-Agent (`72-83`, `85-98`). - -### Quirks - -| Quirk | Evidence | -|-------|----------| -| Model id mapping table (`kiro-auto` → `auto`, etc.) | `728-749` | -| Stable `conversationId` hash from messages | `751-765` | -| Tool name truncated to 64 chars | `121` | -| No thinking stream — text + tools only | `602-671` | -| 401 → refresh + single retry | `562-579` | - -### Models - -**8** static entries in `special.ts:82-91` (not in `models.json`). Default `kiro-auto` -(`descriptors.ts:222`). - -### opencodex gap - -Proprietary payload + IDE impersonation headers. Eventstream decoder could be **shared** with -Bedrock port, but request builder and auth import path are Kiro-specific. - ---- - -## 5. `cursor` - -### Wire protocol - -- **Not an LLM REST API.** HTTP/2 Connect to `AgentService/Run` (`cursor.ts:133-134`, `355-368`). -- **Content-Type:** `application/connect+proto` (`360`). -- **Framing:** 5-byte Connect frames (`175-180`, `412-421`); protobuf payloads - (`AgentClientMessage` / `AgentServerMessage`). -- **Streaming:** `interactionUpdate` cases — `textDelta`, `thinkingDelta`, `toolCallStarted`, - `toolCallDelta`, `toolCallCompleted`, `turnEnded` (`1955-2098`). -- **Bidirectional:** server sends `execServerMessage`, `kvServerMessage`; client must respond - (`611-622`, `968-1203`). - -### Auth - -- Bearer access token required (`337-340`). -- OAuth: PKCE + browser login + poll (`utils/oauth/cursor.ts:4-76`); refresh via - `exchange_user_api_key` (`98+`). -- Descriptor: `oauthProvider: "cursor"` (`descriptors.ts:291`). - -### Conversation / prompt state - -- `buildGrpcRequest` maintains `conversationState`, blob store, `rootPromptMessagesJson` - (`2509-2640`) — critical for multi-turn (`2273-2276`). -- Host override system prompt so model doesn't assume Cursor IDE (`2293-2297`). -- Tools advertised via `requestContext` exec handshake, not in initial Run request - (`2616-2617`, `977-998`). - -### Exec / tools (major complexity) - -- Shell, read, write, grep, ls, mcp, etc. handlers (`1005-1167`). -- Without `execHandlers`, tools return "Tool not available" / "Not implemented" - (`1256-1257`, `1108`, `1135`). -- Heartbeat every 5s (`468-479`). - -### Models - -**145** entries in `models.json`. Dynamic discovery when API key present -(`special.ts:46-58` → `fetchCursorUsableModels`). - -### opencodex gap - -Requires new adapter with HTTP/2 + protobuf + optional exec bridge. Cannot map to any existing -five adapters. Highest risk: Codex tool calls vs Cursor native tool loop mismatch. - ---- - -## Cross-reference: jawcode type system - -`types.ts:35-61` registers all five APIs in `KnownApi` / `ApiOptionsMap`. Provider slugs -(`types.ts:101-149`) include all five survey subjects. Stream dispatch: -`register-builtins.ts:358-371` (gemini-cli/antigravity), `366-371` (vertex), plus bedrock, -cursor, kiro lazy loaders in the same file. - -## opencodex adapter interface (target shape) - -Every port must satisfy (`adapters/base.ts:8-20`): - -```typescript -buildRequest(parsed, incoming?) → { url, method, headers, body } -parseStream(response) → AsyncGenerator -parseResponse?(response) → Promise -``` - -OAuth pattern reference: `oauth/index.ts:11-17` (`login`, `refresh`, `providerConfig`, -`defaultModel`) — exemplar flow in `oauth/xai.ts:1-71` (PKCE, discovery, token refresh). diff --git a/devlog/_fin/140_remaining-provider-ports/02_port-plan.md b/devlog/_fin/140_remaining-provider-ports/02_port-plan.md deleted file mode 100644 index 02905db1a..000000000 --- a/devlog/_fin/140_remaining-provider-ports/02_port-plan.md +++ /dev/null @@ -1,243 +0,0 @@ -# 140.02 — Port Plan (adapter design, auth, sequencing) - -**Planning only.** No implementation in cycle 140. Approval gates each sub-phase before code -lands (same contract as `110/00_overview.md:83-84`, `120/00_overview.md:64-66`). - -## Design principle: five adapters → extend or add - -`resolveAdapter()` (`server.ts:72-86`) is a closed switch. Options per provider: - -1. **Extend `google`** — new `OcxProviderConfig` fields (e.g. `googleMode: "ai-studio" | "vertex" | "cloud-code-assist"`) interpreted inside `createGoogleAdapter`. -2. **Add adapter ids** — e.g. `bedrock`, `kiro`, `cursor` with new files under `src/adapters/` and new `case` arms in `resolveAdapter()`. - -Adding adapters does not break the "five adapter" *concept* if we treat Google variants as one -adapter with modes; the table below shows both views. - ---- - -## Per-provider port design - -### A. `google-vertex` — extend `google` adapter (MEDIUM) - -| Aspect | Plan | -|--------|------| -| Adapter | Reuse `google` (`adapters/google.ts`) | -| URL | Branch on config: Vertex path template from `google-vertex.ts:38-51` vs AI Studio `:109-110` | -| Auth | `authMode: "gcp-adc"` (new) or env-driven: refresh Bearer via ported `google-auth.ts` logic; optional `x-goog-api-key` path unchanged | -| Body | Same `messagesToGeminiFormat` — already matches `google-shared` output | -| Config | `project`, `location`, `GOOGLE_CLOUD_*` env aliases | -| Models | Static seed **13** from jawcode bundle + optional future discovery | -| OAuth | **None** — ADC / service account / API key only | - -**Effort drivers:** ADC token cache + refresh in opencodex (port ~250 lines from -`google-auth.ts`), config schema, tests with mocked token endpoint. - -**Risk:** LOW–MEDIUM — wire matches existing `parseStream` SSE parser (`google.ts:119-184`). - ---- - -### B. `google-antigravity` — extend `google` adapter + OAuth (MEDIUM) - -| Aspect | Plan | -|--------|------| -| Adapter | Reuse `google` with `googleMode: "cloud-code-assist-antigravity"` | -| URL | Configurable base + `/v1internal:streamGenerateContent?alt=sse` (`google-gemini-cli.ts:338`) | -| Auth | New `OAUTH_PROVIDERS` entry `google-antigravity` mirroring jawcode `utils/oauth/google-antigravity.ts:160-172` (login + refresh + project id in stored JSON) | -| Body | Wrap flat Gemini body in CCA envelope `{ project, model, request, requestType, userAgent, requestId }` — port `buildRequest` antigravity branches (`681-796`) | -| Headers | Antigravity User-Agent, optional `anthropic-beta` for Claude (`317-324`) | -| Credential shape | Store `{ token, projectId, refreshToken, expiresAt }` as serialized JSON in oauth store (jawcode `parseGeminiCliCredentials` pattern, `143-179`) | -| Models | Dynamic discovery port from `fetchAntigravityDiscoveryModels` — **15** bundled + live fetch | - -**Effort drivers:** OAuth onboarding (`loadCodeAssist` / `onboardUser`, `116-172`), antigravity -quirks table in `01_provider-survey.md`, endpoint fallback list. - -**Risk:** MEDIUM — Google OAuth client id/secret must ship as env/config -(`GOOGLE_ANTIGRAVITY_CLIENT_ID`, `google-antigravity.ts:10-11`); quota/tier provisioning -failures are user-visible. - ---- - -### C. `amazon-bedrock` — new `bedrock` adapter (HARD) - -| Aspect | Plan | -|--------|------| -| Adapter | **New** `src/adapters/bedrock.ts` | -| Wire | Port `ConverseStreamRequest` builder + event handlers from `amazon-bedrock.ts:204-511` | -| Auth | Port `resolveAwsCredentials` + `signRequest` (`aws-credentials.ts`, `aws-sigv4.ts`) — keep zero `@aws-sdk/*` | -| Streaming | Port `decodeEventStream` (`aws-eventstream.ts`) to `src/adapters/` or `src/lib/` | -| Config | `region`, `profile`, optional `AWS_*` env; model ids as Bedrock ARNs/cross-region ids | -| Bridge | Map eventstream → `AdapterEvent` (text/thinking/tool_call_start/delta/end/done) | -| Models | Seed from jawcode filter logic (`openai-compat.ts:2177-2203`) — **119** rows | - -**Effort drivers:** SigV4 correctness, thinking/cache/tool edge cases, large model matrix. - -**Risk:** HIGH — auth misconfiguration silent failures; thinking signature errors already -diagnosed in jawcode (`amazon-bedrock.ts:355-374`). - ---- - -### D. `kiro` — new `kiro` adapter (HARD) - -| Aspect | Plan | -|--------|------| -| Adapter | **New** `src/adapters/kiro.ts` | -| Wire | Port `buildPayload` + `parseKiroPayload` (`kiro.ts:145-305`) | -| Auth | **Import-first**, not full browser OAuth: read kiro-cli SQLite + refresh (`kiro.ts:392-487`, `utils/oauth/kiro.ts:1-50`); optional `ocx login kiro` = import path | -| Headers | Port fingerprint User-Agent builder (`85-98`) — required for upstream acceptance | -| Streaming | Reuse eventstream decoder shared with Bedrock port | -| Models | Static **8** from `special.ts:82-91` | -| Config | `region`, `profileArn`, `KIRO_ACCESS_TOKEN` | - -**Effort drivers:** Anti-detection headers, token refresh, conversation id stability. - -**Risk:** HIGH — upstream may rate-limit non-IDE clients; ToS/impersonation policy decision -needed before shipping. - ---- - -### E. `cursor` — new `cursor` adapter (HARDEST) - -| Aspect | Plan | -|--------|------| -| Adapter | **New** `src/adapters/cursor.ts` (+ protobuf codegen or shared buf module) | -| Transport | HTTP/2 client (`cursor.ts:355-368`) — Bun `http2.connect` | -| Auth | Port `oauth/cursor.ts` poll + refresh into `OAUTH_PROVIDERS.cursor` | -| MVP scope | **Text + thinking only**, exec handlers stubbed (reject server tools with structured error) OR map Codex tool calls to MCP exec if feasible later | -| State | Port minimal `conversationState` + `rootPromptMessagesJson` (`2316-2599`) for multi-turn | -| Models | Dynamic `fetchCursorUsableModels` when logged in — **145** catalog rows | - -**Effort drivers:** Protobuf schema maintenance, bidirectional stream, exec loop (optional phase). - -**Risk:** VERY HIGH — without exec bridge, Cursor models that invoke native tools may stall; -with exec bridge, opencodex becomes a partial Cursor CLI host. - -**Recommended sub-phases:** (1) login + single-turn text, (2) multi-turn state, (3) tool/exec -parity (optional / separate approval). - ---- - -## Difficulty table - -| Provider | Adapter strategy | Auth work | Wire complexity | Est. effort | Est. risk | -|----------|------------------|-----------|-----------------|-------------|-----------| -| `google-vertex` | Extend `google` | ADC / API key | Low (Gemini SSE) | **MEDIUM** | Low–Med | -| `google-antigravity` | Extend `google` | OAuth + project onboard | Med (CCA envelope) | **MEDIUM** | Med | -| `amazon-bedrock` | **New `bedrock`** | SigV4 + AWS chain | High (eventstream) | **HARD** | High | -| `kiro` | **New `kiro`** | Token import + refresh | High (eventstream) | **HARD** | High | -| `cursor` | **New `cursor`** | OAuth poll | Very high (Connect/proto/exec) | **HARD+** | Very high | - -MLB-style rough grades (20–80): vertex **60**, antigravity **58**, bedrock **45**, kiro **42**, -cursor **35** on "port readiness." - ---- - -## Recommended sequencing - -``` -140.1 google-vertex ──► extend google + GCP auth (validates adapter hooks) -140.2 google-antigravity ──► extend google + OAuth (reuses 140.1 hook work) -140.3 amazon-bedrock ──► new adapter + shared eventstream lib -140.4 kiro ──► new adapter (reuses eventstream from 140.3) -140.5 cursor ──► new adapter (isolated; largest unknown) -``` - -**Rationale:** - -1. **Vertex before Antigravity** — both Gemini SSE parsers; vertex avoids OAuth/onboarding - before CCA envelope work. -2. **Bedrock before Kiro** — establishes `decodeEventStream` once (`kiro.ts:31`, `amazon-bedrock.ts:37`). -3. **Cursor last** — no shared code with prior ports; exec/MCP scope can ship independently. - -Parallelization: 140.1 + 140.3 can run in parallel after plan approval (disjoint files). - ---- - -## Shared infrastructure (cross-cutting) - -| Component | Consumers | jawcode source | -|-----------|-----------|----------------| -| Eventstream decoder | bedrock, kiro | `aws-eventstream.ts` | -| SigV4 signer | bedrock | `aws-sigv4.ts`, `aws-credentials.ts` | -| GCP ADC | vertex | `google-auth.ts` | -| OAuth registry entry | antigravity, cursor | pattern: `oauth/index.ts:19-64`, `oauth/xai.ts` | -| `google` adapter hooks | vertex, antigravity | refactor `google.ts:93-116` | - ---- - -## opencodex config sketch (non-binding) - -Illustrative only — not implemented this cycle: - -```json -{ - "google-vertex": { - "adapter": "google", - "googleMode": "vertex", - "authMode": "gcp-adc", - "project": "${GOOGLE_CLOUD_PROJECT}", - "location": "${GOOGLE_CLOUD_LOCATION}", - "models": ["gemini-3-pro-preview"] - }, - "google-antigravity": { - "adapter": "google", - "googleMode": "cloud-code-assist", - "authMode": "oauth", - "baseUrl": "https://daily-cloudcode-pa.googleapis.com", - "defaultModel": "gemini-3-pro-high" - }, - "amazon-bedrock": { - "adapter": "bedrock", - "authMode": "aws", - "region": "us-east-1", - "defaultModel": "us.anthropic.claude-opus-4-6-v1" - } -} -``` - ---- - -## Risks & mitigations - -| Risk | Impact | Mitigation | -|------|--------|------------| -| Bridge RC1/RC3 (phase 110) on new adapters | Codex stream errors | Reuse `done` terminal + heartbeat patterns from existing adapters | -| OAuth secret distribution (Antigravity) | Login blocked | Document env vars; no secrets in repo | -| AWS/Kiro credential exposure in logs | Security | Follow opencodex redaction conventions from existing oauth store | -| Cursor tool deadlock without exec | Hung streams | MVP: cap models to non-agent/composer-only; document limitation | -| Bedrock model id drift | Wrong ARN | Live sync from models.dev or periodic jawcode bundle import | -| Sixth adapter proliferation | Maintenance | Keep Google as one adapter with modes; share eventstream module | - ---- - -## Verification plan (for implementation phases) - -| Provider | Minimal proof | -|----------|---------------| -| vertex | `ocx` stream with ADC; compare SSE parts to jawcode golden | -| antigravity | `ocx login google-antigravity`; one tool call + thinking model | -| bedrock | Converse stream against `us.anthropic.claude-*`; SigV4 unit test | -| kiro | Import kiro-cli token; `kiro-auto` single turn | -| cursor | OAuth login; single-turn text on `composer-*`; multi-turn regression | - -Run `bun test` + `bun x tsc --noEmit` after each sub-phase (baseline per `110/00_overview.md:85-86`). - ---- - -## Out of scope (implementation cycles) - -- `google-gemini-cli` provider (dead; Antigravity replaces) -- Full Cursor exec parity (shell/MCP/computer-use) -- Bedrock non-Converse APIs (InvokeModel legacy) -- Kiro browser OAuth from scratch (import-only v1) -- WebSocket upstream for any of these (phase 120 scope) -- Catalog `supports_websockets` changes - ---- - -## Approval checklist (before coding) - -- [ ] User approves adapter extension vs new-id layout for Google modes -- [ ] User approves Antigravity OAuth env var requirement -- [ ] User approves Cursor MVP scope (text-only vs exec) -- [ ] User approves Kiro IDE-impersonation headers ethically / ToS -- [ ] Sequencing 140.1–140.5 accepted or reordered diff --git a/devlog/_fin/140_remaining-provider-ports/03_execution-roadmap.md b/devlog/_fin/140_remaining-provider-ports/03_execution-roadmap.md deleted file mode 100644 index b04faf7b5..000000000 --- a/devlog/_fin/140_remaining-provider-ports/03_execution-roadmap.md +++ /dev/null @@ -1,86 +0,0 @@ -# 140.03 — Phased execution roadmap (1 phase = 1 PABCD pass) - -> Status: **PLAN ONLY** — no source modified. Turns the `02` survey into a phased, jawcode-grounded -> execution plan. Each provider = one decade phase = **one PABCD implementation pass** (gated by user -> approval before code lands, same contract as `02:3-4`). Follows the dev skill (modular, verify-first, -> decade numbering). Grounded in jawcode's actual port source (researched per-provider; cites in each phase doc). - ---- - -## The five phases - -| Phase | Provider | Strategy | Difficulty | jawcode core | opencodex doc | -|:-----:|----------|----------|:----------:|--------------|---------------| -| **10** | `google-vertex` | **extend `google`** + ADC | MEDIUM | `google-vertex.ts`, `google-auth.ts` | `10_phase1_google-vertex.md` | -| **20** | `google-antigravity` | **extend `google`** + OAuth | MEDIUM | `google-antigravity.ts`, `oauth/google-antigravity.ts` | `20_phase2_google-antigravity.md` | -| **30** | `amazon-bedrock` | **new adapter** + SigV4 + eventstream | HARD | `amazon-bedrock.ts`, `aws-sigv4.ts`, `aws-eventstream.ts` | `30_phase3_amazon-bedrock.md` | -| **40** | `kiro` | **new adapter** (import auth, reuses eventstream) | HARD | `kiro.ts`, `oauth/kiro.ts` | `40_phase4_kiro.md` | -| **50** | `cursor` | **new adapter** (HTTP/2+protobuf+exec) | HARD+ | `providers/cursor.ts` (2705L) | `50_phase5_cursor.md` → **`devlog/350`** | - -## Sequencing + why (build order matters because of shared infra) - -``` -10 google-vertex ─┐ establishes the google "mode" hook (config-branch in createGoogleAdapter) -20 google-antigravity ─┘ REUSES the 10 google-mode hook + adds OAuth -30 amazon-bedrock ─┐ establishes the shared eventstream decoder + AWS SigV4 (src/lib/) -40 kiro ─┘ REUSES the 30 eventstream decoder; adds import-auth + fingerprint headers -50 cursor ── isolated (no shared code); largest unknown → last (detailed in 350) -``` - -- **10 before 20:** both are Gemini-SSE over the `google` adapter; vertex's mode-hook + ADC land first so antigravity only adds the OAuth + CCA-envelope branch. -- **30 before 40:** bedrock ports the AWS **eventstream binary decoder** into `src/lib/`; kiro reuses it verbatim (`kiro.ts:31` imports the same decoder). -- **50 last:** cursor shares nothing with the others and needs a transport escape hatch — see `350` for the full design. -- **Parallelizable after approval:** 10 (google branch) and 30 (eventstream/sigv4) touch disjoint files → can run concurrently; 20 waits on 10, 40 waits on 30. - -## Shared infrastructure (build once, reuse) - -| Module (NEW in opencodex) | Built in | Reused by | jawcode source | -|---------------------------|:--------:|-----------|----------------| -| `google` adapter mode-hook (`googleMode` config branch) | 10 | 20 | `google.ts` extension | -| GCP ADC resolver (`src/lib/gcp-adc.ts`) | 10 | — | `google-auth.ts` | -| OAuth registry pattern (`OAUTH_PROVIDERS` entry) | 20 | 50 | `oauth/index.ts:36-61` | -| AWS SigV4 signer (`src/lib/aws-auth.ts`, zero `@aws-sdk`) | 30 | — | `aws-sigv4.ts`, `aws-credentials.ts` | -| **AWS eventstream decoder** (`src/lib/eventstream-decoder.ts`) | 30 | **40** | `aws-eventstream.ts` | -| Transport escape hatch (`runTurn` hook) | 50 | — | (opencodex `350` §2) | - -## The "1 phase = 1 PABCD pass" contract - -Each phase doc (`10`–`50`) is scoped to a single PABCD implementation cycle: **P** plan + approval gate → -**A** port shared/auth modules → **B** build the adapter + wire `resolveAdapter` → **C** verify -(`bun test` + `bun x tsc --noEmit` + the per-phase minimal proof from `02:212-222`) → **D** record + ship. -No phase bundles a second provider; a phase may *internally* have sub-steps but stays one approval unit. - -## Global gates (per phase, from dev skill + `02`) - -1. `bun x tsc --noEmit` clean + `bun test` green (baseline per `110/00_overview.md:85-86`). -2. The phase's minimal proof (live or mocked stream, `02:212-222`). -3. No regression in existing adapters (anthropic/google/openai/azure/openai-responses). -4. Approval checklist items for that provider (`02:237-244`) signed off **before** coding. -5. Devlog record of the pass (this folder). - -## Effort rollup (rough, from `02` grades + per-phase briefs) - -| Phase | MVP effort | Risk | -|:-----:|:----------:|:----:| -| 10 vertex | 5–7d | Low–Med | -| 20 antigravity | 5d | Med | -| 30 bedrock | 7–10d | High | -| 40 kiro | 4–6d (after 30) | High | -| 50 cursor | 7–11d (MVP text) | Very High | - -> Order of value/risk: ship 10→20 first (Gemini reach, low risk), then 30→40 (AWS-family), then 50 (cursor) -> as an isolated, separately-approved effort. - -## Audit record (PABCD-A, jaw Backend employee) - -A backend specialist independently audited `10`–`50` against jawcode + opencodex source. **Verdict: -PASS-with-fixes.** Confirmed sound: the extend-`google` vs new-adapter split, the sequencing -(10→20 google-mode hook; 30→40 shared `decodeEventStream`, verified `kiro.ts:31` + `amazon-bedrock.ts:37` -both import it; 50 isolated), the 1-phase-1-PABCD framing, and the auth approaches. **Fixed findings** -(all in Phase 20 + cites): (1) antigravity `parseStream` is **not** unchanged — it nests under -`response.candidates` vs opencodex's top-level `chunk.candidates`, so Phase 20 adds a mode-aware parser; -(2) `projectId` must be read from the stored credential (server injects only the bare token); -(3) added the `cloudcode-pa.googleapis.com` prod-fallback endpoint; (4) corrected jawcode cite prefixes -to `utils/oauth/…`. - -→ Per-phase plans: `10`–`50`. Cursor detail: `devlog/350_cursor-provider-add/`. diff --git a/devlog/_fin/140_remaining-provider-ports/05_reference-repos.md b/devlog/_fin/140_remaining-provider-ports/05_reference-repos.md deleted file mode 100644 index 82c6bfe66..000000000 --- a/devlog/_fin/140_remaining-provider-ports/05_reference-repos.md +++ /dev/null @@ -1,76 +0,0 @@ -# 140.05 — Pinned external reference repos (vertex + antigravity) - -> Pinned for the **un-hardened** provider ports still ahead of us: Phase 10 `google-vertex` -> and Phase 20 `google-antigravity`. Cursor (Phase 50) and Kiro (Phase 40) are essentially -> hardened already, so this doc deliberately does NOT re-pin cursor/kiro references — it -> captures the external SOT we will lean on the way we leaned on jawcode for those two. -> -> Source-of-truth hierarchy for these ports: -> 1. **opencodex itself** — our adapter contract + existing `google` adapter is the primary SOT. -> 2. **jawcode** (`packages/ai/src/providers/*.ts`) — the original internal port reference. -> 3. **External repos below** — independent, actively-maintained implementations to cross-check -> wire/auth/quirks against. Treat as evidence to verify, not code to copy. - ---- - -## Primary external SOT — `router-for-me/CLIProxyAPI` - -- URL: https://github.com/router-for-me/CLIProxyAPI -- Lang: Go · License: MIT · ~38.7k stars · actively maintained (commits within days). -- Scope: wraps **Antigravity**, ChatGPT Codex, Claude Code, Grok as OpenAI/Gemini/Claude/Codex - compatible APIs. Ships **both** Vertex and Antigravity as first-class, fully-tested providers — - this is the closest external analogue to what we are building. - -Why it is the right reference: its internal layering maps almost 1:1 onto opencodex's adapter model. - -| CLIProxyAPI layer | opencodex equivalent | -|-------------------|----------------------| -| `internal/auth/*` (OAuth + credential parse/refresh) | `src/oauth/*`, `src/lib/gcp-adc.ts` | -| `internal/translator/*` (request/response body conversion) | body conversion in `src/adapters/google.ts` | -| `internal/runtime/executor/*` (stream, retry, refresh, signature) | `buildRequest`/`parseStream` + stabilization logic | -| `internal/signature/*` (gemini sanitize/validate) | gemini param/tool normalization | - -### Vertex — file pointers (Phase 10 cross-check) - -- Auth/credentials: `internal/auth/vertex/vertex_credentials.go`, `internal/auth/vertex/keyutil.go` -- Executor (stream + payload): `internal/runtime/executor/gemini_vertex_executor.go` -- Payload helpers: `internal/runtime/executor/helps/vertex_payload_helpers.go` -- Config compat: `internal/config/vertex_compat.go` -- Import/management: `internal/cmd/vertex_import.go`, `internal/api/handlers/management/vertex_import.go` - -### Antigravity — file pointers (Phase 20 cross-check) - -- Auth (OAuth + project onboard): `internal/auth/antigravity/auth.go`, `constants.go`, `filename.go` -- Login command: `internal/cmd/antigravity_login.go` -- Executor: `internal/runtime/executor/antigravity_executor.go` - - refresh: `antigravity_refresh_test.go`; signature: `antigravity_executor_signature_test.go` - - **reasoning/thoughtSignature replay**: `antigravity_reasoning_replay.go` (+ cache - `internal/cache/antigravity_reasoning_replay_cache.go`) — this is the well-known Antigravity - footgun (Claude+Gemini-3 thought-signature replay); cross-check our Phase 20 parser against it. - - grounding URLs: `internal/runtime/executor/helps/antigravity_grounding_urls.go` -- Translators (CCA envelope, per client wire): - - Gemini: `internal/translator/antigravity/gemini/antigravity_gemini_request.go` + `_response.go` - - Claude: `internal/translator/antigravity/claude/*` (incl. `signature_validation.go`, `web_search.go`) - - OpenAI chat + responses: `internal/translator/antigravity/openai/**` -- Model list fetcher: `cmd/fetch_antigravity_models/main.go`, version `internal/misc/antigravity_version.go` - -## Secondary references (smaller, narrower, still useful) - -- `comgunner/antigravity-studio` — pure-Python Antigravity (Cloud Code Assist) client: text chat + - image-gen + agentic. Good for reading the raw CCA request/response shape without Go noise. -- `twobotass/antigravity-quota` — Antigravity quota check **with auto token refresh**. Useful for - the refresh/quota stabilization slice (mirrors what Kiro got). -- `synthalorian/hermes-gemini-setup-guide` and `synthalorian/claw-code-gemini-setup-guide` — both - document the Cloud Code Assist free-tier OAuth flow and the **thoughtSignature replay fix**; - handy prose confirmation of the replay quirk above. - -## How to use these (parity intent) - -The goal is to bring Phase 10/20 to the same hardening bar Cursor and Kiro now sit at: -retry/backoff, per-attempt timeout, actionable error classification + secret redaction, -fail-closed truncation, estimated-usage tagging, and auth-refresh robustness. When implementing, -diff our behavior against CLIProxyAPI's executor/auth for each of those areas and record the -evidence in the corresponding phase doc (`10_…`, `20_…`). - -> Note: external repos are reference evidence under their own licenses (CLIProxyAPI is MIT). -> Verify wire/auth details against them; do not paste their source into opencodex. diff --git a/devlog/_fin/140_remaining-provider-ports/10_phase1_google-vertex.md b/devlog/_fin/140_remaining-provider-ports/10_phase1_google-vertex.md deleted file mode 100644 index e31abd433..000000000 --- a/devlog/_fin/140_remaining-provider-ports/10_phase1_google-vertex.md +++ /dev/null @@ -1,54 +0,0 @@ -# 140.10 — Phase 1: google-vertex (extend `google`, MEDIUM) - -> One PABCD pass. Extends the existing `google` adapter — **no new adapter id**. Establishes the -> `googleMode` config-branch hook reused by Phase 20. Grounded in jawcode (cites are jawcode paths). -> -> External cross-check: `router-for-me/CLIProxyAPI` Vertex executor/auth — see `05_reference-repos.md`. - ---- - -## Goal - -Let opencodex stream Gemini via **Vertex AI** (project/location endpoints + GCP auth) by reusing the -existing `google` adapter's SSE parser and message conversion — adding only a mode branch + a GCP token resolver. - -## What we port (jawcode) - -- **Auth — ADC resolver** (`google-auth.ts:64-246`, ~180 LOC, pure TS + WebCrypto, no Node deps): - - sources (priority): `GOOGLE_APPLICATION_CREDENTIALS` (service_account RS256 JWT exchange `:102-139`, or authorized_user refresh `:141-153`) → `~/.config/gcloud/application_default_credentials.json` → GCE metadata server (`:155-172`). - - token cache + 60s refresh skew (`:56-58`) + inflight-promise dedup (`:54,230-245`). -- **URL templates** (`google-vertex.ts:38-51,79-81`): ADC → `https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/google/models/{model}:streamGenerateContent?alt=sse`; `location==="global"` → `aiplatform.googleapis.com`; API-key path → `x-goog-api-key`. -- **Body:** unchanged — Vertex reuses the same Gemini params (`messagesToGeminiFormat`); opencodex `google.ts:16-67` is line-for-line equivalent to jawcode `google-shared.ts:convertMessages`. **No new wire.** - -## opencodex fit - -- **Config** (`src/types.ts` `OcxProviderConfig`): add `googleMode?: "ai-studio" | "vertex"`, `project?`, `location?`. -- **Adapter** (`src/adapters/google.ts createGoogleAdapter`): branch `buildRequest` on `googleMode` (default `"ai-studio"` → backward compatible). Vertex branch builds the project/location URL + resolves the ADC Bearer (or `x-goog-api-key`). `parseStream` **unchanged** (mode-agnostic SSE, `google.ts:120-184`). -- **Dispatch** (`src/server.ts:186 resolveAdapter`): **no new case** — stays `google`. -- **Models:** seed ~13 Vertex Gemini ids in the registry. -- **NEW file:** `src/lib/gcp-adc.ts` (the ported ADC resolver). - -## Sub-steps (this PABCD pass) - -1. **A:** port `google-auth.ts` → `src/lib/gcp-adc.ts` (JWT RS256 sign, file/env/metadata sources, cache+skew+dedup). Unit-test JWT signature + cache refresh (mocked token endpoint + time). -2. **B:** add `googleMode`/`project`/`location` to config; extend `createGoogleAdapter` with the vertex branch (URL + auth); leave `parseStream` + `resolveAdapter` untouched. -3. **B:** seed the 13 models + a config example. -4. **C:** mocked integration (ADC → token → buildRequest → parseStream → done) + API-key fallback + location→host mapping; existing AI-Studio google tests still green; `tsc`/`bun test`. - -## Risks - -| Risk | Mitigation | -|------|------------| -| JWT RS256 signing bug → 401 | use WebCrypto; unit-test signature shape vs a known sample | -| token refresh race | inflight-promise dedup (jawcode `:230-245`); test parallel callers | -| location→host mismatch → 404 | hardcode the jawcode mapping; test global + regional | -| AI-Studio regression | default `googleMode:"ai-studio"`; assert existing configs unchanged | - -## Verify (minimal proof) - -Stream one prompt with ADC; compare SSE parts to a jawcode golden (`02:216`). API-key path also streams. - -## Depends-on / enables - -- **Depends-on:** none (ships first). -- **Enables:** the `googleMode` hook reused by Phase 20 (antigravity). diff --git a/devlog/_fin/140_remaining-provider-ports/11_phase1_vertex-http-hardening.md b/devlog/_fin/140_remaining-provider-ports/11_phase1_vertex-http-hardening.md deleted file mode 100644 index 6b161889f..000000000 --- a/devlog/_fin/140_remaining-provider-ports/11_phase1_vertex-http-hardening.md +++ /dev/null @@ -1,262 +0,0 @@ -# 140.11 — Phase 1b: google-vertex HTTP hardening (retry/timeout + error classification) - -> Brings Vertex (the `google` adapter `googleMode:"vertex"` branch) to the **Kiro/Cursor -> stabilization bar** for the HTTP layer. Copy-paste-ready. SOT = opencodex's own Kiro pattern -> (`kiro-retry.ts`, `kiro-errors.ts`); external cross-check = `router-for-me/CLIProxyAPI` -> (`internal/runtime/executor/gemini_vertex_executor.go`). See `05_reference-repos.md`. - ---- - -## Why - -Today the `google` adapter does **not** implement `fetchResponse`, so `server.ts:677` falls back to -`fetchWithHeaderTimeout` — a single attempt with a connect-header timeout and **no retry, no -backoff, no actionable error classification, no secret redaction**. Kiro routes its upstream call -through `fetchKiroWithRetry` (`src/adapters/kiro-retry.ts`) and normalizes errors via -`safeKiroHttpErrorMessage` (`src/adapters/kiro-errors.ts`). Vertex must match that bar. - -CLIProxyAPI confirms the gap is real on the upstream too: its Vertex executor does **no** retry and -returns the raw body as `statusErr{code,msg}` on any non-2xx. We do better by classifying + retrying -transient codes, mirroring Kiro. - -## Vertex error contract (cross-checked, paste-ready) - -Vertex returns the standard Google API error envelope: - -```json -{"error":{"code":429,"message":"…","status":"RESOURCE_EXHAUSTED","details":[…]}} -``` - -| HTTP | `status` enum | Meaning | Classify as | -|------|---------------|---------|-------------| -| 429 | `RESOURCE_EXHAUSTED` | rate limit / quota | `Vertex AI rate limit exceeded` (retryable) | -| 401 | `UNAUTHENTICATED` | missing/expired token | `Vertex AI authentication failed` (non-retryable) | -| 403 | `PERMISSION_DENIED` | creds ok, no access / API disabled / wrong project | `Vertex AI access denied` (non-retryable) | -| 400 | `INVALID_ARGUMENT` | bad model/region/schema | `Vertex AI invalid request` (non-retryable) | -| 404 | `NOT_FOUND` | model not found | `Vertex AI invalid request` (non-retryable) | -| 503 | `UNAVAILABLE` | overload / transient | `Vertex AI server overloaded` (retryable) | -| 500 | `INTERNAL` | server error | `Vertex AI upstream error` (retryable) | - -> Retryable set mirrors Kiro: `429, 500, 502, 503, 504`. Quota-exhausted (a distinct -> `RESOURCE_EXHAUSTED` flavor with `QuotaFailure`/no `retryDelay`) is NOT retried — same split Kiro -> makes between "rate limit" and "quota exhausted". - ---- - -## Part 1 — Plain explanation - -When a Vertex call hits a temporary failure (rate limit, server overload, dropped connection), we -now retry a few times with growing, jittered delays instead of failing immediately — exactly like -Kiro. When it fails for real, the caller gets a short, classified message ("Vertex AI rate limit -exceeded: …") with file paths and secrets stripped out, instead of a raw Google error blob. - -## Part 2 — Diff-level plan - -### NEW `src/adapters/google-errors.ts` - -Mirror of `kiro-errors.ts`, specialized to the Google/Vertex envelope. Reuses the shared -`redactSecretString` and the same absolute-path redaction. - -```ts -import { redactSecretString } from "../redact"; - -const ABSOLUTE_PATH_PATTERN = /(?:\/Users\/[^ "';,]+|\/home\/[^ "';,]+|[A-Za-z]:\\Users\\[^ "';,]+)/g; - -function sanitizeGoogleErrorText(value: string): string { - return redactSecretString(value).replace(ABSOLUTE_PATH_PATTERN, "[REDACTED_PATH]"); -} - -function safeString(value: unknown): string | undefined { - return typeof value === "string" && value.trim().length > 0 ? value.trim() : undefined; -} - -/** Pull the human detail out of the Google API error envelope `{error:{message,status,code}}`. */ -function googleErrorDetail(payloadText: string): { message?: string; status?: string } { - const trimmed = payloadText.trim(); - if (!trimmed || (!trimmed.startsWith("{") && !trimmed.startsWith("["))) { - return { message: trimmed || undefined }; - } - try { - const parsed = JSON.parse(trimmed) as { error?: { message?: unknown; status?: unknown } }; - const err = parsed.error; - return { message: safeString(err?.message), status: safeString(err?.status) }; - } catch { - return {}; - } -} - -function classifyGoogle(status: number | undefined, enumStatus: string | undefined, text: string): string { - const lower = `${enumStatus ?? ""} ${text}`.toLowerCase(); - const quotaExhausted = - lower.includes("quotafailure") || - lower.includes("quota exceeded") || - lower.includes("exceeded your current quota") || - lower.includes("billing"); - if (enumStatus === "RESOURCE_EXHAUSTED" && quotaExhausted) return "Vertex AI quota exhausted"; - if (status === 429 || enumStatus === "RESOURCE_EXHAUSTED" || lower.includes("rate limit")) { - return "Vertex AI rate limit exceeded"; - } - if (status === 401 || enumStatus === "UNAUTHENTICATED" || lower.includes("unauthenticated") || lower.includes("invalid authentication") || lower.includes("expired")) { - return "Vertex AI authentication failed"; - } - if (status === 403 || enumStatus === "PERMISSION_DENIED" || lower.includes("permission denied") || lower.includes("access denied")) { - return "Vertex AI access denied"; - } - if (status === 503 || enumStatus === "UNAVAILABLE" || lower.includes("overloaded") || lower.includes("unavailable")) { - return "Vertex AI server overloaded"; - } - if (status === 400 || status === 404 || enumStatus === "INVALID_ARGUMENT" || enumStatus === "NOT_FOUND" || lower.includes("invalid") || lower.includes("not found") || lower.includes("malformed")) { - return "Vertex AI invalid request"; - } - return "Vertex AI upstream error"; -} - -export function safeVertexHttpErrorMessage(status: number, payloadText: string): string { - const { message, status: enumStatus } = googleErrorDetail(payloadText); - const prefix = classifyGoogle(status, enumStatus, [message, enumStatus].filter(Boolean).join(" ")); - const detail = message ? sanitizeGoogleErrorText(message).slice(0, 500) : `HTTP ${status}`; - return `${prefix}: ${detail}`; -} - -/** Vertex's retryable HTTP set (mirrors Kiro). Quota-exhausted is classified above and not retried. */ -export function retryableVertexStatus(status: number): boolean { - return status === 429 || status === 500 || status === 502 || status === 503 || status === 504; -} -``` - -### NEW `src/adapters/google-http.ts` - -A near-verbatim copy of `kiro-retry.ts`, swapping the error normalizer + retryable predicate. Keeps -the abort-aware sleep, per-attempt timeout (`AbortSignal.any([parent, AbortSignal.timeout])`), -`Retry-After` honoring, and exponential backoff with 0.8–1.2× jitter. - -```ts -import type { AdapterFetchContext, AdapterRequest } from "./base"; -import { retryableVertexStatus, safeVertexHttpErrorMessage } from "./google-errors"; - -const VERTEX_RETRY_ATTEMPTS = 3; -const VERTEX_RETRY_BASE_MS = 250; -const VERTEX_RETRY_MAX_MS = 2_000; - -function retryAfterMs(headers: Headers): number | undefined { - const raw = headers.get("retry-after")?.trim(); - if (!raw) return undefined; - const seconds = Number(raw); - if (Number.isFinite(seconds)) return Math.max(0, seconds * 1000); - const dateMs = Date.parse(raw); - if (!Number.isFinite(dateMs)) return undefined; - return Math.max(0, dateMs - Date.now()); -} - -function retryDelayMs(attempt: number, headers?: Headers): number { - const retryAfter = headers ? retryAfterMs(headers) : undefined; - if (retryAfter !== undefined) return Math.min(retryAfter, VERTEX_RETRY_MAX_MS); - const exp = Math.min(VERTEX_RETRY_BASE_MS * (2 ** attempt), VERTEX_RETRY_MAX_MS); - return Math.floor(exp * (0.8 + Math.random() * 0.4)); -} - -function abortError(signal?: AbortSignal): unknown { - return signal?.reason ?? new DOMException("The operation was aborted", "AbortError"); -} - -async function sleepWithAbort(ms: number, signal?: AbortSignal): Promise { - if (ms <= 0) return; - if (signal?.aborted) throw abortError(signal); - await new Promise((resolve, reject) => { - let timer: ReturnType; - const cleanup = () => { clearTimeout(timer); signal?.removeEventListener("abort", onAbort); }; - const onAbort = () => { cleanup(); reject(abortError(signal)); }; - timer = setTimeout(() => { cleanup(); resolve(); }, ms); - signal?.addEventListener("abort", onAbort, { once: true }); - }); -} - -function signalWithAttemptTimeout(parent: AbortSignal | undefined, timeoutMs: number): AbortSignal { - const timeout = AbortSignal.timeout(timeoutMs); - return parent ? AbortSignal.any([parent, timeout]) : timeout; -} - -async function normalizeFinalVertexError(res: Response): Promise { - if (res.ok) return res; - const payloadText = await res.clone().text().catch(() => ""); - const headers = new Headers(res.headers); - headers.delete("content-encoding"); - headers.delete("content-length"); - return new Response(safeVertexHttpErrorMessage(res.status, payloadText), { - status: res.status, statusText: res.statusText, headers, - }); -} - -export async function fetchVertexWithRetry(request: AdapterRequest, ctx: AdapterFetchContext = {}): Promise { - const timeoutMs = ctx.timeoutMs ?? 100_000; - let lastError: unknown; - for (let attempt = 0; attempt < VERTEX_RETRY_ATTEMPTS; attempt++) { - if (ctx.abortSignal?.aborted) throw abortError(ctx.abortSignal); - try { - const res = await fetch(request.url, { - method: request.method, headers: request.headers, body: request.body, - signal: signalWithAttemptTimeout(ctx.abortSignal, timeoutMs), - }); - if (!retryableVertexStatus(res.status) || attempt === VERTEX_RETRY_ATTEMPTS - 1) return normalizeFinalVertexError(res); - await res.body?.cancel().catch(() => {}); - await sleepWithAbort(retryDelayMs(attempt, res.headers), ctx.abortSignal); - } catch (err) { - if (ctx.abortSignal?.aborted) throw err; - lastError = err; - if (attempt === VERTEX_RETRY_ATTEMPTS - 1) throw err; - await sleepWithAbort(retryDelayMs(attempt), ctx.abortSignal); - } - } - throw lastError ?? new Error("Vertex fetch failed"); -} -``` - -### MODIFY `src/adapters/google.ts` — add `fetchResponse` (vertex-only) - -The retry wrapper must apply **only** to the Vertex branch; AI-Studio Gemini keeps the default path -(it has its own behavior and tests). Gate on `provider.googleMode`. - -```ts -// add import at top -import { fetchVertexWithRetry } from "./google-http"; - -// inside createGoogleAdapter(provider), add alongside buildRequest/parseStream: - fetchResponse: provider.googleMode === "vertex" - ? (request: AdapterRequest, ctx?: AdapterFetchContext): Promise => fetchVertexWithRetry(request, ctx) - : undefined, -``` - -> `AdapterFetchContext`/`AdapterRequest` are already imported by the adapter? If not, extend the -> existing `./base` import. `server.ts:677` already prefers `adapter.fetchResponse` when present and -> falls through to `fetchWithHeaderTimeout` when it is `undefined`, so AI-Studio is untouched. - -## Tests (NEW `tests/google-vertex-http.test.ts`) - -1. `fetchVertexWithRetry` retries a 503 then succeeds (mock fetch, count attempts). -2. retries a thrown network error then succeeds. -3. does NOT retry a 400/401/403 (single attempt) and returns a `safeVertexHttpErrorMessage` body. -4. honors `Retry-After` (seconds + HTTP-date), capped at `VERTEX_RETRY_MAX_MS`. -5. aborts promptly when `ctx.abortSignal` fires mid-backoff. -6. `safeVertexHttpErrorMessage` classifies each enum row above and redacts an embedded - `Authorization: Bearer …` + an absolute `/Users/...` path. -7. AI-Studio google (`googleMode` unset) still uses the default fetch path (adapter `fetchResponse` - is `undefined`). - -## Verify - -`bun x tsc --noEmit` clean + `bun test ./tests/` green (no regression in existing `google-adapter` -/ `adapter-usage` tests). Minimal proof: a mocked 503→200 retry and a classified-redacted 403. - -## Depends-on / enables - -- Depends-on: Phase 10 (vertex branch exists). -- Enables: Phase 20 antigravity reuses `fetchVertexWithRetry` + `google-errors` for its CCA endpoint. - ---- - -## ✅ Implemented (commit `b4a772b`) - -- NEW `src/adapters/google-errors.ts` — `safeGoogleHttpErrorMessage(label,…)` + `safeVertexHttpErrorMessage` / `safeAntigravityHttpErrorMessage` + `retryableGoogleStatus`. (Generalized with a `label` param so antigravity reuses it.) -- NEW `src/adapters/google-http.ts` — `fetchGoogleWithRetry(label,…)` + `fetchVertexWithRetry` / `fetchAntigravityWithRetry`. -- `src/adapters/google.ts` — `fetchResponse` spread for `googleMode` vertex|cloud-code-assist; ai-studio stays undefined. -- Tests: `tests/google-vertex-http.test.ts` (9). Suite 1015/0, tsc clean. diff --git a/devlog/_fin/140_remaining-provider-ports/12_phase1_vertex-stream-usage-hardening.md b/devlog/_fin/140_remaining-provider-ports/12_phase1_vertex-stream-usage-hardening.md deleted file mode 100644 index cbaf7c8ff..000000000 --- a/devlog/_fin/140_remaining-provider-ports/12_phase1_vertex-stream-usage-hardening.md +++ /dev/null @@ -1,202 +0,0 @@ -# 140.12 — Phase 1c: google-vertex stream + usage + ADC-refresh hardening - -> Second hardening slice for Vertex: fail-closed stream truncation, correct usage tagging, and a -> hardened ADC token exchange (timeout + bounded retry). Copy-paste-ready. SOT = opencodex Kiro -> pattern (`kiro-truncation.ts`, `usage-log.ts`) + the existing `gcp-adc.ts`; external cross-check = -> CLIProxyAPI `helps/usage_helpers.go`. See `05_reference-repos.md`. - ---- - -## What this covers - -1. **Fail-closed truncation** — if a Vertex stream ends with `finishReason: MAX_TOKENS` (or the - stream is cut) while a tool call is mid-emit, surface an error instead of silently yielding a - half-built `functionCall`. Mirrors `kiro-truncation.ts`. -2. **Authoritative usage** — Vertex DOES return `usageMetadata` on the terminal chunk - (cross-checked: `promptTokenCount / candidatesTokenCount / thoughtsTokenCount / totalTokenCount / - cachedContentTokenCount`). So unlike Kiro/Cursor, Vertex usage is NOT estimated — keep it - `reported`, and make sure `usage-log.ts` does NOT force-tag `google-vertex` as estimated. -3. **ADC refresh hardening** — `gcp-adc.ts` `postForToken` is a single `fetch` with no timeout and - no retry. Add a per-attempt timeout + bounded retry on transient failures, matching the Cursor - `refreshCursorToken` hardening. - ---- - -## Part 1 — Plain explanation - -If Vertex stops a response early in the middle of a tool call, we now report that clearly instead -of handing back a broken half-call. Token counts that Vertex actually reports stay accurate (not -marked as guesses). And fetching the Google access token now survives a flaky network: it times out -per try and retries transient errors a couple of times before giving up. - -## Part 2 — Diff-level plan - -### A) Fail-closed truncation in `parseStream` (`src/adapters/google.ts`) - -Current loop (`google.ts:~146-180`) emits `tool_call_start`/`delta`/`end` per `functionCall` and -accumulates `usageMetadata`, then emits one terminal `done`. It never inspects `finishReason`. Add: - -- Track the last seen `finishReason` from `candidates[0].finishReason`. -- A Vertex `functionCall` arrives as a complete `{name, args}` object, so mid-call byte truncation - is not the failure mode (unlike Cursor's streamed text). The real fail-closed trigger is a - terminal `finishReason` of `MAX_TOKENS` (or `OTHER`/`SAFETY` cutting off a started-but-unfinished - turn). Treat `MAX_TOKENS` while at least one tool call was started this turn as truncation. - -NEW helper file `src/adapters/google-truncation.ts` (mirror of `kiro-truncation.ts`): - -```ts -import { redactSecretString } from "../redact"; - -const TRUNCATION_REASONS = new Set(["MAX_TOKENS", "MALFORMED_FUNCTION_CALL"]); - -/** Vertex finishReason values that mean the turn was cut off, not cleanly stopped. */ -export function isVertexTruncationReason(finishReason: string | undefined): boolean { - return finishReason !== undefined && TRUNCATION_REASONS.has(finishReason); -} - -export function vertexTruncationErrorMessage(reason?: string): string { - const suffix = reason ? ` (${redactSecretString(reason).slice(0, 160)})` : ""; - return `Vertex AI response truncated upstream before the turn completed${suffix}`; -} -``` - -Patch in `parseStream` (sketch): - -```ts - let toolCallsStarted = 0; - let lastFinishReason: string | undefined; - // …inside the candidates loop: - lastFinishReason = candidates[0].finishReason ?? lastFinishReason; - // when emitting a functionCall: toolCallsStarted++; - // …after the read loop, BEFORE the terminal done: - if (isVertexTruncationReason(lastFinishReason) && toolCallsStarted > 0) { - yield { type: "error", message: vertexTruncationErrorMessage(lastFinishReason) }; - return; - } -``` - -> Scope guard: only apply this when `provider.googleMode === "vertex"` if we want AI-Studio Gemini -> behavior byte-identical. Recommended: apply to both (a `MALFORMED_FUNCTION_CALL` truncation error -> is correct for AI-Studio too), but if the existing AI-Studio tests assert silent completion on -> `MAX_TOKENS`, gate on `googleMode`. Decide in A-audit by reading `tests/google-adapter.test.ts`. - -### B) Usage tagging stays `reported` (`src/usage-log.ts`) - -Vertex returns authoritative `usageMetadata`, so it must NOT be force-estimated. The current -`isEstimatedUsageProvider` only tags `kiro*`/`cursor`, so `google-vertex` is already correct — but -pin it with a guard + a test so a future "tag all gemini as estimated" change can't regress it. - -- The mapping function is **`usageFromGemini`** (`src/adapters/google.ts:115`), NOT `mapGeminiUsage` - (no such symbol exists). It already sets `inputTokens`/`outputTokens`/`reasoningOutputTokens`/ - `cachedInputTokens` from `usageMetadata` and does NOT set `estimated: true` — both correct. - Field mapping it already does: - -| Vertex field | OcxUsage | Status | -|---|---|---| -| `promptTokenCount` | `inputTokens` | already mapped | -| `candidatesTokenCount` | `outputTokens` | already mapped | -| `thoughtsTokenCount` | `reasoningOutputTokens` | already mapped (conditional) | -| `cachedContentTokenCount` | `cachedInputTokens` | already mapped (conditional) | -| `totalTokenCount` | (derive if 0: input+output+reasoning) | **NEW — not currently read; add it** | - - So the only behavior change here is the `totalTokenCount` derive-if-0 (optional polish, matching - CLIProxyAPI's `parseGeminiFamilyUsageDetail`). The estimated-tagging guard below is the real point. - -- Test: a `google-vertex` request whose stream carries `usageMetadata` ends with - `usageStatus === "reported"` (NOT `estimated`). - -### C) ADC token-exchange hardening (`src/lib/gcp-adc.ts`) - -`postForToken` is a single `fetch` with the caller's signal but no timeout and no retry. Harden it -the way Cursor's `refreshCursorToken` was hardened (timeout + bounded transient retry), keeping the -"never log token/key/refresh" guarantee. - -```ts -const TOKEN_TIMEOUT_MS = 15_000; -const TOKEN_ATTEMPTS = 3; -const TOKEN_RETRY_BASE_MS = 300; - -function tokenRetryDelayMs(attempt: number): number { - const exp = TOKEN_RETRY_BASE_MS * 2 ** attempt; - return Math.floor(exp * (0.8 + Math.random() * 0.4)); -} -function isRetryableTokenStatus(s: number): boolean { - return s === 429 || s === 500 || s === 502 || s === 503 || s === 504; -} -function tokenTimeoutSignal(parent: AbortSignal | undefined): AbortSignal { - const timeout = AbortSignal.timeout(TOKEN_TIMEOUT_MS); - return parent ? AbortSignal.any([parent, timeout]) : timeout; -} - -async function postForToken(body: URLSearchParams, signal: AbortSignal | undefined, fetchImpl: FetchImpl): Promise { - let lastError: unknown; - for (let attempt = 0; attempt < TOKEN_ATTEMPTS; attempt++) { - if (signal?.aborted) throw signal.reason ?? new Error("token exchange aborted"); - let response: Response; - try { - response = await fetchImpl(OAUTH_TOKEN_URL, { - method: "POST", - headers: { "Content-Type": "application/x-www-form-urlencoded" }, - body: body.toString(), - signal: tokenTimeoutSignal(signal), - }); - } catch (err) { - if (signal?.aborted) throw err; - lastError = err; - if (attempt === TOKEN_ATTEMPTS - 1) break; - await new Promise(r => setTimeout(r, tokenRetryDelayMs(attempt))); - continue; - } - if (response.ok) return (await response.json()) as TokenResponse; - // Do NOT echo the body for auth errors (it can include grant details); use status only. - if (!isRetryableTokenStatus(response.status) || attempt === TOKEN_ATTEMPTS - 1) { - throw new Error(`Google OAuth token exchange failed (${response.status})`); - } - lastError = new Error(`Google OAuth token exchange failed (${response.status})`); - await response.body?.cancel().catch(() => {}); - await new Promise(r => setTimeout(r, tokenRetryDelayMs(attempt))); - } - throw lastError instanceof Error ? lastError : new Error("Google OAuth token exchange failed"); -} -``` - -> Behavior change worth flagging in A-audit: the old `postForToken` appended the raw response body -> to the error message (`…failed (${status}): ${detail}`). The detail can leak grant/account hints, -> so the hardened version drops it (status only). If a test asserts on the old `: detail` suffix, -> update it. - -The metadata-server fetch (`fetchMetadataToken`) already has a 2s timeout and swallows errors — -leave it. The in-flight dedup + cache + skew in `getVertexAccessToken` are already correct; do not -touch them. - -## Tests - -- `tests/google-vertex-stream.test.ts`: a stream ending `finishReason:MAX_TOKENS` after a - `tool_call_start` yields a terminal `error` (truncation), and a clean `finishReason:STOP` stream - yields `done` with `reported` usage carrying the mapped token fields. -- `tests/gcp-adc.test.ts` (extend): `postForToken` retries a 503 then succeeds; retries a thrown - network error then succeeds; fails fast on 400/401 (single attempt); token value never appears in - a thrown error message. -- `tests/usage-*.test.ts` (extend): `google-vertex` usage with `usageMetadata` → `reported`. - -## Verify - -`bun x tsc --noEmit` clean + `bun test ./tests/` green. Minimal proof: MAX_TOKENS-mid-tool → -truncation error; 503→200 token-exchange retry. - -## Depends-on / enables - -- Depends-on: Phase 10 + 11. -- Enables: Phase 20 antigravity reuses the truncation helper and inherits the hardened ADC/OAuth - refresh shape for its own token endpoint. - ---- - -## ✅ Implemented (commit `c225642`) - -- NEW `src/adapters/google-truncation.ts` — `isVertexTruncationReason` + `vertexTruncationErrorMessage`. -- `src/adapters/google.ts` parseStream — tracks `lastFinishReason` + `toolCallsStarted`; fail-closed - error on MAX_TOKENS/MALFORMED_FUNCTION_CALL mid tool call (vertex + cloud-code-assist). -- `src/lib/gcp-adc.ts` `postForToken` — per-attempt 15s timeout + bounded retry (429/5xx/network), - status-only error (no body leak). Usage stays `reported` (no estimated tag for google-vertex). -- Tests: `tests/google-vertex-stream.test.ts` + retry cases in `tests/gcp-adc.test.ts`. Suite 1023/0. diff --git a/devlog/_fin/140_remaining-provider-ports/20_phase2_google-antigravity.md b/devlog/_fin/140_remaining-provider-ports/20_phase2_google-antigravity.md deleted file mode 100644 index dfa7588e8..000000000 --- a/devlog/_fin/140_remaining-provider-ports/20_phase2_google-antigravity.md +++ /dev/null @@ -1,61 +0,0 @@ -# 140.20 — Phase 2: google-antigravity (extend `google` + OAuth, MEDIUM) - -> One PABCD pass. Reuses Phase 10's `googleMode` hook; adds OAuth + the Cloud-Code-Assist (CCA) -> envelope. Grounded in jawcode (cites are jawcode paths). -> -> External cross-check: `router-for-me/CLIProxyAPI` antigravity auth/executor/translator + the -> thoughtSignature reasoning-replay quirk — see `05_reference-repos.md`. - ---- - -## Goal - -Stream Gemini/Claude through Google **Antigravity** (Cloud Code Assist) — an OAuth-authed endpoint that -wraps the flat Gemini body in a CCA envelope. Reuses the `google` adapter (mode = `cloud-code-assist`). - -## What we port (jawcode) - -- **OAuth + project onboard** (`utils/oauth/google-antigravity.ts:116-207`): - - `loginAntigravity` (`:160-172`) → `runGoogleOAuthLogin` with `discoverProject`: call `/v1internal:loadCodeAssist`; if no project, `/v1internal:onboardUser` retry loop (5×, 2s) → `operation.response.cloudaicompanionProject`. - - refresh via `https://oauth2.googleapis.com/token` (`:177-207`). - - env secrets `GOOGLE_ANTIGRAVITY_CLIENT_ID/SECRET` (`:10-11`); scopes `cloud-platform`/userinfo/cclog (`:15-21`). - - credential JSON `{ refresh, access, expires, projectId }` (`:201-206`). -- **CCA envelope** (`google-gemini-cli.ts:735-856`): wrap the flat Gemini `request` in - `{ project, model, request, requestType:"agent", userAgent:"antigravity", requestId:"agent-{uuid}" }`; - endpoint `${baseUrl}/v1internal:streamGenerateContent?alt=sse` (`:349`). -- **Headers** (`:314-352`): Antigravity User-Agent; `anthropic-beta: interleaved-thinking-2025-05-14` for Claude+reasoning (`:101-102,329`). -- **Quirks** (`:814-842`): drop `maxOutputTokens` for non-Claude; inject system instruction for Claude+Gemini-3; tool mode `VALIDATED` for Claude forced choice; `parametersJsonSchema → parameters` normalize (`:729`). sessionId = hash of first user text (`:686-714`). - -## opencodex fit - -- **OAuth** (`src/oauth/`): NEW `src/oauth/google-antigravity.ts` (login + refresh + discoverProject/onboard). Extend `OAuthCredentials` (`src/oauth/types.ts`) with `projectId?`. Register in `OAUTH_PROVIDERS` (`src/oauth/index.ts:36-61`). -- **projectId wiring (audit fix):** the CCA envelope needs `project` from the stored credential, but opencodex's server injects only the bare access token into `apiKey` (`server.ts:299-301`). So the adapter must read the credential's `projectId` directly (`getCredential("google-antigravity").projectId`) — store `{token, projectId}` like jawcode `parseGeminiCliCredentials` (`google-gemini-cli.ts:143-159`), and use it in both the envelope and the refresh closure. -- **Adapter** (`src/adapters/google.ts`): extend the Phase-10 `googleMode` branch with `cloud-code-assist` → wrap body in CCA envelope + antigravity headers. -- **⚠️ `parseStream` IS changed for antigravity (audit fix):** antigravity nests output under `chunk.response.candidates` + thinking parts (jawcode `google-gemini-cli.ts:399-402`), but opencodex's current parser reads **top-level** `chunk.candidates` (`google.ts:155`). Phase 20 adds a **mode-aware** `parseStream` that unwraps `response` and handles thinking deltas. (Phase 10 vertex is top-level, so it left `parseStream` untouched — this change is antigravity-only.) -- **Registry:** `adapter:"google"`, `authKind:"oauth"`, `oauthId:"google-antigravity"`, baseUrl `daily-cloudcode-pa.googleapis.com` **with prod fallback `cloudcode-pa.googleapis.com`** (`google-gemini-cli.ts:69-71`), ~15 bundled models. - -## Sub-steps (this PABCD pass) - -1. **A:** port `src/oauth/google-antigravity.ts` (login/refresh/onboard); extend creds with `projectId`; register in `OAUTH_PROVIDERS`. Unit-test mocked `loadCodeAssist`/`onboardUser`. -2. **B:** extend `createGoogleAdapter` with the CCA-envelope branch + headers + a **mode-aware `parseStream`** (unwrap `response.candidates` + thinking deltas) + the quirks table (maxTokens drop, system-instr inject, tool-mode map, schema normalize, sessionId hash). -3. **B:** registry entry + 15 models. -4. **C:** `ocx login google-antigravity` → creds with projectId; envelope-builder asserts (`project`, `requestType:"agent"`); one text stream + one tool call (schema normalization); `tsc`/`bun test`. - -## Risks - -| Risk | Mitigation | -|------|------------| -| OAuth client id/secret distribution | env vars, no secrets in repo (`02:204`); sandbox creds for test | -| onboard timeout / quota | retry loop (5×,2s) + clear error | -| CCA envelope rejected upstream | validate envelope vs jawcode golden | -| Claude thinking-beta header missing | gate on `model.includes("claude") && reasoning` | -| tool schema normalize bug | port `normalizeSchemaForCCA` exactly; test complex tool sets | - -## Verify (minimal proof) - -`ocx login google-antigravity`; one tool-call + thinking-model stream (`02:217`). - -## Depends-on / enables - -- **Depends-on:** Phase 10 (the `googleMode` hook + the `google` adapter branch structure). -- **Enables:** the `OAUTH_PROVIDERS` registration pattern reused by Phase 50 (cursor). diff --git a/devlog/_fin/140_remaining-provider-ports/21_phase2_antigravity-oauth-cca.md b/devlog/_fin/140_remaining-provider-ports/21_phase2_antigravity-oauth-cca.md deleted file mode 100644 index 42b88a31b..000000000 --- a/devlog/_fin/140_remaining-provider-ports/21_phase2_antigravity-oauth-cca.md +++ /dev/null @@ -1,225 +0,0 @@ -# 140.21 — Phase 2b: google-antigravity OAuth + projectId + CCA envelope (hardened wire) - -> Builds the Antigravity (Cloud Code Assist) wire on top of the Phase 10/20 `googleMode` hook, and -> bakes in the same HTTP hardening as Vertex (Phase 11) from day one. Copy-paste-ready. -> External SOT cross-check = `router-for-me/CLIProxyAPI` (`internal/auth/antigravity/*`, -> `internal/runtime/executor/antigravity_executor.go`, `internal/translator/antigravity/gemini/*`). -> See `05_reference-repos.md`. The reasoning-replay quirk is split into `22_…`. - ---- - -## Corrections to the existing 20_ plan (from CLIProxyAPI cross-check) - -The original `20_phase2_google-antigravity.md` was grounded in jawcode. The CLIProxyAPI audit -surfaces details to pin precisely: - -1. **Default streaming host is `daily-cloudcode-pa.googleapis.com`**, with `cloudcode-pa.googleapis.com` - as the prod fallback. `loadCodeAssist` uses the **prod** host; `onboardUser` uses the **daily** - host. Make the base configurable, daily-first, prod as fallback. -2. The envelope field **`userAgent` is the literal string `"antigravity"`** — a body field, NOT the - HTTP `User-Agent` header (those are separate). -3. `request.safetySettings` is attached then **deleted** before send — net: send none. -4. **`anthropic-beta: interleaved-thinking-…` is NOT part of the CCA wire.** Do not add it (the - original 20_ plan mentioned it; drop that). Claude-on-Antigravity thinking is handled in-body via - signature sanitization (Phase 22), not a header. -5. Response chunks nest under **`response.candidates`** (confirmed); our parser must unwrap `response`. -6. `requestType` defaults to `"agent"`; `requestId` = `"agent-" + uuid`. - -## Constants (paste-ready, from CLIProxyAPI `internal/auth/antigravity/constants.go`) - -``` -ClientID 1071006060591-tmhssin2h21lcre235vtolojh4g403ep.apps.googleusercontent.com -ClientSecret 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 -CallbackPort 51121 -TokenEndpoint https://oauth2.googleapis.com/token -AuthEndpoint https://accounts.google.com/o/oauth2/v2/auth -APIEndpoint https://cloudcode-pa.googleapis.com (prod; loadCodeAssist) -DailyAPIEndpoint https://daily-cloudcode-pa.googleapis.com (default stream + onboardUser) -APIVersion v1internal -Scopes cloud-platform, userinfo.email, userinfo.profile, cclog, experimentsandconfigs -``` - -> The client_id/secret above are public OAuth client identifiers embedded in the Antigravity -> desktop client (not user secrets); they are how CLIProxyAPI authenticates the OAuth flow. Treat -> them as config defaults overridable by env (`GOOGLE_ANTIGRAVITY_CLIENT_ID/SECRET`). - ---- - -## Part 1 — Plain explanation - -Antigravity is Google's Cloud Code Assist endpoint. To use it we log in with Google OAuth, discover -the user's cloud project (auto-onboarding if needed), then wrap each Gemini request in a small -"envelope" with the project id and a request id before streaming. We build it with the same retry, -timeout, and error-classification armor Vertex got, so it's production-grade from the start. - -## Part 2 — Diff-level plan - -### A) Config + registry - -`src/types.ts` — extend the `google` mode union: - -```ts - googleMode?: "ai-studio" | "vertex" | "cloud-code-assist"; -``` - -`src/providers/registry.ts` — add the provider entry (adapter stays `google`, no new `resolveAdapter` case): - -```ts - { id: "google-antigravity", label: "Google Antigravity", adapter: "google", - baseUrl: "https://daily-cloudcode-pa.googleapis.com", authKind: "oauth", - oauthId: "antigravity", // OAUTH_PROVIDERS key (derive.ts: oauthId ?? id) - dashboardUrl: "https://antigravity.google", defaultModel: "gemini-3-pro", - googleMode: "cloud-code-assist", jawcodeBundle: "google", - extraMetadataAliases: ["antigravity", "gemini-antigravity"] }, -``` - -> `oauthId` matters: every other `authKind:"oauth"` registry entry sets it, and `derive.ts:147` -> resolves the `OAUTH_PROVIDERS` registration key as `entry.oauthId ?? entry.id`. With -> `oauthId:"antigravity"` the OAuth provider in `src/oauth/index.ts` must be registered under the key -> `"antigravity"`. (Omit `oauthId` only if you instead key `OAUTH_PROVIDERS` by `"google-antigravity"`.) - -### B) OAuth provider — NEW `src/oauth/google-antigravity.ts` - -Register in `OAUTH_PROVIDERS` (`src/oauth/index.ts`) under key `"antigravity"` (matching `oauthId` -above). Implements login + refresh + project discovery. **`OAuthCredentials` (`src/oauth/types.ts`) -gains an optional `projectId?: string`** — note this is the OAuth credential type, NOT -`OcxProviderConfig` (which already uses `project`/`location` for Vertex and needs no change here). -The server injects only the bare access token into `apiKey`, so the adapter reads `projectId` from -the stored credential — same audit fix the 20_ plan noted. `OcxProviderConfig`'s allowed-key union -(`types.ts:58`) does NOT need editing for this (no new `OcxProviderConfig` field). - -Project discovery (mirror CLIProxyAPI `FetchProjectID` / `OnboardUser`): - -```ts -// 1. POST https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist -// body: { metadata: { ideType: "ANTIGRAVITY" } } -// extract project from cloudaicompanionProject | projectId | project(.id) -// 2. if no project → onboarding loop on the DAILY host: -// POST https://daily-cloudcode-pa.googleapis.com/v1internal:onboardUser -// body: { tier_id: , metadata: { ide_type:"ANTIGRAVITY", ide_name:"antigravity", ide_version: } } -// poll up to 5 attempts, 2s sleep between, each attempt 30s timeout, -// until 200 && done===true → response.cloudaicompanionProject -``` - -Refresh (mirror CLIProxyAPI `refreshTokenSingleFlight`), hardened like Cursor/Vertex token refresh: - -```ts -// POST https://oauth2.googleapis.com/token -// form: client_id, client_secret, grant_type=refresh_token, refresh_token -// proactive refresh when expiry < now + 50min skew -// single-flight per refresh token (coalesce concurrent refreshes) -// per-attempt 15s timeout + bounded retry (3x) on 429/5xx/network; honor Retry-After -// NEVER log token/refresh/secret -``` - -### C) Adapter — extend the `googleMode` branch in `src/adapters/google.ts` - -Add a `cloud-code-assist` arm next to `vertex`. It builds the daily-host CCA URL, wraps the flat -Gemini body in the envelope, sets the whitelist headers, and resolves the project from the stored -credential. - -```ts -if (provider.googleMode === "cloud-code-assist") { - const base = provider.baseUrl || "https://daily-cloudcode-pa.googleapis.com"; - const method = parsed.stream ? "streamGenerateContent" : "generateContent"; - const streamParam = parsed.stream ? "?alt=sse" : ""; - const url = `${base}/v1internal:${method}${streamParam}`; - - const project = resolveAntigravityProjectId(provider); // from credential.projectId - if (!project) throw new Error("Antigravity requires a discovered Google Cloud project id (re-run login)."); - - const envelope = { - model: parsed.modelId, - userAgent: "antigravity", // literal body field - requestType: "agent", - project, - requestId: `agent-${crypto.randomUUID()}`, - request: { - ...body, // the flat Gemini body (contents/tools/generationConfig…) - sessionId: stableAntigravitySessionId(parsed), // sha256(first user text) → masked int, "-" prefix - }, - }; - // body.systemInstruction already correct; ensure request.model is NOT present; send NO safetySettings. - delete (envelope.request as Record).model; - delete (envelope.request as Record).safetySettings; - - const token = provider.apiKey; // server injects the OAuth access token here - const headers: Record = { - "Content-Type": "application/json", - "User-Agent": ANTIGRAVITY_REQUEST_UA, - Authorization: `Bearer ${token}`, - }; - return { url, method: "POST", headers, body: JSON.stringify(envelope) }; -} -``` - -Helpers (small, in `google.ts` or a new `src/adapters/google-antigravity-wire.ts`): -- `stableAntigravitySessionId(parsed)`: sha256 of first user message text → BigEndian uint64 masked - `0x7FFFFFFFFFFFFFFF`, prefixed `-`; fallback random `-<19 digits>`. -- `resolveAntigravityProjectId(provider)`: reads the stored credential's `projectId`. -- `ANTIGRAVITY_REQUEST_UA`: the Antigravity client UA string (config-overridable). - -### D) parseStream — unwrap `response` - -Antigravity chunks are `{"response":{"candidates":[…],"usageMetadata":…},"traceId":…}`. The existing -google `parseStream` reads top-level `chunk.candidates`. Add a mode-aware unwrap so CCA reads -`chunk.response.candidates` (and `chunk.response.usageMetadata`). Keep AI-Studio/Vertex on the -top-level path. - -```ts -const root = provider.googleMode === "cloud-code-assist" - ? (chunk.response as Record | undefined) ?? chunk - : chunk; -const candidates = root.candidates as …; -const usageMeta = root.usageMetadata as …; -``` - -### E) HTTP hardening — reuse Vertex's wrapper - -`fetchResponse` for `cloud-code-assist` routes through `fetchVertexWithRetry` (Phase 11) — same -retry/timeout/backoff. Error classification reuses `safeVertexHttpErrorMessage`; Antigravity's 429 -RESOURCE_EXHAUSTED + `ErrorInfo.reason` (QUOTA_EXHAUSTED vs RATE_LIMIT_EXCEEDED) maps onto the same -"quota exhausted" vs "rate limit" split. Generalize the prefix from "Vertex AI" to a provider-aware -label, or add a thin `safeAntigravityHttpErrorMessage` that delegates with an `"Antigravity"` prefix. - -```ts - fetchResponse: (provider.googleMode === "vertex" || provider.googleMode === "cloud-code-assist") - ? (request, ctx) => fetchVertexWithRetry(request, ctx) - : undefined, -``` - -## Tests - -- `tests/google-antigravity-wire.test.ts`: envelope shape (project/userAgent/requestType/requestId/ - request.sessionId; no `request.model`, no `safetySettings`); daily-host stream URL `?alt=sse`; - parseStream unwraps `response.candidates` and `response.usageMetadata`. -- `tests/google-antigravity-oauth.test.ts`: loadCodeAssist project extraction across the three key - shapes; onboardUser poll loop (not-done→done); refresh retry on 503 then success; single-flight - coalescing; token never logged. -- stable sessionId is deterministic for the same first user text. - -## Verify - -`bun x tsc --noEmit` clean + `bun test ./tests/` green. Minimal proof: a mocked envelope build + a -`response`-wrapped SSE chunk parsed into text/tool events; a 503→200 refresh retry. - -## Depends-on / enables - -- Depends-on: Phase 10/11/12 (googleMode hook, `fetchVertexWithRetry`, `google-errors`, truncation). -- Enables: Phase 22 (thoughtSignature reasoning-replay) layers onto this wire. - ---- - -## ✅ Implemented (commit `0909f4a`) - -- `src/types.ts` + `src/providers/registry.ts` — `googleMode` union gains `"cloud-code-assist"`; - NEW `google-antigravity` registry entry (adapter `google`, authKind `oauth`, keyed by its id). -- `src/oauth/types.ts` — `OAuthCredentials.projectId?`. `src/oauth/index.ts` — `OAUTH_PROVIDERS["google-antigravity"]` - + `getOAuthCredentialProjectId`. NEW `src/oauth/google-antigravity.ts` (PKCE login, loadCodeAssist/onboardUser - discovery, hardened refresh). -- `src/providers/derive.ts` — `providerConfigSeed` now propagates `googleMode`/`project`/`location` (was dropped). -- `src/server.ts` — injects credential `projectId` → `provider.project` for cloud-code-assist. -- `src/adapters/google.ts` + NEW `src/adapters/google-antigravity-wire.ts` — CCA envelope (daily host, - `userAgent` literal, stable sessionId), `response`-unwrap in parseStream, `fetchAntigravityWithRetry`. -- Tests: `tests/google-antigravity-wire.test.ts` + `tests/google-antigravity-oauth.test.ts`; - `provider-registry-parity` alias snapshot updated. Suite 1034/0, tsc clean. diff --git a/devlog/_fin/140_remaining-provider-ports/22_phase2_antigravity-reasoning-replay-hardening.md b/devlog/_fin/140_remaining-provider-ports/22_phase2_antigravity-reasoning-replay-hardening.md deleted file mode 100644 index 9d4499253..000000000 --- a/devlog/_fin/140_remaining-provider-ports/22_phase2_antigravity-reasoning-replay-hardening.md +++ /dev/null @@ -1,160 +0,0 @@ -# 140.22 — Phase 2c: antigravity thoughtSignature reasoning-replay + stabilization close-out - -> The Antigravity-specific hardening that has no Vertex analogue: **thoughtSignature reasoning -> replay**, plus the truncation/usage stabilization close-out. Copy-paste-ready. External SOT -> cross-check = `router-for-me/CLIProxyAPI` (`internal/runtime/executor/antigravity_reasoning_replay.go`, -> `internal/cache/antigravity_reasoning_replay_cache.go`, `internal/translator/antigravity/gemini/*`). -> See `05_reference-repos.md`. - ---- - -## The quirk (why this phase exists) - -Gemini-3 (and Claude-on-Antigravity) interleaved thinking is **stateless upstream**. Every model -content part in a CCA response carries a `thoughtSignature` (aliases: `thoughtSignature`, -`thought_signature`, `extra_content.google.thought_signature`). To continue a thinking/tool-call -chain across turns, the previous turn's signature **must be echoed back** on the matching model -content part in the next request. Miss it and upstream returns **HTTP 400** (invalid/missing -signature), breaking multi-turn agentic flows — the single biggest Antigravity footgun. - -CLIProxyAPI handles this with a per-session **reasoning-replay cache**: it observes signatures on -the response stream, caches them keyed by `model + session`, and re-injects them into -`request.contents[ci].parts[pi].thoughtSignature` on the next request. It also reconstructs model -function-call parts before the matching `functionResponse`. - -Important applicability split (cross-checked): -- **Gemini/Flash/Agent models → use the replay cache.** -- **Claude-on-Antigravity → does NOT use the cache.** It sanitizes signatures inline instead - (drop empty/incompatible thinking blocks, strip non-model/non-thinking signature fields). - -## Part 1 — Plain explanation - -Antigravity needs us to "remember" a little cryptographic receipt the model attaches to each -thinking step, and hand it back on the next request. If we forget, the server rejects the whole -turn. This phase records those receipts per conversation and replays them, so multi-step tool/think -chains keep working instead of dying with a 400. - -## Part 2 — Diff-level plan - -### A) NEW `src/adapters/google-antigravity-replay.ts` - -A small in-memory cache + observe/apply pair, mirroring CLIProxyAPI's accumulator. Keep it -self-contained and adapter-local (no server wiring needed beyond the google adapter). - -```ts -interface ReplayItem { - type: "thought_signature" | "function_call_part"; - contentIndex: number; - partIndex: number; - thoughtSignature?: string; - // function_call_part only: - name?: string; - callId?: string; - args?: unknown; -} - -interface ReplayEntry { items: ReplayItem[]; expiresAtMs: number; } - -const MIN_SIGNATURE_LEN = 16; // CLIProxyAPI minAntigravityThoughtSignatureReplayLen -const REPLAY_TTL_MS = 60 * 60 * 1000; // 1h -const REPLAY_MAX_ENTRIES = 10_240; // evict batch 128 when exceeded - -const replayCache = new Map(); - -function replayKey(model: string, sessionId: string): string { - return `${model}::session:${sessionId}`; -} - -/** Observe one parsed CCA chunk's `response.candidates[0].content.parts`; record signatures + fn-calls. */ -export function observeAntigravityReplay(model: string, sessionId: string, parts: unknown[]): void { /* … */ } - -/** Re-inject cached signatures + function-call parts into the outgoing request.contents. */ -export function applyAntigravityReplay(model: string, sessionId: string, contents: unknown[]): unknown[] { /* … */ } - -/** Drop the entry when upstream rejects a signature (clear-on-invalid). */ -export function clearAntigravityReplay(model: string, sessionId: string): void { - replayCache.delete(replayKey(model, sessionId)); -} - -/** Gemini/Flash/Agent only — Claude uses inline sanitization, not the cache. */ -export function antigravityUsesReplayCache(model: string): boolean { - return !/claude/i.test(model); -} - -export function __resetAntigravityReplayCache(): void { replayCache.clear(); } -``` - -Behavior to mirror precisely: -- Only record signatures with `length >= MIN_SIGNATURE_LEN`. -- Dedup function-call parts by `(name, callId, args)`; on apply, dedup re-injection by `callId`. -- TTL 1h; when `replayCache.size > REPLAY_MAX_ENTRIES`, evict the 128 oldest by `expiresAtMs`. -- Cache key = `model + ":session:" + sessionId` where `sessionId` is the stable id from Phase 21. - -### B) Wire observe/apply into the google adapter (cloud-code-assist only) - -- **buildRequest (CCA arm):** before serializing the envelope, if `antigravityUsesReplayCache(model)` - call `applyAntigravityReplay(model, sessionId, request.contents)` to re-inject signatures. -- **parseStream (CCA unwrap path):** as each chunk's `response.candidates[0].content.parts` is read, - call `observeAntigravityReplay(model, sessionId, parts)`. The adapter therefore needs the stable - `sessionId` available in `parseStream` — pass it via a per-request closure (the adapter is created - per provider; thread the sessionId through `runTurn`/a request-scoped field, or recompute it from - the parsed request the same deterministic way). -- **clear-on-invalid:** when a 400 with a signature-related message is seen - (`safeVertexHttpErrorMessage` classifies it as `invalid request`), call - `clearAntigravityReplay(model, sessionId)` so the next attempt starts clean. - -### C) Claude-on-Antigravity inline sanitization (no cache) - -For `claude*` models routed through Antigravity, instead of the cache: -- Strip thinking blocks with empty/invalid signatures from outgoing `request.contents` - (mirror `StripEmptySignatureThinkingBlocks` / `StripInvalidBypassSignatureThinkingBlocks`). -- Drop signature fields on non-model / non-thinking parts. - -Keep this a small pure helper `sanitizeAntigravityClaudeSignatures(contents)` with focused tests. - -### D) Stabilization close-out (reuse Phase 11/12 modules) - -- **Truncation fail-closed:** reuse `google-truncation.ts` (Phase 12). Antigravity `finishReason` - lives under the unwrapped `response.candidates[0].finishReason`; the same MAX_TOKENS / - MALFORMED_FUNCTION_CALL check applies. -- **Estimated usage:** Antigravity DOES return `usageMetadata` (renamed from `cpaUsageMetadata` - upstream; the parser already unwraps `response`). So usage is `reported`, NOT estimated — do not - add `google-antigravity` to `isEstimatedUsageProvider`. Pin with a test. -- **Error classification + redaction:** reuse `google-errors.ts` (Phase 11) with an Antigravity - label; map 429 `RESOURCE_EXHAUSTED` + `ErrorInfo.reason` (`QUOTA_EXHAUSTED` → quota exhausted, - `RATE_LIMIT_EXCEEDED` → rate limit) onto the existing split. - -## Tests - -- `tests/google-antigravity-replay.test.ts`: observe two chunks with signatures → cached; apply - re-injects into matching `contents[ci].parts[pi].thoughtSignature`; signatures shorter than - `MIN_SIGNATURE_LEN` are ignored; function-call parts dedup by callId; TTL expiry drops entries; - `antigravityUsesReplayCache("claude-…")` is false; clear-on-invalid empties the entry. -- `tests/google-antigravity-claude-signatures.test.ts`: empty-signature thinking blocks stripped; - signature fields removed from non-model parts. -- truncation: a CCA stream ending `response.candidates[0].finishReason:MAX_TOKENS` mid tool call → - terminal error. usage: a `usageMetadata`-bearing CCA stream → `reported`. - -## Verify - -`bun x tsc --noEmit` clean + `bun test ./tests/` green. Minimal proof: observe→apply round-trip -re-injects a signature; a Claude request with an empty-signature thinking block is sanitized. - -## Depends-on / enables - -- Depends-on: Phase 21 (CCA wire + sessionId), Phase 11/12 (errors/truncation/usage modules). -- Enables: Phase 2 (google-antigravity) reaches the Cursor/Kiro hardening bar — closes the 140 track - for the two Gemini-family ports. - ---- - -## ✅ Implemented (commit `23806df`) - -- NEW `src/adapters/google-antigravity-replay.ts` — `observeAntigravityReplay` / `applyAntigravityReplay` - / `clearAntigravityReplay` / `antigravityUsesReplayCache` (Gemini-only, 1h TTL, 10240-entry bound, - ≥16-char signatures, `extra_content.google.thought_signature` alias). -- `src/adapters/google-antigravity-wire.ts` — `sanitizeAntigravityClaudeSignatures` (Claude inline path). -- `src/adapters/google.ts` — per-request closure (model/session); buildRequest applies replay (Gemini) - or Claude sanitize; parseStream observes signatures from the unwrapped model parts. -- Tests: `tests/google-antigravity-replay.test.ts` (9). Suite 1043/0, tsc clean. -- Final verification (glm-5.2 subagent): all 7 integration items PASS, no bug/concurrency/security risk. diff --git a/devlog/_fin/140_remaining-provider-ports/30_phase3_amazon-bedrock.md b/devlog/_fin/140_remaining-provider-ports/30_phase3_amazon-bedrock.md deleted file mode 100644 index 09e39d30f..000000000 --- a/devlog/_fin/140_remaining-provider-ports/30_phase3_amazon-bedrock.md +++ /dev/null @@ -1,53 +0,0 @@ -# 140.30 — Phase 3: amazon-bedrock (new adapter + SigV4 + eventstream, HARD) - -> One PABCD pass. NEW adapter. **Establishes the shared AWS eventstream decoder + SigV4 signer** -> (reused by Phase 40 kiro). Grounded in jawcode (cites are jawcode paths). - ---- - -## Goal - -Stream Bedrock **ConverseStream** (Claude/etc. on AWS) through opencodex — porting SigV4 auth and the -AWS binary **eventstream** decoder, both with **zero `@aws-sdk/*` deps** (jawcode is pure WebCrypto/TS). - -## What we port (jawcode) - -- **SigV4** (`aws-sigv4.ts:104-218`): `signRequest` → signed headers (host, x-amz-date, x-amz-content-sha256, authorization, ±x-amz-security-token); HMAC signing-key chain (kSecret→kDate→kRegion→kService→kSigning); WebCrypto `sha256Hex`/`hmac`; RFC-3986 canonicalization. -- **Credential chain** (`aws-credentials.ts:50-496`, zero `@aws-sdk`): env keys → `~/.aws/credentials`+`config` INI (SSO) → SSO portal fetch + `~/.aws/sso/cache` → `credential_process` → EC2 IMDSv2. Cache per (profile,region) + 60s skew. -- **EventStream decoder** (`aws-eventstream.ts:43-185`) — **the shared module**: big-endian framing `[total_len u32][headers_len u32][prelude_crc][headers][payload][msg_crc]`; `crc32` (poly 0xEDB88320), `decodeMessage` (CRC-checked), `async* decodeEventStream` (chunk-boundary stitching). -- **ConverseStream builder** (`amazon-bedrock.ts:204-256,570-810`): messages → WireMessage; tool results batched into one user msg (`:662-696`); thinking config `{type:"enabled",budget_tokens,display:"summarized"}`; **thinking-signature edge case** — only `anthropic.claude*` keep `signature`, else demote to `[Thinking]: text` (`:631-651`). -- **Endpoint:** `POST https://bedrock-runtime.{region}.amazonaws.com/model/{modelId}/converse-stream`, `Accept: application/vnd.amazon.eventstream`. - -## opencodex fit - -- **NEW** `src/adapters/bedrock.ts` (`createBedrockAdapter`): `buildRequest` (ConverseStream + SigV4-sign) + `parseStream` (eventstream → `AdapterEvent`). -- **NEW shared** `src/lib/aws-auth.ts` (SigV4 + credential chain) and `src/lib/eventstream-decoder.ts` (the decoder — **40 reuses this**). -- **Dispatch:** add `case "bedrock"` in `resolveAdapter` (`server.ts:186`). -- **Config:** `awsRegion?`, `awsProfile?` (no `apiKey` — creds from env/`~/.aws`/SSO/IMDS). Models: ~119 seed. -- **Event bridge:** `contentBlockStart/Delta` text→`text_delta`, toolUse→`tool_call_start`/`tool_call_delta`, `reasoningContent.text`→`thinking_delta`, signature→capture-not-emit, `contentBlockStop`(tool)→`tool_call_end`, `messageStop`→stopReason, `metadata`→usage. - -## Sub-steps (this PABCD pass) - -1. **A:** port `aws-eventstream.ts` → `src/lib/eventstream-decoder.ts`; unit-test CRC32 (`"123456789"→0xcbf43926`) + frame decode + multi-chunk stitch. -2. **A:** port SigV4 + credential chain → `src/lib/aws-auth.ts`; unit-test signature vs jawcode golden (fixed date). -3. **B:** `createBedrockAdapter` — ConverseStream builder (tool batching, thinking-signature demotion) + the eventstream→AdapterEvent bridge; wire `resolveAdapter`. -4. **C:** mocked Converse stream → AdapterEvent sequence + usage; SigV4 unit; existing adapters green; `tsc`/`bun test`. - -## Risks - -| Risk | Mitigation | -|------|------------| -| SigV4 mismatch → silent 403 | fixed-date unit test vs jawcode golden hex | -| thinking-signature corruption (multi-turn) | port demote-to-text for non-Anthropic exactly (`:631-651`); test both | -| CRC / chunk-boundary decode error | port CRC table verbatim; test known vectors + stitched frames | -| tool-result ordering | batch consecutive toolResults into one user msg (`:662-696`) | -| credentials unresolved (silent) | test env/profile/SSO/IMDS paths; log credential source in debug | - -## Verify (minimal proof) - -Converse stream against `us.anthropic.claude-*` + a SigV4 unit test (`02:218`). - -## Depends-on / enables - -- **Depends-on:** none (can run parallel with Phase 10 — disjoint files). -- **Enables:** `src/lib/eventstream-decoder.ts` reused by **Phase 40 (kiro)** — kiro cannot `parseStream` until this ships. diff --git a/devlog/_fin/140_remaining-provider-ports/40_phase4_kiro.md b/devlog/_fin/140_remaining-provider-ports/40_phase4_kiro.md deleted file mode 100644 index 1b710a979..000000000 --- a/devlog/_fin/140_remaining-provider-ports/40_phase4_kiro.md +++ /dev/null @@ -1,54 +0,0 @@ -# 140.40 — Phase 4: kiro (new adapter, import auth, HARD) - -> One PABCD pass. NEW adapter. **Import-first auth** (read kiro-cli SQLite, not browser OAuth) + -> **reuses Phase 30's eventstream decoder**. Grounded in jawcode (cites are jawcode paths). - ---- - -## Goal - -Stream kiro (AWS CodeWhisperer agent) through opencodex via import-first auth + the shared AWS -eventstream decoder. Requires anti-detection fingerprint headers for upstream acceptance. - -## What we port (jawcode) - -- **Auth — import-first** (`kiro.ts:392-487`, `utils/oauth/kiro.ts:1-50`): - - read kiro-cli SQLite: `~/Library/Application Support/kiro-cli/data.sqlite3` (mac) / `~/.kiro/sso/cache.db` (linux); token keys `kirocli:social:token`/`kirocli:odic:token`/`codewhisperer:odic:token` (`:313-317`). - - token shape `{ access_token, refresh_token, expires_at, profile_arn?, region? }`. - - refresh `POST https://prod.{region}.auth.desktop.kiro.dev/refreshToken` (`:371-389`); 60s skew; use-stale-on-refresh-fail (`:409-429`). - - config `KIRO_ACCESS_TOKEN`, `KIRO_PROFILE_ARN`. -- **Wire** (`kiro.ts:49-104,145-305`): endpoint `https://runtime.{region}.kiro.dev/`; `Content-Type: application/x-amz-json-1.0`; `Accept: application/vnd.amazon.eventstream`; `x-amz-target: AmazonCodeWhispererStreamingService.GenerateAssistantResponse`. `buildPayload` (history + currentContent + tools, name max 64) ; `parseKiroPayload` heuristics for `{content}`/`{name,input,toolUseId}`/`{stop}`/`{usage}`. -- **Fingerprint headers** (`kiro.ts:72-104`): `sha256(hostname-username-…)` machine fingerprint + `aws-sdk-js/… KiroIDE-{version}-{fp}` User-Agent — **required** or upstream rejects. Stable conversation id = hash of first-3 + last message (`:751-765`). -- **Streaming** (`kiro.ts:31,591`): reuses `decodeEventStream` from `aws-eventstream.ts` → **the Phase 30 module**. -- **Models** (`special.ts:82-91`): 8 static (`kiro-auto`, `claude-sonnet-4.5`, `claude-haiku-4.5`, `deepseek-3.2`, `minimax-m2.5`, `glm-5`, `qwen3-coder-next`, …). - -## opencodex fit - -- **NEW** `src/oauth/kiro.ts` (SQLite read + refresh; import-only, no PKCE) → register in `OAUTH_PROVIDERS`. -- **NEW** `src/adapters/kiro.ts` (`buildRequest` payload+fingerprint headers+conversation-id; `parseStream` via the **shared `src/lib/eventstream-decoder.ts` from Phase 30**). -- **Dispatch:** add `case "kiro"` in `resolveAdapter` (`server.ts:186`). Models: 8 static. - -## Sub-steps (this PABCD pass) - -1. **A:** `src/oauth/kiro.ts` — SQLite token read (mac+linux paths) + manual-paste fallback + desktop refresh; register in `OAUTH_PROVIDERS`. Test both import paths. -2. **B:** `src/adapters/kiro.ts` — `buildPayload` port + fingerprint/User-Agent + conversation-id hash; `parseStream` over `decodeEventStream` (Phase 30) + `parseKiroPayload` heuristics; wire `resolveAdapter`; 8-model lookup. -3. **C:** `ocx login kiro` (import) → single-turn text on `kiro-auto`; assert stream + 401→refresh; `tsc`/`bun test`. - -## Risks - -| Risk | Mitigation | -|------|------------| -| anti-detection headers rejected | port fingerprint + User-Agent exactly; document beta/rate-limit | -| **ToS / IDE-impersonation** | ⚠️ **explicit user decision before shipping** (`02:242`); recommend a transparency note | -| token refresh fails silently → hang | 60s skew + one 401-retry then clear error (`:409-429`) | -| SQLite read on non-mac | linux path + manual-paste fallback; test both | -| eventstream decoder absent | **HARD DEP: Phase 30 must ship `src/lib/eventstream-decoder.ts` first** | - -## Verify (minimal proof) - -Import kiro-cli token; `kiro-auto` single-turn text (`02:219`). - -## Depends-on / enables - -- **Depends-on:** **Phase 30** (the shared eventstream decoder) — auth/payload work can proceed in parallel, but `parseStream` lands after 30. -- **Enables:** nothing downstream (kiro is a leaf). diff --git a/devlog/_fin/140_remaining-provider-ports/50_phase5_cursor.md b/devlog/_fin/140_remaining-provider-ports/50_phase5_cursor.md deleted file mode 100644 index 354d8829d..000000000 --- a/devlog/_fin/140_remaining-provider-ports/50_phase5_cursor.md +++ /dev/null @@ -1,47 +0,0 @@ -# 140.50 — Phase 5: cursor (new adapter, HTTP/2+protobuf+exec, HARD+) - -> One PABCD pass (MVP). The **hardest** port, fully isolated. **Detailed plan: `devlog/350_cursor-provider-add/`.** -> This doc is the roadmap stub; 350 is the source of truth for cursor. - ---- - -## Goal - -Stream cursor agent models (text + thinking) through opencodex over cursor's HTTP/2 Connect+protobuf -transport. MVP = login + single/multi-turn **text**; exec bridge **stubbed**. - -## Why it is last + isolated - -Cursor shares **zero** code with Phases 10–40. It breaks all three opencodex transport assumptions -(HTTP/1.1 fetch / JSON / unidirectional SSE) at once → HTTP/2 + protobuf + bidirectional agent RPC. -The `ProviderAdapter` interface (`base.ts:8-20`) cannot express it without a transport escape hatch. - -## The full plan lives in 350 - -`devlog/350_cursor-provider-add/`: -- `00_overview.md` — the 3 structural clashes + scope. -- `01_cursor-anatomy.md` — protocol (HTTP/2 Connect framing, `@bufbuild/protobuf`, PKCE oauth poll, `GetUsableModels` discovery, exec + KV handshake, checksum-skipped). -- `02_opencodex-fit.md` — the **optional `runTurn()` adapter hook** escape hatch (additive; isolates HTTP/2+protobuf to `src/adapters/cursor/`). -- `03_phased-plan.md` — sub-phases: **0 transport spike → 1 oauth → 2 text+KV → 3 state → 4 models = MVP**; 5 exec bridge optional/separate-approval. -- `04_risks-and-decisions.md` — decisions (reuse jawcode protobuf, stub exec, skip checksum) + risks. - -## Phase-50 summary (for this roadmap) - -| Aspect | Plan (detail in 350) | -|--------|----------------------| -| Adapter | NEW `src/adapters/cursor/` (transport submodule) via optional `runTurn` hook | -| Transport | `node:http2` `http2.connect` + Connect framing + `@bufbuild/protobuf` | -| Auth | PKCE poll OAuth (`OAUTH_PROVIDERS.cursor`) — reuses the Phase-20 registry pattern | -| MVP scope | text + thinking; exec STUB + in-memory KV handshake (or turn stalls) | -| Models | ~145 seed + dynamic `GetUsableModels` | -| exec bridge | **optional Phase 5** — separate approval (makes opencodex a partial Cursor CLI host) | - -## Sub-steps / Risks / Verify - -See `350` — `03_phased-plan.md` (sub-steps + dependency graph) and `04_risks-and-decisions.md`. -Minimal proof: OAuth login → single-turn text on a `composer-*` model → multi-turn regression (`02:220`). - -## Depends-on / enables - -- **Depends-on:** the `OAUTH_PROVIDERS` registration pattern (established Phase 20); otherwise isolated. -- **Enables:** nothing downstream. Ships independently of 10–40. diff --git a/devlog/_fin/141_kiro-adapter-impl/00_plan.md b/devlog/_fin/141_kiro-adapter-impl/00_plan.md deleted file mode 100644 index d686554ef..000000000 --- a/devlog/_fin/141_kiro-adapter-impl/00_plan.md +++ /dev/null @@ -1,33 +0,0 @@ -# 141.00 — kiro adapter on dev (implementation MOC) - -> Branch: `feat/kiro-on-dev` (off `dev`, cursor-free). Implements the kiro provider for -> opencodex independent of the cursor stack. Grounded in jawcode + the 260628 live-confirmed -> CodeWhisperer contract (codex `015_reverse-engineering/45_ki_codewhisperer_wire_stream_oauth.md`). -> Supersedes-by-execution: `140_remaining-provider-ports/40_phase4_kiro.md` (the original port plan). - -## Why a dev-based branch -- `dev` is cursor-free (adapters: anthropic/azure/base/google/image/openai-chat/openai-responses). -- kiro has **zero** dependency on the cursor stack (22k-line protobuf/mcp/native-exec). Its only - hard dep is the AWS eventstream decoder — which does **not** exist on any branch yet, so we port it. -- Therefore kiro stacks cleanly on dev; cursor stays separate. - -## Work-phase map (one full PABCD per phase) -- **P1 — eventstream decoder** (`src/lib/eventstream-decoder.ts` + test). Foundational. ← THIS -- **P2 — kiro OAuth** (`src/oauth/kiro.ts`): import-first kiro-cli SQLite read (mac/linux) + - desktop refresh + manual-paste fallback; register in OAUTH_PROVIDERS. -- **P3 — kiro adapter** (`src/adapters/kiro.ts`): `buildRequest` (conversationState; toolUses.input - = JSON **object**; toolResult adjacency; fingerprint/KiroIDE UA; stable conversationId) + - `parseStream` (eventstream → AdapterEvent; **discriminate by stop/input, not name**). -- **P4 — wiring**: registry entry (`adapter:"kiro"`, runtime.{region}.kiro.dev) + `resolveAdapter` - case + 8 static models. -- **P5 — verify**: `tsc`/`bun test`; `ocx login kiro` import + single-turn live smoke; ToS note. - -## Correctness carried from the jawcode live debugging (must-have from day one) -1. `toolUses[].input` = JSON object (NOT JSON.stringify string) — else REQUEST_BODY_INVALID. -2. toolResults ride on the userInputMessage following their assistant turn (history adjacency). -3. Stream tool events repeat `name`+`toolUseId` on every chunk → discriminate `stop`→`input`→`name`. - -## Risks -- anti-detection headers required (fingerprint + KiroIDE UA) or upstream rejects. -- ToS: third-party harness impersonation — explicit transparency note before ship. -- SQLite read on non-mac → linux path + manual-paste fallback. diff --git a/devlog/_fin/141_kiro-adapter-impl/20_phase2_oauth.md b/devlog/_fin/141_kiro-adapter-impl/20_phase2_oauth.md deleted file mode 100644 index 833682588..000000000 --- a/devlog/_fin/141_kiro-adapter-impl/20_phase2_oauth.md +++ /dev/null @@ -1,71 +0,0 @@ -# 141.20 — Phase 2: kiro OAuth (import-first), plan - -> Branch `feat/kiro-on-dev`. NEW `src/oauth/kiro.ts` + register in `OAUTH_PROVIDERS`. -> Port source: jawcode `packages/ai/src/providers/kiro.ts` (readKiroCliSqlite, refreshKiroDesktopToken, -> resolveKiroAuth) + `utils/oauth/kiro.ts`. Contract: codex `015/45_ki_codewhisperer_wire_stream_oauth.md` §3. - -## opencodex contract (verified from src/oauth) -- `OAuthProviderDef = { login(ctrl, opts), refresh(refreshToken, signal), providerConfig, defaultModel }`. -- `OAuthCredentials = { refresh, access, expires(epoch ms), email?, accountId? }`. -- import-first precedent: `xai/anthropic` use `{importLocal:"fallback"}`; `kimi` uses a direct import login. - kiro is **import-only** (no browser/PKCE) → `login` reads kiro-cli SQLite; manual-paste as fallback. - -## Design -### NEW `src/oauth/kiro.ts` -- `loginKiro(ctrl): Promise` - - read kiro-cli SQLite via `bun:sqlite` (readonly): - mac `~/Library/Application Support/kiro-cli/data.sqlite3`, linux `~/.kiro/sso/cache.db`; - table `auth_kv`, keys `kirocli:social:token`/`kirocli:odic:token`/`codewhisperer:odic:token`; - value JSON `{access_token, refresh_token, expires_at, profile_arn?, region?}`. - - map → `{ access, refresh, expires: Date(expires_at).getTime() }`. - - if no SQLite token: `ctrl.onManualCodeInput()` manual-paste fallback (accept raw access token; or - `KIRO_ACCESS_TOKEN` env). `ctrl.onProgress` for status. -- `refreshKiroToken(refresh, signal): Promise` - - `POST https://prod.{region}.auth.desktop.kiro.dev/refreshToken` body `{refreshToken}` → - `{accessToken, refreshToken?, expiresIn}`; `expires = Date.now()+expiresIn*1000` (60s skew handled by caller). - - region: from stored cred / `KIRO_REGION` / default `us-east-1`. -- register in `OAUTH_PROVIDERS.kiro = { login: loginKiro, refresh: refreshKiroToken, providerConfig: oauthConfig("kiro"), defaultModel: oauthDefaultModel("kiro") }`. - - **needs**: `kiro` entry in the provider registry so `deriveOAuthProviderConfig("kiro")`/`deriveOAuthDefaultModel("kiro")` - resolve. providerConfig: `{ adapter:"kiro", baseUrl:"https://runtime.{region}.kiro.dev", authMode:"oauth" }` - (registry add is Phase 4 wiring — for Phase 2, mirror chatgpt's inline providerConfig to avoid ordering dep, - OR land the registry entry here. **audit: which is cleaner for opencodex?**) - -### profileArn / region (NOT in OAuthCredentials) -- `OAuthCredentials` has no profileArn/region. Like jawcode, the **adapter** (Phase 3) resolves profileArn at - request time from SQLite (`profile_arn`) or `KIRO_PROFILE_ARN`, and region from cred/env/default. Phase 2 - stores only access/refresh/expires. **audit: confirm this split is acceptable vs needing a cred extension.** - -## Sub-steps (Phase 2 PABCD) -- A: Backend employee audits THIS plan vs opencodex `src/oauth/{index,types,store,login-cli}.ts` + kimi/xai - precedent + jawcode source. Resolve the two audit questions (inline vs registry providerConfig; profileArn split). -- B: implement `src/oauth/kiro.ts` + register; tests (`tests/kiro-oauth.test.ts`): SQLite import (temp db), - manual-paste fallback, refresh mapping, expires skew. -- C: `bun test` + `bun x tsc --noEmit`; Backend verify DONE. - -## Risks -- SQLite path differs per OS → mac+linux paths + manual-paste fallback (test both). -- profileArn absence → adapter must error clearly (Phase 3). -- ToS: import-first reuses installed kiro-cli creds — transparency note at ship. - -## Audit resolution (Backend, PASS + B1–B4) — 260628 - -- **CORRECTION (B2):** kimi is a **device-auth grant**, NOT a local-token import. The real import - precedent to mirror is `src/oauth/local-token-detect.ts` (`detectGrokCliToken`/`detectClaudeCodeToken`). -- **Q1 → add registry entry now (not inline).** `OAUTH_PROVIDERS.kiro.providerConfig = oauthConfig("kiro")` - is evaluated **eagerly at module load**; `oauthConfig` throws if the registry lacks `kiro` → importing - `oauth/index.ts` would crash the CLI. So Phase 2 MUST also land the registry entry in `src/providers/registry.ts`: - `{ id:"kiro", adapter:"kiro", baseUrl:"https://runtime.us-east-1.kiro.dev", authKind:"oauth", oauthId:"kiro", defaultModel:"kiro-auto" }`. - baseUrl seed is static (no `{region}` template) → Phase 3 adapter rewrites per region. Registry entry also - feeds init wizard/preset/oauth-id discovery (`derive.ts`). -- **Q2 → adapter-time resolution confirmed.** `getValidAccessToken()` returns only `cred.access`; the rest of - the credential is never surfaced via the standard path. So profileArn (SQLite `profile_arn` / `KIRO_PROFILE_ARN`) - and region (`KIRO_REGION`/default) are resolved in the **Phase 3 adapter** at request time. Phase 2 stores only - `{refresh, access, expires}`. Do NOT extend `OAuthCredentials`. -- **B3 (GUI):** GUI `startLoginFlow` builds `ctrl` without `onManualCodeInput` (only CLI wires it). So with no - SQLite token AND no `KIRO_ACCESS_TOKEN`, `loginKiro` must throw a clear "no kiro-cli token found" error — never hang. -- **B4 (manual paste):** call `ctrl.onManualCodeInput?.()` **directly** inside `kiro.ts` and treat the return as a - **raw access token** (kiro does not use OAuthCallbackFlow / the callback-server code parser). - -→ Phase 2 build scope (revised): `src/oauth/kiro.ts` (SQLite import via bun:sqlite + env + guarded manual-paste + -desktop refresh) **+** `src/providers/registry.ts` kiro entry **+** `OAUTH_PROVIDERS.kiro` registration **+** -`tests/kiro-oauth.test.ts`. diff --git a/devlog/_fin/141_kiro-adapter-impl/30_phase3_adapter.md b/devlog/_fin/141_kiro-adapter-impl/30_phase3_adapter.md deleted file mode 100644 index 653a3b0e3..000000000 --- a/devlog/_fin/141_kiro-adapter-impl/30_phase3_adapter.md +++ /dev/null @@ -1,86 +0,0 @@ -# 141.30 — Phase 3: kiro adapter (buildRequest + parseStream), plan - -> Branch `feat/kiro-on-dev`. NEW `src/adapters/kiro.ts` (implements `ProviderAdapter`) + `resolveAdapter` -> case in `src/server.ts`. Depends on Phase 1 (`src/lib/eventstream-decoder.ts`) + Phase 2 (oauth/registry). -> Port: jawcode `packages/ai/src/providers/kiro.ts` `buildPayload`/`parseKiroPayload` (+ the 260628 -> live-confirmed fixes). Contract: codex `015/45_ki_codewhisperer_wire_stream_oauth.md`. - -## opencodex contract (verified from src/adapters/base.ts) -```ts -interface ProviderAdapter { - name: string; - buildRequest(parsed: OcxParsedRequest, incoming?: IncomingMeta): { url; method; headers; body }; // SYNC on dev - parseStream(response: Response): AsyncGenerator; - parseResponse?(response: Response): Promise; -} -``` -- `buildRequest` is **sync** on dev (no Promise) and `IncomingMeta = { headers: Headers }`. -- `resolveAdapter(providerConfig)` switch in `server.ts:282` (cases: openai-chat/anthropic/openai-responses/google/azure...). Add `case "kiro"`. - -## Design (NEW src/adapters/kiro.ts) -### buildRequest(parsed, incoming) -- `url = https://runtime.{region}.kiro.dev/` (region from `KIRO_REGION` / default us-east-1; registry seed is us-east-1). -- headers: `authorization: Bearer `, `content-type: application/x-amz-json-1.0`, - `accept: application/vnd.amazon.eventstream`, `x-amz-target: AmazonCodeWhispererStreamingService.GenerateAssistantResponse`, - KiroIDE-spoof `user-agent`/`x-amz-user-agent` (sha256(hostname-username) fingerprint), `x-amzn-kiro-agent-mode: vibe`, - `x-amzn-kiro-profile-arn: `, `amz-sdk-invocation-id: `. -- body: `conversationState` from `parsed` (Responses-shaped input → CW history): - - **toolUses[].input = JSON object** (NOT stringified) ← carried fix. - - **toolResults adjacency** (each on the userInputMessage after its assistant turn) ← carried fix. - - stable `conversationId` (hash of first-3 + last message), `chatTriggerType:"MANUAL"`, `origin:"AI_EDITOR"`. - - tools from `parsed` tool defs (name ≤64). -### parseStream(response) -- `for await (msg of decodeEventStream(response.body!))` → JSON.parse(msg.payload) → - **discriminate `stop` → `input` → `name`** (carried fix; CW repeats name on every tool event) → - emit `AdapterEvent`s (text delta, tool-call start/args-delta/done, completed). - -## AUDIT QUESTIONS (Backend A-phase) -- **Q1 token flow:** With sync `buildRequest`, how does the adapter obtain the resolved kiro access token? - Does the server pre-resolve the OAuth token and pass it via `incoming.headers` (authMode), or must the - adapter read it? Check `server.ts` around `resolveAdapter` (L282) + the request pipeline (L437) + how - anthropic/xai (oauth) adapters receive their token. Determine the idiomatic path for kiro. -- **Q2 profileArn (sync):** profileArn must be resolved at buildRequest time. SQLite read is sync (bun:sqlite), - so a sync `readKiroProfileArn()` (SQLite `profile_arn` / `KIRO_PROFILE_ARN`) is feasible — confirm acceptable, - and where region should come from. -- **Q3 AdapterEvent shape:** Read `src/types.ts` `AdapterEvent` (L177+) and an existing `parseStream` - (openai-responses.ts / anthropic.ts) to fix the EXACT event variants kiro must emit for: text delta, - tool-call (id/name/args-delta), and completion. Map CW stream events → those variants. -- **Q4 OcxParsedRequest shape:** Read `src/types.ts` `OcxParsedRequest` (L1+) to know the input/messages/tools - shape buildRequest receives (Responses-API-derived) and how to build CW history from it. - -## Sub-steps -- A: Backend audits this plan + answers Q1-Q4 with file:line (real adapter token flow + AdapterEvent + OcxParsedRequest). -- B: implement `src/adapters/kiro.ts` + `resolveAdapter` case; tests `tests/kiro-adapter.test.ts` (buildRequest body - shape incl. input=object + toolResult adjacency + headers; parseStream over encodeMessage frames incl. tool-arg - accumulation with name-repeat). Backend B-verify. -- C: bun test + tsc; D: close. - -## Carried correctness (must-have) -input=object · toolResult adjacency · stream discriminate stop/input/name (the 3 jawcode live fixes). - -## Audit resolution (Backend, PASS + C1/C2) — 260628 - -- **Q1 (token):** server pre-resolves the OAuth token async (`server.ts:~L414`) and it reaches the adapter via - `provider.apiKey` (token already injected because kiro is `authKind/authMode:"oauth"`, `server.ts:~L412`). - → kiro `buildRequest` uses the provided token; **must NOT re-resolve**. (Confirm exact accessor in B.) -- **Q2 (profileArn, sync):** sync `bun:sqlite` read OK inside buildRequest. Isolate `readKiroProfileArn()` - (SQLite `profile_arn` / `KIRO_PROFILE_ARN`), memoize, hard-fail with a clear error when absent. - region precedence `KIRO_REGION → baseUrl → us-east-1`. -- **Q3 (AdapterEvent variants, exact):** `text_delta{text}` · `tool_call_start{id,name}` (once per tool id) · - `tool_call_delta{arguments}` (raw partial-JSON string; or whole JSON in one delta if CW gives atomically) · - `tool_call_end{}` (close before next start / before done) · `done{usage?}` · `error{message}`. - Ordering invariant from openai-chat.ts L233/253/282. This is exactly why the stop→input→name discrimination - matters: emit start once, route name-repeat chunks to delta, close on stop. -- **Q4 (OcxParsedRequest):** `context = { systemPrompt?: string[]; messages: OcxMessage[]; tools?: OcxTool[] }`. - `OcxToolCall.arguments` is **already a parsed object** → pass straight into CW `toolUses[].input` (input=object - fix is native here). toolResult adjacency precedent: anthropic.ts L160-185 (scan j=i+1 while role==='toolResult', - group onto the following user turn) — replicate for CW. -- **C1:** rely on Phase-2 oauth token injection; don't re-resolve in adapter. -- **C2:** profileArn helper memoized + hard-fail; region precedence as above. -- **B-validate:** endpoint `runtime.{region}.kiro.dev` + `x-amz-target` against jawcode kiro.ts (already confirmed - this session: host + `AmazonCodeWhispererStreamingService.GenerateAssistantResponse`). - -→ Phase 3 build scope (revised): `src/adapters/kiro.ts` (buildRequest: token from provider.apiKey + sync -profileArn/region + conversationState w/ input=object + toolResult adjacency + KiroIDE fingerprint headers + -stable conversationId; parseStream: decodeEventStream + stop/input/name discrimination → the 6 AdapterEvent -variants) + `resolveAdapter` `case "kiro"` + `tests/kiro-adapter.test.ts`. diff --git a/devlog/_fin/141_kiro-adapter-impl/50_phase5_live_smoke.md b/devlog/_fin/141_kiro-adapter-impl/50_phase5_live_smoke.md deleted file mode 100644 index 2e3c7a348..000000000 --- a/devlog/_fin/141_kiro-adapter-impl/50_phase5_live_smoke.md +++ /dev/null @@ -1,36 +0,0 @@ -# 141.50 — Phase 5: live smoke (PASSED, self-served) - -> Branch `feat/kiro-on-dev`. End-to-end verification of the kiro adapter against the REAL -> CodeWhisperer backend, without Codex/proxy — driving the adapter directly -> (loginKiro import → createKiroAdapter().buildRequest → fetch → parseStream). - -## Result — PASS (2026-06-28) -Imported the installed kiro-cli token (access 233ch, refresh present, profileArn -`arn:aws:codewhisperer:us-east-…`, region us-east-1), single-turn prompt -"Reply with exactly: hello-ocx-kiro": -``` -[req] url=https://runtime.us-east-1.kiro.dev/ bodyBytes=305 -[res] status=200 OK -[events] text_delta,text_delta,done -[assembled-text] "hello-ocx-kiro" -``` -→ auth (oauth import) + conversationState wire + AWS eventstream decode + stop/input/name -parse + AdapterEvent emission all verified against the live backend. Exact expected text returned. - -## What this proves end-to-end -- `loginKiro` reads the real kiro-cli SQLite token and `resolveKiroProfileArn`/`resolveKiroRegion` - return real values. -- `buildRequest` produces a CodeWhisperer-accepted body (200, not the historical - REQUEST_BODY_INVALID — the input=object + adjacency fixes hold on the wire). -- `decodeEventStream` + `parseKiroEvent` correctly stream a real eventstream response to text. - -## Not covered live (sufficiency note) -- A tool-call round-trip was not exercised live (would need a multi-turn result feedback). The tool - wire (input=object), stream tool-event discrimination (name-repeat → start-once/delta), and - toolResult adjacency are covered by unit tests (kiro-adapter.test.ts) and were the exact jawcode - live-confirmed fixes. Text round-trip proves the shared auth/wire/stream path. - -## Goal status -All 5 work-phases complete on feat/kiro-on-dev, each employee-verified; live smoke self-served. -Completion is proven; awaiting explicit user finalize (`goal done`). No push; ToS note present in -the registry entry (third-party harness, import-first). diff --git a/devlog/_fin/141_kiro-adapter-impl/60_phase6_codex_cli_e2e.md b/devlog/_fin/141_kiro-adapter-impl/60_phase6_codex_cli_e2e.md deleted file mode 100644 index c0c7022e2..000000000 --- a/devlog/_fin/141_kiro-adapter-impl/60_phase6_codex_cli_e2e.md +++ /dev/null @@ -1,42 +0,0 @@ -# Phase 6 — Full Codex CLI E2E (real `codex exec` → ocx → kiro) - -## Symptom reported by user -Interactive Codex (model `kiro/claude-sonnet-4.6`) showed "Reconnecting… high demand, -temporary errors" and never produced output. "안되는데?" → "codex exec으로 성공시켜봐". - -## Root cause (confirmed, not guessed) -The proxy listening on `localhost:10100` was the **globally-installed published build** -`/opt/homebrew/lib/node_modules/@bitkyc08/opencodex/dist` (v2.6.0) — which has **no kiro -adapter**. Evidence: `grep 'kiro/' /Users/jun/.codex/opencodex-catalog.json` returned -**empty** → kiro models were never advertised to Codex, so `kiro/*` requests failed. -My kiro implementation lives only on `feat/kiro-on-dev` in the workspace, never published. - -## Fix -Replace the running proxy on 10100 with the **branch build**: -1. `kill -9` the stale published proxy holding 10100. -2. `bun run src/cli.ts start --port 10100` (dev run — `bin/ocx.mjs` is the npm shim that - execs the *bundled published* package, so dev MUST use `src/cli.ts` directly). -3. Branch `start` auto-injected **23 models incl. 11 `kiro/*`** into the Codex catalog - (`/Users/jun/.codex/opencodex-catalog.json`) — verified `kiro/claude-sonnet-4.6` present. - -## Verification (live, real Codex CLI 0.142.3) -- Direct `POST /v1/responses` to the proxy → **HTTP 200** + full Responses SSE - (`response.created → output_item.added → content_part.added → output_text.delta* → - output_text.done`), assembled text "Hi! What are you working on?". Proves - proxy + kiro adapter + Responses re-emission end-to-end. -- `codex exec --model kiro/claude-sonnet-4.6 "Reply with exactly: hello-from-kiro-via-codex"` - → **`hello-from-kiro-via-codex`** (exact), RC=0, 14,547 tokens. -- `codex exec … "What is 17*23? …"` → **`391`** (correct), RC=0. - -## Note on the earlier 429 -The first `codex exec` hit `429 Too Many Requests (exceeded retry limit)`. A direct -`/v1/responses` curl seconds later returned 200, and both subsequent `codex exec` runs -succeeded — so the 429 was **transient CodeWhisperer throttling** from the prior -interactive "Reconnecting 1/5…" retry storm, not a proxy/adapter defect. - -## Operational caveat (for the user) -10100 is now served by the **workspace dev proxy** (`bun run src/cli.ts start`, PID at -write-time 62725). To make this permanent in the normal `ocx` flow, the published install -must be updated to include the kiro adapter (publish from `feat/kiro-on-dev`, then -`ocx update`) — otherwise a future `ocx start` from the global binary reverts to the -kiro-less build. The full Codex CLI path is now proven working against the branch build. diff --git a/devlog/_fin/141_kiro-adapter-impl/60_websearch_parseresponse_fix.md b/devlog/_fin/141_kiro-adapter-impl/60_websearch_parseresponse_fix.md deleted file mode 100644 index 433e60a67..000000000 --- a/devlog/_fin/141_kiro-adapter-impl/60_websearch_parseresponse_fix.md +++ /dev/null @@ -1,41 +0,0 @@ -# 60 — web_search → parseResponse fix (kiro-only live failure) - -## Symptom -Real Codex CLI → opencodex proxy (`localhost:10100`) → kiro showed `Reconnecting… 1/5…5/5` -then `We're currently experiencing high demand, which may cause temporary errors.` **for kiro -only** — every other provider worked. Reproduced headlessly with `codex exec -m -kiro/claude-sonnet-4.6` (exit 1). - -## Diagnosis (evidence-driven) -1. Minimal `curl` to `/v1/responses` with a kiro model **succeeded** (200, streamed exact text) - → adapter + eventstream + Responses bridge are fine for a plain request. -2. Captured the **real** 53 KB Codex request via a throwaway capture server (codex `-c - model_providers.opencodex.base_url=…`). Diff vs the minimal request: 17 `tools` (incl. - hosted `web_search`), 21 KB instructions, `reasoning:{effort:"none"}`. -3. Replayed the captured request to the running proxy → **HTTP 500** - `{"error":{"message":"web-search sidecar requires a non-streaming adapter"}}`. -4. Source: `src/web-search/loop.ts:111` — when Codex sends the `web_search` tool the proxy runs - the model in a **non-streaming** agentic loop and hard-requires `adapter.parseResponse`. - anthropic/google/openai-chat implement it; **kiro had only `parseStream`** → 500 → Codex - retried → "high demand". That is exactly why *only* kiro failed. - -## Fix (commit dd1d924) -`src/adapters/kiro.ts`: hoisted the eventstream decode into a module-level -`parseKiroStream(response)` generator; `parseStream` delegates to it and a new -`parseResponse(response): Promise` drains it into an array. CodeWhisperer only -ever returns an AWS eventstream (no non-stream mode), so draining is the correct equivalent. -Registry/buildRequest unchanged. Tests: `tests/kiro-adapter.test.ts` +2 (parseResponse present; -parity with parseStream). - -## Verification -- Restarted the workspace proxy (it ran pre-fix code in memory; Bun has no hot-reload). -- `codex exec -m kiro/claude-sonnet-4.6 "…hello-codex-exec-kiro"` → **exit 0**, replied - `hello-codex-exec-kiro`. -- Captured 53 KB request replay → now streams to `response.completed` (no 500). -- Adapter 10/10, full suite 608/608, `tsc` 0. Backend employee read-only verify: DONE, plus a - preemptive scan found **no other kiro-only adapter-capability gap** (passthrough/vision are - config/forward concerns, not adapter methods). - -## Follow-up (not a bug, config-only) -If kiro models cannot accept image input natively, list them in `noVisionModels` so the vision -sidecar pre-describes images (analogous to the reasoning-ignore registry metadata). diff --git a/devlog/_fin/142_kiro-token-usage/00_plan.md b/devlog/_fin/142_kiro-token-usage/00_plan.md deleted file mode 100644 index c3fe7ee88..000000000 --- a/devlog/_fin/142_kiro-token-usage/00_plan.md +++ /dev/null @@ -1,95 +0,0 @@ -# 142 — kiro token-usage accounting (heuristic sidecar) - -## Problem (root cause, verified) -- Codex relies on `response.completed.usage.{input_tokens,output_tokens,total_tokens}` to track - the context window and trigger **auto-compact** (`model_auto_compact_token_limit`). -- The kiro adapter emits `yield { type: "done" }` with **no usage** (`src/adapters/kiro.ts:297`). -- `src/bridge.ts:356` builds `response.completed` with `usage: responsesUsage(event.usage)`; - `responsesUsage(undefined)` → `{input_tokens:0, output_tokens:0, total_tokens:0}` (bridge.ts:12-13). -- CodeWhisperer `GenerateAssistantResponse` provides **no reliable usage**. jawcode parses a `usage` - number but its consumer does nothing (`case "usage": break;`, jawcode kiro.ts:746) → Usage stays 0. -- Net effect: with kiro, Codex shows 0 token usage and **never auto-compacts** → context overflow. - -## 2026-06-29 correction — nested accumulation regression -- The v1 fix made Kiro usage non-zero, but it estimated `input_tokens` from the **entire serialized - Kiro payload body**. That payload intentionally includes full `conversationState.history`. -- Codex stores `TokenUsageInfo.total_token_usage` by **adding every response's last usage** - (`TokenUsageInfo::append_last_usage()` in codex-rs `protocol/src/protocol.rs`), while its active - context check uses `last_token_usage.total_tokens` plus local post-model items - (`core/src/context_manager/history.rs`). -- Therefore a stateless provider adapter must not report the full Kiro history body as each - response's additive usage. Doing so re-adds old user/assistant/tool history every turn and produces - the user-visible "nested usage keeps stacking" symptom. -- Kiro upstream payload construction remains unchanged: full history is still sent to CodeWhisperer. - Only the usage number reported back to Codex is changed to a current-turn delta. - -## Goal -Add a heuristic token-estimation **sidecar** so kiro emits non-zero, reasonable usage and Codex's -usage display + auto-compact work. Prefer any real CW usage number if the stream ever provides one. - -## Research grounding (web) -- Rule of thumb: 1 token ≈ 4 chars (English prose), ≈ 0.75 words. -- Empirical model ratios (~±10%): GPT chars/3.6, Claude chars/3.5, Gemini chars/3.8. -- Code / JSON / non-English consume MORE tokens per char (lower chars-per-token). -- Codex traffic is code/JSON/tool-arg heavy → use a **conservative 3.5 chars/token** for kiro text - models so we slightly over-count rather than under-count (under-counting delays auto-compact → - context overflow, the worse failure). - -## Kiro models (all TEXT LLMs — sidecar applies to all) -`kiro-auto, claude-opus-4.8/4.7/4.6, claude-sonnet-4.6/4.5, claude-haiku-4.5, deepseek-3.2, -minimax-m2.5, glm-5, qwen3-coder-next` — all text models; the char heuristic applies uniformly. - -## Design - -### New module `src/lib/token-estimate.ts` (sidecar, reusable + testable) -- `charsPerToken(modelId?: string): number` — model-aware ratio; kiro text models → 3.5, - generic default → 4. -- `estimateTokens(text: string, modelId?: string): number = max(0, ceil(len / charsPerToken))`. -- Pure, dependency-free (no tokenizer dep added). - -### Wire into `src/adapters/kiro.ts` -- `createKiroAdapter` is created **per request** (`server.ts:440` → fresh factory each call), so a - closure variable is race-free. -- `buildRequest`: build the full Kiro payload exactly as before, but estimate reported `inputTokens` - from the current-turn suffix only: - - include user/developer/tool-result messages after the last assistant message; - - do **not** count old assistant output/tool-call args as new input; - - count system prompt + tool definitions only on the first model turn, so stable prompt overhead is - not re-added forever in Codex's additive session usage. - Store `inputTokens` + `modelId` in per-request closure vars. -- `parseKiroStream(response, modelId?, inputTokens?)`: accumulate output chars from `text_delta` + - `tool_call_delta.arguments`; on terminal `done`, emit - `done{usage:{inputTokens: inputTokens ?? 0, outputTokens: estimateTokens(accumulatedOutput, modelId)}}`. - Params are optional (default ratio + 0 input) so the generator stays usable standalone. -- **Both call sites forward closure values** — `parseStream` AND `parseResponse` (kiro.ts:341, the - web-search sidecar path) must pass `modelId`+`inputTokens`, else web-search emits zero usage - (Backend audit fix #2). -- (Optional) if a real CW `usage` number appears, prefer it. v1: heuristic only; CW currently sends none. - -> Superseded note: the earlier audit statement "per-turn input_tokens == cumulative context size -> because full history is re-sent each turn" was wrong for Codex's additive `total_token_usage`. -> Full history still matters for Kiro upstream context, but the value emitted in -> `response.completed.usage` must be a current-turn delta to avoid repeated old-history addition. - -## Slices (single PABCD pass, 2 build steps) -1. **Estimator** `src/lib/token-estimate.ts` + `tests/token-estimate.test.ts` - (ratios, ceil, empty string, model-aware, monotonicity). -2. **Wiring** kiro adapter input estimate + output accumulation + `done{usage}`; extend - `tests/kiro-adapter.test.ts` to assert `done` carries non-zero input/output tokens. - -## Verification -- `bun test tests/token-estimate.test.ts tests/kiro-adapter.test.ts` green. -- `bun test tests/` full suite no regression. -- `bun x tsc --noEmit` → 0. -- Regression added: `tests/kiro-adapter.test.ts` now proves old history remains in the Kiro request - body while `done.usage.inputTokens` stays stable for the same latest user input, and proves - tool-result follow-ups do not re-count prior assistant tool-call args. -- Self-served live check (optional): drive the adapter and assert the `done` event usage is non-zero - (→ bridge emits non-zero `response.completed.usage` → Codex auto-compact engages). -- Backend read-only verification at A (plan) and B (build). - -## Commits (atomic, no push) -- `feat(kiro): add heuristic token-estimate sidecar (src/lib/token-estimate.ts)` -- `feat(kiro): emit estimated usage so Codex usage + auto-compact work` -- `fix(kiro): report current-turn usage instead of full history body` -- devlog commits per phase (`git add -f`, devlog/ is gitignored). diff --git a/devlog/_fin/142_kiro-token-usage/10_context_usage_sot_plan.md b/devlog/_fin/142_kiro-token-usage/10_context_usage_sot_plan.md deleted file mode 100644 index 892dc5244..000000000 --- a/devlog/_fin/142_kiro-token-usage/10_context_usage_sot_plan.md +++ /dev/null @@ -1,44 +0,0 @@ -# 142.10 — Kiro contextUsagePercentage as source of truth - -## Problem - -The current Kiro adapter ignores Kiro's terminal `contextUsagePercentage` frames and reports only -a current-turn heuristic input delta. That prevents the old full-history nesting regression, but it -also makes Codex's context-window UI collapse to near-zero after short second+ turns because Codex -uses `last_token_usage.total_tokens` as the active context size when a model context window is known. - -Kiro Gateway already treats `contextUsagePercentage` as the authoritative context signal: -`total_tokens = (contextUsagePercentage / 100) * max_input_tokens`. - -## Plan - -1. Extend `/Users/jun/Developer/new/700_projects/opencodex/src/adapters/kiro-events.ts` - to parse terminal `{"usage": ...}` and `{"contextUsagePercentage": ...}` JSON frames. -2. Extend `/Users/jun/Developer/new/700_projects/opencodex/src/adapters/kiro.ts` - so `parseKiroStream` stores the latest context percentage and, when a fixed model window is known, - emits `done.usage.totalTokens` from that Kiro-derived absolute context total. -3. Keep fallback behavior unchanged for `kiro-auto` and streams without context percentage: - current-turn input heuristic plus output estimate. `kiro-auto` must not inherit provider-level - `contextWindow`, because Kiro Auto is a router with no fixed source-of-truth window. -4. Extend `/Users/jun/Developer/new/700_projects/opencodex/src/types.ts` and - `/Users/jun/Developer/new/700_projects/opencodex/src/bridge.ts` with optional - `OcxUsage.totalTokens`, preserving the existing `input + output` default for all other adapters. -5. Update Kiro stream tests to prove: - - parser preserves `contextUsagePercentage`; - - known-window Kiro models use percentage-derived `totalTokens`; - - `kiro-auto` falls back to the existing heuristic even when provider-level `contextWindow` exists; - - previous "current-turn only" regressions still avoid full-history input recounting. - -## Constraint - -Codex currently derives both `last_token_usage` and additive `total_token_usage` from the same -Responses `usage` object. opencodex can make the Kiro-provided absolute context visible in -`last_token_usage.total_tokens`, which fixes context-window display and auto-compact decisions, but -opencodex cannot independently tell Codex "use absolute for last and delta for accumulated total" -without a Codex-side protocol change. - -## Verification - -- `bun test tests/kiro-stream.test.ts` -- `bun test tests/kiro-adapter.test.ts tests/kiro-stream.test.ts tests/bridge.test.ts` -- `bun x tsc --noEmit` diff --git a/devlog/_fin/143_kiro-gateway-parity/00_plan.md b/devlog/_fin/143_kiro-gateway-parity/00_plan.md deleted file mode 100644 index e36682520..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/00_plan.md +++ /dev/null @@ -1,48 +0,0 @@ -# 143 — Kiro <-> kiro-gateway Parity (catch-up) - -Goal: raise opencodex's Kiro (CodeWhisperer) surface to kiro-gateway parity and -stabilize it. One FULL PABCD cycle per surface. Each surface closes with a -typecheck + targeted test and one atomic commit. - -Upstream reference: https://github.com/jwadow/kiro-gateway (Python/FastAPI, -~14.8K LOC across kiro/*.py). opencodex Kiro surface today: src/adapters/kiro.ts -(474 lines) + src/oauth/kiro.ts + shared src/lib/eventstream-decoder.ts, -src/lib/token-estimate.ts, plus shared sidecars (src/vision, src/web-search). - -## Gap matrix (gateway has -> opencodex kiro status) - -| # | Surface | Gateway implementation | opencodex status | Phase | -|---|---------|------------------------|------------------|-------| -| 1 | Native image input | convert_images_to_kiro_format -> userInputMessage.images = [{format, source:{bytes}}] (converters_core.py 641-704, 1354-1362, 1520-1562). Parses OpenAI image_url data URLs + Anthropic image.source.base64 (185-297). | MISSING — userContentText (kiro.ts 91-94) drops every image part; payload has no images field. Vision sidecar also inactive (kiro not in noVisionModels). | 10 | -| 2 | Retry/backoff | network_errors.py classifies 403/429/5xx as retryable; routes retry with account failover. | MISSING — no retry/backoff in adapter or transport for kiro. | 20 | -| 3 | Payload size guard | payload_guards.py trim_payload_to_limit trims oldest history pairs under a byte cap; repairs orphaned tool results. | MISSING — no size guard / history trim. | 30 | -| 4 | Thinking parse-back | thinking_parser.py FSM extracts // blocks from the response stream and emits them as reasoning_content. | PARTIAL — opencodex injects request-side thinking tags (kiro.ts 196-209) but does NOT parse response-side thinking blocks into reasoning events. | 40 | -| 5 | Smart model-name normalization | model_resolver.py normalize_model_name maps versioned/dashed slugs (claude-sonnet-4-5-20250929, claude-3-7-sonnet) to canonical ids. | WEAK — mapModelId (kiro.ts 56-58) only strips a kiro- prefix; no versioned-slug normalization. | 50 | -| 6 | Truncation recovery | truncation_recovery.py detects mid-stream truncation (Issue #56) and injects a synthetic recovery message so the model adapts. | MISSING — no truncation detection/recovery. | 60 | - -## Surfaces intentionally NOT in scope (already at/above parity or out of band) - -- Multi-account failover: opencodex has its own Codex multi-account pool - (codexAccounts) at a different layer; kiro single-credential import is the - documented design. Not a kiro-adapter gap. -- Anthropic/OpenAI dual API ingress: opencodex normalizes upstream-in via its - own Responses parser; gateway's dual ingress is its own front door. Out of band. -- Web search: opencodex already has the shared web-search sidecar wired to kiro - (parseResponse + loop). At parity. -- Token usage: closed in plan 142. CW emits no usage; heuristic estimate stands. - -## PABCD discipline - -- Phase 0 (this doc): design + gap matrix only. No code. -- Phases 10/20/30/40/50/60: one surface each = one full P->A->B->C->D cycle. -- Verification per surface: bun x tsc --noEmit + the surface's targeted - bun test tests/... . Atomic commit per surface. -- Risk tiering: surfaces 1 (payload shape) and 3 (history trim) touch request - construction -> STANDARD+; others LIGHT/STANDARD. - -## Completion criteria - -All six surfaces implemented, each with passing typecheck + tests and an atomic -commit, and a closing parity audit confirming no silent image drop, retry on -transient upstream failures, bounded payload, response-side reasoning surfaced, -versioned model slugs resolved, and truncation surfaced rather than swallowed. diff --git a/devlog/_fin/143_kiro-gateway-parity/01_review_revised_roadmap.md b/devlog/_fin/143_kiro-gateway-parity/01_review_revised_roadmap.md deleted file mode 100644 index d1733ba46..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/01_review_revised_roadmap.md +++ /dev/null @@ -1,37 +0,0 @@ -# 01 - Revised roadmap (external GPT-Pro review folded in) - -An external code-level review of feat/kiro-on-dev vs kiro-gateway reprioritized -the work toward functional hardening. User direction: multi-account failover is -OUT of scope; push functional hardening harder. - -## Status - -- Phase 10 native images: DONE (commit 494be0d, 5 tests, tsc + 23 tests pass). - -## Revised phase order (P0 first) - -| Phase | Tier | Surface | Source | -|-------|------|---------|--------| -| 70 | P0 | Stream exception/error is TERMINAL - stop parsing, no `done` after an upstream exception frame. | review 2.2 / P0-1 | -| 80 | P0 | Kiro-specific HTTP retry/backoff: refresh-on-401/403, 429/5xx exp backoff + jitter, first-token retry (pre-first-byte only), abort upstream on client disconnect. | review 1.7 / 2.3 / P0-2 | -| 90 | P0 | OAuth singleflight refresh + SQLite reload-before-refresh + busy-timeout; API-region vs runtime-region split. | review 2.4 / P0-3 | -| 100 | P0 | Resume / tool-result correctness: E2E matrix for tool-call continuation and compact/resume; repair orphaned tool results in current-turn-only payloads. | review 1.5 / 2.5 / P0-4 | -| 110 | P0 | Eventstream decoder hardening: frame-size cap, header-length bounds, per-header read bounds, malformed-frame fuzz tests. | review 2.1 / P0-5 | -| 120 | P1 | Tool schema sanitization (drop empty required, additionalProperties), long-description handling, orphaned/no-tools fallback. | review 1.4 / P1-2,3 | -| 130 | P1 | Model list/resolver: versioned-slug normalization, max effort advertise, missing official models reconciled. | review 1.2 / P1-4,5 (folds old Phase 50) | -| 140 | P1 | Kiro-specific error mapping: auth/region/quota/model-unavailable to actionable Codex errors. | review 3.5 / P1 | -| 150 | P2 | Truncation detection/recovery. | review 1.4 / P2 (folds old Phase 60) | -| 160 | P2 | Tag usage as estimated + calibration fixtures; debug observability. | review 3.1 / P2 | - -## Superseded earlier stubs - -- Old Phase 20 (retry) -> Phase 80 (expanded). -- Old Phase 30 (payload guard) -> folded into Phase 100/120 request shaping. -- Old Phase 40 (thinking parse-back) -> P2 follow-up; deprioritized below correctness/hardening. -- Old Phase 50 (model normalization) -> Phase 130. -- Old Phase 60 (truncation) -> Phase 150. - -## Discipline unchanged - -One full PABCD cycle per phase. Each closes with bun x tsc --noEmit + targeted -bun test and one atomic commit. Independent verifier dispatch per phase. diff --git a/devlog/_fin/143_kiro-gateway-parity/05_hardening_review.md b/devlog/_fin/143_kiro-gateway-parity/05_hardening_review.md deleted file mode 100644 index fdf03e11b..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/05_hardening_review.md +++ /dev/null @@ -1,36 +0,0 @@ -# 05 — Hardening phase map (external GPT-Pro review folded in) - -Scope decision (user): multi-account failover is OUT of scope. Focus on -functional parity + production hardening of the single-credential Kiro adapter. - -Phase 10 (native images) is DONE — commit 494be0d. - -## Re-prioritized phase map - -| Phase | Priority | Surface | Closes | -|-------|----------|---------|--------| -| 70 | P0 | Stream exception/error is terminal (no `done` after an upstream error frame) | parseKiroStream emitting both error + done | -| 80 | P0 | Kiro HTTP retry/backoff: 401/403 refresh-once, 429/5xx exp backoff w/ jitter, first-token retry BEFORE any emitted delta, abort upstream on client disconnect | no transient-failure resilience | -| 90 | P0 | OAuth singleflight refresh + SQLite reload-before-refresh + busy-timeout | refresh races overwrite creds; silent DB swallow | -| 100 | P0 | Resume / tool-result correctness test matrix (current-turn-only risk) | unverified resume + tool continuation | -| 110 | P0 | Eventstream decoder hardening: frame-size cap, header-length bounds, per-read bounds, fuzz tests | binary-protocol crash/corruption risk | -| 120 | P1 | Tool schema sanitization (strip additionalProperties / empty required), long-description -> system prompt, orphaned/no-tools fallback | Kiro 400s from unsupported schema/tool-result-without-def | -| 130 | P1 | Model resolver: versioned-slug normalization + `max` effort advertise | mis-routed versioned slugs; max effort hidden | -| 140 | P1 | Kiro-specific error mapping (auth/region/quota/model -> actionable Codex errors) | generic upstream failure opacity | -| 150 | P2 | Truncation detection/recovery | silent mid-stream truncation | -| 160 | P2 | Usage estimated-tagging + calibration note | estimated usage indistinct from authoritative | -| 170 | P2 | Debug observability (redacted payload + raw frame behind flag) | future crash-guard cost | - -## Order rationale - -P0 first (survivability/correctness), each one full PABCD cycle with tsc + -targeted tests + atomic commit. Phase 70 is the smallest, highest-leverage P0 -(a correctness bug: a failed upstream call can currently look partially -successful), so it leads. - -## Already-at-parity (no work) - -- Web search sidecar wired to kiro (parseResponse + loop). -- Token usage heuristic (plan 142) — only needs P2 tagging polish. -- Reasoning effort via fake-thinking tags (request side); response-side - parse-back tracked as the original Phase 40 and folded with truncation/P1. diff --git a/devlog/_fin/143_kiro-gateway-parity/100_phase_resume_tool_result_correctness.md b/devlog/_fin/143_kiro-gateway-parity/100_phase_resume_tool_result_correctness.md deleted file mode 100644 index 3b089e023..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/100_phase_resume_tool_result_correctness.md +++ /dev/null @@ -1,55 +0,0 @@ -# Phase 100 (P0-4) - Resume / tool-result correctness - -## Problem - -For `previousResponseId`, `kiroPayloadMessages()` uses -`currentTurnInputMessages()`, which currently slices after the last assistant and -filters assistants out: - -```ts -return messages.slice(lastAssistant + 1).filter(m => m.role !== "assistant"); -``` - -If the current turn starts with `toolResult`, the transmitted Kiro payload has -`userInputMessageContext.toolResults` but no preceding `assistantResponseMessage` -with matching `toolUses`. Kiro can reject or misinterpret that payload. - -## Scope - -Repair only resumed tool-result continuations. Keep normal resumed user text -current-turn-only, and keep usage accounting current-turn-only so old tool args -are not re-counted. - -## File changes - -### MODIFY src/adapters/kiro.ts - -1. Split payload slicing from usage slicing: - - `currentTurnUsageMessages(messages)` = old behavior (`slice(lastAssistant+1)`, no assistant) - - `currentTurnPayloadMessages(messages)`: - - if no current toolResult: same as old behavior - - if current tail has a toolResult: include the minimal prior exchange from - after the previous assistant through current tail. This preserves: - `last user/developer -> last assistant(toolUses) -> toolResult(s)`. -2. `kiroPayloadMessages()` should use `currentTurnPayloadMessages`. -3. `estimateKiroInputTokens()` should use `currentTurnUsageMessages` to avoid - charging old user/tool-call context again. - -### MODIFY tests/kiro-adapter.test.ts - -Add resumed tool-result tests: - -- previousResponseId with `user -> assistant toolCall -> toolResult` includes - history with the assistant toolUse and currentMessage with toolResults. -- usage for that resumed toolResult remains based only on the tool output (not - previous user text or huge assistant args). -- normal previousResponseId with latest user text still has no history. - -## Verification - -- bun x tsc --noEmit -- bun test tests/kiro-adapter.test.ts - -## Commit - -fix(kiro): preserve tool-use context for resumed tool results diff --git a/devlog/_fin/143_kiro-gateway-parity/10_phase1_native_images.md b/devlog/_fin/143_kiro-gateway-parity/10_phase1_native_images.md deleted file mode 100644 index eb6af6252..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/10_phase1_native_images.md +++ /dev/null @@ -1,85 +0,0 @@ -# Phase 10 — Native image input (userInputMessage.images) - -## Problem - -CodeWhisperer's userInputMessage accepts an images array (proven by kiro-gateway: -converters_core.py 641-704 + 1354-1362 + 1520-1562). opencodex's kiro adapter -models userInputMessage as { content: string } only, so userContentText -(kiro.ts 91-94) silently drops every OcxImageContent part. Images vanish. - -## Target wire shape (from gateway, native Kiro IDE format) - - userInputMessage.images = [ - { format: "jpeg", source: { bytes: "" } }, - ... - ] - -- format = media subtype: "image/jpeg" -> "jpeg". -- source.bytes = pure base64 (strip any "data:...;base64," prefix). -- Images attach to userInputMessage directly, NOT userInputMessageContext. -- Remote https image URLs are NOT fetchable here -> skip with a text marker, - matching gateway's "URL-based images not supported" behavior. - -## opencodex source shape - -OcxImageContent (types.ts 70-76): { type:"image"; imageUrl: data|https URL; detail? }. -Carried on user/developer/toolResult messages as OcxContentPart[]. - -## Plan (diff-level) - -### MODIFY src/adapters/kiro.ts - -1. Extend KiroUserInputMessage interface: add optional `images?: KiroImage[]`. - New interface: - interface KiroImage { format: string; source: { bytes: string }; } - -2. Add helpers near userContentText: - - parseDataUrlImage(imageUrl): { format; bytes } | undefined - * Only handles `data:` URLs. Returns undefined for https (not fetchable). - * Split on first ","; derive media subtype from the header; bytes = tail. - - extractKiroImages(content): KiroImage[] - * Maps OcxContentPart[] image parts via parseDataUrlImage; drops https. - -3. In buildKiroPayload user/developer branch (around line 233): after computing - `text`, also compute images via extractKiroImages and attach to the entry's - userInputMessage when non-empty. mkUser must accept optional images. - -4. For the FINAL currentMessage userInputMessage: ensure images from the last - user turn survive (currentEntry is popped from history or freshly built). - Since images are already attached to the history entry that becomes - currentEntry, popping preserves them. For the synthetic "(tool results)" / - "(continue)" carriers there are no images by construction. - -5. Leave userContentText unchanged (text extraction stays text-only); images - travel through the separate images field, not inlined as text. - -### Interaction with vision sidecar - -Native image support means we do NOT add kiro to noVisionModels. Kiro now sees -images directly. The sidecar remains the fallback for genuinely text-only -providers. (If a future Kiro model rejects images, that specific model id can be -added to noVisionModels later — not in scope now.) - -## Tests (NEW tests/kiro-images.test.ts) - -- data URL image -> userInputMessage.images[0] == {format:"png", source:{bytes:"..."}} -- media prefix stripped: "data:image/jpeg;base64,AAAA" -> bytes "AAAA", format "jpeg" -- https URL image -> skipped (no images field, no throw) -- mixed text+image -> content has text, images has the image -- no image -> no images field (back-compat: payload identical to before) - -## Verification - -- bun x tsc --noEmit -- bun test tests/kiro-images.test.ts tests/kiro-adapter.test.ts - -## Commit - -feat(kiro): send images natively via userInputMessage.images (gateway parity) - -## Audit note (Backend, PASS) - -toolResult image parts (kiro.ts:263, userContentText(tr.content)) are ALSO -dropped today — e.g. Codex view_image output. That is pre-existing behavior, not -a regression. Phase 10 scopes to user/developer messages only. toolResult image -forwarding is deferred to Phase 11 (follow-up) to keep this slice atomic. diff --git a/devlog/_fin/143_kiro-gateway-parity/110_phase_eventstream_hardening.md b/devlog/_fin/143_kiro-gateway-parity/110_phase_eventstream_hardening.md deleted file mode 100644 index 643579a03..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/110_phase_eventstream_hardening.md +++ /dev/null @@ -1,49 +0,0 @@ -# Phase 110 (P0-5) - Eventstream decoder hardening - -## Problem - -`src/lib/eventstream-decoder.ts` validates CRCs and chunking, but lacks hard -bounds: - -- no maximum frame size -- no `headersLen <= total - 16` guard -- `parseHeaders()` reads with DataView/subarray without checking every required - byte exists first -- bogus large `total` can make the stream buffer grow indefinitely while waiting - for a never-completing frame - -## File changes - -### MODIFY src/lib/eventstream-decoder.ts - -- Add `MAX_MESSAGE_LEN = 16 * 1024 * 1024`. -- In `decodeMessage`: - - reject `total > MAX_MESSAGE_LEN` - - reject `headersLen > total - MIN_MESSAGE_LEN` -- In `parseHeaders`: - - add a small `need(n, label)` helper before every DataView read / fixed-size - subarray read - - throw clear `eventstream: truncated header ...` errors instead of letting - DataView RangeError or silent short subarrays through -- In `decodeEventStream`: - - reject advertised `total > MAX_MESSAGE_LEN` as soon as the first 4 bytes are available - - reject an incomplete single buffered frame if buffer growth exceeds - `MAX_MESSAGE_LEN` - -### MODIFY tests/eventstream-decoder.test.ts - -Add tests: - -- advertised total length over cap throws -- header length exceeding payload boundary throws -- truncated string header throws a controlled eventstream error -- one valid frame split at every byte boundary still decodes (small fuzz) - -## Verification - -- bun x tsc --noEmit -- bun test tests/eventstream-decoder.test.ts tests/kiro-adapter.test.ts - -## Commit - -fix(eventstream): bound frame and header parsing diff --git a/devlog/_fin/143_kiro-gateway-parity/120_phase_tool_schema_sanitization.md b/devlog/_fin/143_kiro-gateway-parity/120_phase_tool_schema_sanitization.md deleted file mode 100644 index c28a23b64..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/120_phase_tool_schema_sanitization.md +++ /dev/null @@ -1,63 +0,0 @@ -# Phase 120 (P1-1) - Tool schema sanitization - -## Problem - -`convertTools()` currently passes tool JSON Schema through directly: - -```ts -inputSchema: { json: (t.parameters ?? {}) } -``` - -kiro-gateway strips schema fields Kiro rejects, especially: - -- `additionalProperties` -- empty `required: []` - -Without this, valid Codex/OpenAI-style tool schemas can become Kiro 400s. - -## Scope - -This phase closes schema sanitization only. It also moves tool conversion out of -`kiro.ts` because that file is already at the 500-line limit. - -Out of scope for this phase: - -- long tool description -> system prompt movement -- orphaned toolResult / no-tools fallback - -Those remain P1 follow-up phases. - -## File changes - -### ADD src/adapters/kiro-tools.ts - -- export `convertKiroTools(parsed: OcxParsedRequest): unknown[]` -- recursively sanitize schemas: - - remove every `additionalProperties` key - - remove `required` if it is an empty array - - preserve non-empty `required`, `properties`, `items`, `oneOf`, `anyOf`, etc. -- keep existing behavior: - - name sliced to 64 chars - - description placeholder when empty, sliced to 1024 - - inputSchema json defaults to `{}` - -### MODIFY src/adapters/kiro.ts - -- remove local `convertTools` -- import/use `convertKiroTools` -- keep file <= 500 lines - -### MODIFY tests/kiro-adapter.test.ts - -Add test that a tool schema with nested `additionalProperties` and empty -`required: []` is sanitized before payload construction, while non-empty -required is preserved. - -## Verification - -- bun x tsc --noEmit -- bun test tests/kiro-adapter.test.ts - -## Commit - -fix(kiro): sanitize tool schemas before CodeWhisperer payload diff --git a/devlog/_fin/143_kiro-gateway-parity/125_phase_tool_fallback_hardening.md b/devlog/_fin/143_kiro-gateway-parity/125_phase_tool_fallback_hardening.md deleted file mode 100644 index 31232cc08..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/125_phase_tool_fallback_hardening.md +++ /dev/null @@ -1,133 +0,0 @@ -# Phase 125 (P1 residual) - Kiro tool fallback hardening - -## Trigger - -Phase 120 closed JSON Schema sanitization only. The external code review still -flags three Kiro tool-compatibility gaps: - -- Long tool descriptions are hard-truncated to 1024 chars. -- Tool results can still be sent as structured `toolResults` when no tool - definitions are present. -- Orphaned tool results can still be sent as structured `toolResults` when the - transmitted payload no longer contains the matching assistant `toolUse`. - -## Current state - -- `src/adapters/kiro-tools.ts` sanitizes schemas and returns raw Kiro tool - specifications, but it truncates descriptions. -- `src/adapters/kiro.ts` converts every assistant `toolCall` to Kiro - `toolUses`, and every `toolResult` to pending structured `toolResults`. -- Resume repair from Phase 100 preserves the previous assistant context for - ordinary resumed tool-result turns, but malformed/no-tools inputs still need - fail-closed text fallback. - -## Diff plan - -### MODIFY `src/adapters/kiro-tools.ts` - -- Add a `convertKiroToolContext(parsed)` export returning: - - `tools: unknown[]` - - `systemAdditions: string[]` -- Keep `convertKiroTools(parsed)` as a compatibility wrapper returning only - `tools`. -- If a tool description is over 1024 chars: - - Replace the Kiro tool definition description with a short pointer such as - `Tool documentation moved to the system prompt: .` - - Add the full description to `systemAdditions` under a deterministic heading. -- Preserve existing schema sanitization behavior from Phase 120. - -### ADD `src/adapters/kiro-wire.ts` - -Keep `src/adapters/kiro.ts` below the 500-line project limit by moving existing -wire helpers out before adding fallback logic: - -- `fingerprint()` -- `osTag()` -- `mapModelId(id)` -- `normalizeToolId(id)` - -The functions keep their current behavior exactly. This is a mechanical -extraction only. - -### MODIFY `src/adapters/kiro.ts` - -- Remove local wire helpers and import them from `src/adapters/kiro-wire.ts`. -- Import `convertKiroToolContext()` instead of only `convertKiroTools()`. -- Append `systemAdditions` to the payload system prefix. Unlike the stable - system prompt, tool documentation additions should be present whenever the - request includes tool definitions, including resumed requests, because Kiro - receives the tool specs on the current request. -- Track structured assistant tool-use IDs while building the transmitted - payload. -- If `parsed.context.tools` converts to zero Kiro tools: - - Do not emit `assistantResponseMessage.toolUses`. - - Do not emit `userInputMessageContext.toolResults`. - - Render assistant tool calls and tool results as plain text context. -- If a `toolResult` has no matching earlier assistant tool-use ID in the - transmitted payload: - - Convert it to a plain user text entry. - - Do not attach it as structured `toolResults`. -- Preserve current behavior for valid tool-call continuation payloads: matching - assistant `toolUses` remain structured and matching `toolResults` stay - adjacent. - -### MODIFY `tests/kiro-adapter.test.ts` - -Add regression coverage: - -- Long tool descriptions are not lost: Kiro tool definition has a short pointer - and the full description appears in the user/system-prefixed payload. -- No-tools fallback converts assistant tool calls and tool results to text and - emits no structured `toolUses`/`toolResults`. -- Orphaned tool results with tools present convert to text and emit no - structured `toolResults`. -- Existing resumed tool-result context test remains green. - -## Verification - -- `bun x tsc --noEmit` -- `bun test tests/kiro-adapter.test.ts` -- `wc -l src/adapters/kiro.ts src/adapters/kiro-tools.ts src/adapters/kiro-wire.ts tests/kiro-adapter.test.ts` - -## Commit - -`a63aa76 fix(kiro): harden tool fallback payloads` - -## Completion evidence - -- Implemented `convertKiroToolContext()` in `src/adapters/kiro-tools.ts`. -- Added `src/adapters/kiro-wire.ts` and `src/adapters/kiro-tool-fallback.ts` - to keep the Kiro adapter below the 500-line project limit. -- Updated `src/adapters/kiro.ts` so long tool docs are appended to the - current request prompt, unsafe tool calls/results degrade to plain text, and - matching tool-use/tool-result continuations remain structured. -- Split stream tests into `tests/kiro-stream.test.ts` and added payload - regressions in `tests/kiro-adapter.test.ts`. - -Verification: - -- `bun x tsc --noEmit` passed. -- `bun test tests/kiro-adapter.test.ts tests/kiro-stream.test.ts tests/kiro-images.test.ts tests/kiro-retry.test.ts tests/kiro-oauth.test.ts` - passed: 66 pass, 0 fail. -- Read-only Backend verifier reran the same typecheck, targeted Kiro tests, - line counts, and reported DONE. -- C-stage root test sweep passed with `bun test tests/*.test.ts`: 717 pass, - 0 fail. -- `bun test tests` is not used as the C-stage pass gate because Bun also - discovers archived `devlog/opencode-cursor/tests/**` fixtures from the - repository snapshot; that broader command currently fails in those archived - tests and is unrelated to this Kiro adapter phase. -- Line counts after the split: - - `src/adapters/kiro.ts`: 481 - - `src/adapters/kiro-wire.ts`: 50 - - `src/adapters/kiro-tool-fallback.ts`: 36 - - `src/adapters/kiro-tools.ts`: 44 - - `tests/kiro-adapter.test.ts`: 237 - - `tests/kiro-stream.test.ts`: 333 - -## Explicit non-goals - -- No truncation recovery state machine; Phase 150 owns stream/tool truncation. -- No payload-size trimming; this phase preserves long tool docs in the prompt - but does not implement history trimming. -- No new tool schema sanitization beyond Phase 120 behavior. diff --git a/devlog/_fin/143_kiro-gateway-parity/130_phase_model_catalog_resolver.md b/devlog/_fin/143_kiro-gateway-parity/130_phase_model_catalog_resolver.md deleted file mode 100644 index bfdc324fa..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/130_phase_model_catalog_resolver.md +++ /dev/null @@ -1,152 +0,0 @@ -# Phase 130 (P1) - Kiro model catalog and resolver parity - -## Trigger - -The external review still flags Kiro model drift and weak model-id handling. -The official Kiro model catalog currently includes models missing from -`src/providers/registry.ts`, and `mapModelId()` only strips a `kiro-` prefix. - -Official source checked on 2026-06-29: - -- `https://kiro.dev/docs/models` -- The page says it was updated 2026-06-19. -- Missing from the current opencodex Kiro list: Claude Opus 4.5, Claude - Sonnet 4.0, and MiniMax M2.1. -- Some Claude rows expose `Max` effort, but Codex catalog entries must remain - Codex-safe (`low`/`medium`/`high`/`xhigh`) because Codex rejects raw `max`. - -## Current state - -- `src/providers/registry.ts` owns Kiro model arrays inline. -- `src/adapters/kiro-wire.ts` owns the runtime `mapModelId()` helper. -- The registry and adapter do not share a model-normalization source of truth. -- Existing tests only cover the older Kiro model list/context table. - -## Diff plan - -### ADD `src/providers/kiro-models.ts` - -Create the canonical Kiro model metadata module: - -- `KIRO_MODELS` - - `kiro-auto` - - `claude-opus-4.8` - - `claude-opus-4.7` - - `claude-opus-4.6` - - `claude-opus-4.5` - - `claude-sonnet-4.6` - - `claude-sonnet-4.5` - - `claude-sonnet-4.0` - - `claude-haiku-4.5` - - `deepseek-3.2` - - `minimax-m2.5` - - `minimax-m2.1` - - `glm-5` - - `qwen3-coder-next` -- `KIRO_MODEL_CONTEXT_WINDOWS` - - Existing 1M/200k/128k/256k values remain. - - Add 200k entries for Claude Opus 4.5, Claude Sonnet 4.0, and MiniMax M2.1. - - Keep `kiro-auto` omitted because Kiro Auto has no fixed context window. -- `KIRO_MODEL_REASONING_EFFORTS` - - Codex-safe efforts for every static Kiro model: - `["low", "medium", "high", "xhigh"]`. - - Do not expose raw `max` in catalog metadata. -- `normalizeKiroModelId(id)` - - Trim/lowercase. - - Strip provider-ish prefixes: `kiro/` and `kiro-`. - - Map `kiro-auto` and `auto` to `auto`. - - Strip trailing date suffixes like `-20250929`. - - Strip trailing effort suffixes: `-low`, `-medium`, `-high`, `-xhigh`, - `-max`. - - Convert dashed numeric versions to dotted versions: - `4-5` -> `4.5`, `m2-1` -> `m2.1`. - - Reorder Claude family aliases: - `claude-4.5-sonnet` -> `claude-sonnet-4.5`. - - Return the normalized id for known and version-normalized ids; leave - unknown non-matching strings unchanged. - -### MODIFY `src/providers/registry.ts` - -- Remove inline Kiro model arrays/context maps. -- Import `KIRO_MODELS`, `KIRO_MODEL_CONTEXT_WINDOWS`, and - `KIRO_MODEL_REASONING_EFFORTS`. -- Keep Kiro `defaultModel: "kiro-auto"`. -- Keep catalog-visible reasoning tiers Codex-safe; no raw `max`. -- Keep Kiro image/native vision behavior out of this phase. Native image - payloads are already implemented in Phase 10; this phase only resolves model - catalog drift and model id normalization. - -### MODIFY `src/adapters/kiro-wire.ts` - -- Import `normalizeKiroModelId` from `src/providers/kiro-models.ts`. -- Change `mapModelId(id)` to use the shared normalizer: - - `kiro-auto` / `auto` -> `auto` - - otherwise return `normalizeKiroModelId(id)`. - -### MODIFY `tests/kiro-adapter.test.ts` - -Add/update regression coverage: - -- Registry contains the official missing models: - - `claude-opus-4.5` - - `claude-sonnet-4.0` - - `minimax-m2.1` -- Context windows contain the new documented entries. -- Auto remains omitted from `modelContextWindows`. -- Reasoning efforts stay Codex-safe for Kiro models and never include raw - `max`. -- `buildRequest()` maps versioned/aliased ids into Kiro payload `modelId`: - - `kiro-auto` -> `auto` - - `claude-sonnet-4-5-20250929` -> `claude-sonnet-4.5` - - `claude-4.5-sonnet-high` -> `claude-sonnet-4.5` - - `minimax-m2-1` -> `minimax-m2.1` - -### MODIFY `tests/token-estimate.test.ts` - -- Add at least one newly added Kiro model to the Kiro token-estimate ratio - table so model-list drift does not silently miss the sidecar. - -## Verification - -- `bun x tsc --noEmit` -- `bun test tests/kiro-adapter.test.ts tests/token-estimate.test.ts tests/provider-registry-parity.test.ts tests/router.test.ts` -- `wc -l src/providers/kiro-models.ts src/providers/registry.ts src/adapters/kiro-wire.ts tests/kiro-adapter.test.ts tests/token-estimate.test.ts` - -## Commit - -`fix(kiro): reconcile model catalog and aliases` - -## Explicit non-goals - -- No live/account-aware Kiro model discovery cache. The user explicitly - deprioritized multi-account/failover complexity; static official-model - reconciliation plus deterministic aliases is the scoped fix. -- No raw `max` catalog value because Codex catalog sanitization rejects it. -- No runtime API call to Kiro for model availability; account/tier/region - availability can still vary and remains a documented limitation. -- No new image-modality matrix. Phase 10 already sends images natively; public - Kiro docs do not provide a complete per-model modality table for this phase. - -## Completion evidence - -- Implemented in `0b8d95c`: - - Added `src/providers/kiro-models.ts` as the shared Kiro model metadata and - normalization owner. - - Updated `src/providers/registry.ts` to consume the shared Kiro model list, - context windows, and Codex-safe reasoning efforts. - - Updated `src/adapters/kiro-wire.ts` so runtime wire model ids use the same - normalizer as the catalog. - - Added regression coverage in `tests/kiro-adapter.test.ts` and - `tests/token-estimate.test.ts`. -- Re-verified on 2026-06-29: - - `bun x tsc --noEmit` passed. - - `bun test tests/kiro-adapter.test.ts tests/token-estimate.test.ts tests/provider-registry-parity.test.ts tests/router.test.ts tests/openai-chat-model-suffix.test.ts` - passed: 50 tests. - - The files that failed during a broad parallel `bun test tests/*.test.ts` - run all passed in isolation: - `tests/codex-routing.test.ts`, `tests/codex-account-store.test.ts`, - `tests/codex-auth-api.test.ts`, `tests/server-auth.test.ts`, and - `tests/api-usage.test.ts`. -- Check verdict: Phase 130 changes are clean. The broad parallel-suite failure - is tracked as test isolation/shared-state behavior, not as a Kiro model - resolver regression. diff --git a/devlog/_fin/143_kiro-gateway-parity/140_phase_kiro_error_mapping.md b/devlog/_fin/143_kiro-gateway-parity/140_phase_kiro_error_mapping.md deleted file mode 100644 index b0ca3ea98..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/140_phase_kiro_error_mapping.md +++ /dev/null @@ -1,127 +0,0 @@ -# Phase 140 (P1) - Kiro actionable error mapping - -## Trigger - -The parity review still flags Kiro failures as too generic. Stream exception -frames and non-2xx HTTP bodies currently become broad `upstream_error` text, so -users cannot distinguish auth expiry, rate limit, quota exhaustion, wrong -profile/region, malformed tool payloads, or unavailable models. - -Multi-account routing/failover remains out of scope. This phase only improves -single-account error fidelity and secret-safe user-facing diagnostics. - -## Current state - -- `src/adapters/kiro-errors.ts` redacts Kiro stream exception text and local - paths, but does not classify common Kiro exception families into actionable - messages. -- `src/adapters/kiro-retry.ts` retries 429/5xx and aborts safely, but final - non-2xx `Response` bodies are still raw provider bodies consumed later by - `src/server.ts`. -- `src/bridge.ts` maps every adapter stream `error` event through - `classifyError(502, "upstream_error", message)`, so stream-side Kiro errors - need recognizable message text. -- `src/errors.ts` already maps context, quota, rate limit, auth, overload, and - invalid-request categories for generic provider errors. - -## Diff plan - -### MODIFY `src/adapters/kiro-errors.ts` - -Add an actionable Kiro error normalization layer while preserving current -redaction behavior: - -- Keep `safeKiroErrorMessage(headers, payloadText)` as the stream-frame API. -- Add `safeKiroHttpErrorMessage(status, headers, payloadText)` for non-2xx HTTP - responses. -- Parse JSON bodies for `__type`, `code`, `error`, `name`, `message`, - `Message`, and `errorMessage`. -- Redact secrets and local paths before returning any message. -- Emit category-specific text that `classifyError()` can recognize: - - throttling/rate-limit types -> `Kiro rate limit exceeded ...` - - auth/access denied/expired token types -> `Kiro authentication failed ...` - - quota exhaustion text -> `Kiro quota exhausted ...` - - validation/profile/region/model/tool schema issues -> `Kiro invalid request ...` - - overloaded/temporary server failures -> `Kiro server overloaded ...` - - unknown -> `Kiro upstream error ...` - -### MODIFY `src/adapters/kiro-retry.ts` - -- Import `safeKiroHttpErrorMessage`. -- When returning the final non-OK response, read a clone/body safely and replace - it with a sanitized actionable text response that preserves status and - headers. -- Keep retry behavior unchanged for retryable 429/5xx before the final attempt. -- Keep abort semantics unchanged. - -### MODIFY `src/errors.ts` - -Extend generic classification keywords so stream-side Kiro messages map -correctly despite bridge status `502`: - -- `throttlingexception`, `rate limited`, and `rate limit exceeded` -> - `rate_limit_error/rate_limit_exceeded`. -- `authentication failed`, `access denied`, `expired token`, - `unauthorizedexception`, `unrecognizedclientexception` -> - `authentication_error/invalid_api_key`. -- `quota exhausted` and non-per-minute quota exhaustion wording -> - `insufficient_quota/insufficient_quota`. -- `validationexception`, `invalid request`, `model unavailable`, `model not - found`, `profile arn`, `region` -> `invalid_request_error`. - -Ordering must preserve the existing transient-429 test: request-per-minute -quota wording remains `rate_limit_exceeded`, not `insufficient_quota`. - -### MODIFY tests - -- `tests/kiro-stream.test.ts` - - Add stream exception cases proving Kiro throttling/auth/validation/model - errors produce actionable redacted messages. - - Bridge classification is covered by `tests/error-fidelity.test.ts`. -- `tests/kiro-retry.test.ts` - - Add final non-OK response cases: - - 403 auth body becomes sanitized Kiro authentication text. - - 400 validation/model body becomes Kiro invalid request text. - - final 429 still returns status 429 and sanitized rate-limit text. -- `tests/error-fidelity.test.ts` - - Add `classifyError()` assertions for Kiro stream messages and keep existing - transient-429 quota behavior. - -## Verification - -- `bun x tsc --noEmit` -- `bun test tests/kiro-stream.test.ts tests/kiro-retry.test.ts tests/error-fidelity.test.ts tests/adapter-error-inline.test.ts` -- `wc -l src/adapters/kiro-errors.ts src/adapters/kiro-retry.ts src/errors.ts tests/kiro-stream.test.ts tests/kiro-retry.test.ts tests/error-fidelity.test.ts` - -## Commit - -`fix(kiro): map upstream failures to actionable errors` - -## Explicit non-goals - -- No multi-account circuit breaker or failover. -- No live Kiro account smoke test; use deterministic wire fixtures. -- No new adapter event schema. This phase works with the existing - `AdapterEvent.error` string and shared bridge classifier. - -## Completion evidence - -- Implemented in `a038784`: - - `src/adapters/kiro-errors.ts` now keeps `safeKiroErrorMessage()` and adds - `safeKiroHttpErrorMessage()` with Kiro-specific redacted category prefixes. - - `src/adapters/kiro-retry.ts` now replaces final non-OK Kiro HTTP bodies - with sanitized actionable text while preserving status, retry count, and - abort behavior. - - `src/errors.ts` now recognizes Kiro rate-limit, auth, quota, validation, - model, and region messages before the generic `status >= 500` fallback. - - Regression coverage was added in `tests/kiro-stream.test.ts`, - `tests/kiro-retry.test.ts`, and `tests/error-fidelity.test.ts`. -- Local verification: - - `bun x tsc --noEmit` passed. - - `bun test tests/kiro-stream.test.ts tests/kiro-retry.test.ts tests/error-fidelity.test.ts tests/adapter-error-inline.test.ts` - passed: 36 tests. - - Line counts stayed under 500 for all touched files. -- Independent verifier: - - Backend verifier reported DONE. - - It reran `bun x tsc --noEmit` and the same four-file target suite, with - 36 pass / 0 fail, and confirmed touched file line counts under 500. diff --git a/devlog/_fin/143_kiro-gateway-parity/150_phase_truncation_detection_recovery.md b/devlog/_fin/143_kiro-gateway-parity/150_phase_truncation_detection_recovery.md deleted file mode 100644 index ceca5e9e6..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/150_phase_truncation_detection_recovery.md +++ /dev/null @@ -1,159 +0,0 @@ -# Phase 150 (P2) - Kiro truncation detection and recovery - -## Trigger - -The original parity map and external review both flag Kiro truncation recovery -as missing. Current `parseKiroStream()` closes an open tool call and emits -`done` when the eventstream ends, even if the tool input never received a stop -event. That can turn an upstream cut-off into a successful Codex tool call with -partial or invalid JSON. - -`kiro-gateway` has a broader truncation recovery subsystem. opencodex should at -least stop silently completing truncated Kiro tool calls and surface a clear, -redacted upstream truncation failure. - -## Current state - -- `src/adapters/kiro.ts` parses Kiro event JSON inline. -- A Kiro tool call is emitted as soon as a `name`/`input` event arrives. -- At stream EOF, `parseKiroStream()` currently emits `tool_call_end` for any - open tool and then emits `done`. -- `bridge.ts` already treats an adapter `error` event as `response.failed`, and - treats a generator EOF without `done`/`error` as `response.incomplete`. -- `bridge.ts` sets its upstream activity flag once per adapter event before the - switch dispatch, so an internal non-visual event can keep the stream alive - without sending Codex-visible output. -- Kiro returns no authoritative usage frame, so ordinary text-only EOF must - continue to be treated as normal completion. - -## Diff plan - -### ADD `src/adapters/kiro-truncation.ts` - -Create a small helper module: - -- `kiroTruncationReason(parsed: Record): string | undefined` - - Detect explicit truncation markers in event JSON: - `finish_reason`, `finishReason`, `stop_reason`, `stopReason`, - `completionReason`, `reason`, or `truncated: true`. - - Treat string values containing `length`, `max_token`, `max-tokens`, - `truncate`, `truncated`, `incomplete`, or `context_length` as truncation. -- `isCompleteKiroToolInput(input: string): boolean` - - Treat empty input as complete only after a real Kiro `stop` event. - - Parse non-empty accumulated input as JSON and require an object/array root. -- `kiroTruncationErrorMessage(reason?: string): string` - - Return a user-facing, redacted message: - `Kiro response truncated upstream before the tool call completed...` - -### MODIFY `src/adapters/kiro.ts` - -- Import the helper module. -- Extend `ParsedKiroEvent` with `type: "truncation"`. -- In `parseKiroEvent()`, return a truncation event when - `kiroTruncationReason(parsed)` detects an explicit marker. -- Change `parseKiroStream()` tool handling: - - Buffer Kiro tool starts/input chunks internally. - - Emit `tool_call_start`, `tool_call_delta`, and `tool_call_end` only after a - real Kiro `stop` event. - - Preserve chunk boundaries when flushing a completed tool call. - - Ignore duplicate `name` starts for the same open tool before input arrives. - - Yield internal `{ type: "heartbeat" }` events while buffering Kiro tool - start/input events so the streaming bridge does not falsely stall-timeout - during long but active tool-call generation. - - On Kiro eventstream `exception`/`error` frames, discard any buffered - unflushed tool call and emit only the upstream error. Do not yield - `tool_call_end`, because no matching `tool_call_start` has been sent to the - bridge under the buffering model. - - If a new tool/content/truncation/EOF arrives while a tool is still open - without `stop`, emit `error` with `kiroTruncationErrorMessage()` and return. - - Do not emit `done` after a truncation error. -- Keep normal text-only EOF as `done` because Kiro has no usage terminal. -- Keep stream exception/error frame behavior from Phase 70 unchanged. - -### MODIFY `src/types.ts` - -- Add a non-visual adapter event: - `{ type: "heartbeat" }`. -- This event is internal to the proxy. It has no Responses API output item and - carries no user-visible content. - -### BRIDGE BEHAVIOR (no code change) - -- Do not modify `src/bridge.ts`; it is already over the repo line limit. -- `bridgeToResponsesSSE()` already sets `activity = true` before dispatching on - `event.type`, so `heartbeat` resets stall tracking even with no switch case. -- `buildResponseJSON()` has no default side effect for unknown event variants, - so `heartbeat` is naturally non-visual. Add tests to lock this behavior. - -### MODIFY `tests/kiro-stream.test.ts` - -Add regression tests: - -- Normal completed tool call still emits start/delta/end/done in the same final - event order, even though Kiro events are buffered until stop. -- Tool input stream ending mid-JSON without stop emits a clear truncation error, - no `done`, and no partial `tool_call_delta`. -- Tool input stream ending with valid JSON but without stop is still treated as - truncation because the upstream did not complete the tool call. -- Explicit Kiro length/truncation marker emits the truncation error and no - `done`. -- Duplicate tool `name` events before input do not create duplicate tool calls. -- Buffered tool input emits internal `heartbeat` events that tests can observe, - but bridge tests prove they are not Codex-visible. -- Update the existing `exception mid-stream closes an open tool call then stops` - regression to the new fail-closed behavior: if the tool was buffered and not - stopped, expect only heartbeat/internal activity plus the upstream error, with - no client-facing `tool_call_start`/`tool_call_end`. - -### MODIFY `tests/bridge.test.ts` - -- Add a regression proving `heartbeat` events do not create SSE output items, - do not change non-streaming JSON output, and still allow surrounding normal - events to complete. - -## Verification - -- `bun x tsc --noEmit` -- `bun test tests/kiro-stream.test.ts tests/bridge.test.ts tests/error-fidelity.test.ts` -- `wc -l src/adapters/kiro.ts src/adapters/kiro-events.ts src/adapters/kiro-truncation.ts src/types.ts tests/kiro-stream.test.ts tests/bridge.test.ts` - -## Commit - -`fix(kiro): surface truncated tool-call streams` - -## Explicit non-goals - -- No full gateway-style persistent recovery memory. -- No attempt to classify ordinary text EOF as truncation without an explicit - marker; Kiro has no terminal usage frame, so that would create false - positives. -- No user-visible heartbeat or progress output. The new `heartbeat` event is - internal and ignored by response builders. - -## Completion evidence - -- Implemented in `c3b10c9`: - - Added `src/adapters/kiro-events.ts` for Kiro event JSON parsing and - explicit truncation marker detection. - - Added `src/adapters/kiro-truncation.ts` for truncation reason detection, - tool-input completeness checks, and user-safe truncation messages. - - Updated `src/adapters/kiro.ts` to buffer tool starts/input until a real - stop event, emit internal `heartbeat` events while buffering, and fail - closed on EOF, exception/error, explicit truncation markers, or content - before tool stop. - - Added internal `{ type: "heartbeat" }` to `AdapterEvent`. - - Added regression tests in `tests/kiro-stream.test.ts` and - `tests/bridge.test.ts`. -- Local verification: - - `bun x tsc --noEmit` passed. - - `bun test tests/kiro-stream.test.ts tests/bridge.test.ts tests/error-fidelity.test.ts` - passed: 42 tests. - - Line counts stayed under 500 for touched files: - `kiro.ts` 478, `kiro-events.ts` 42, `kiro-truncation.ts` 33, - `types.ts` 347, `kiro-stream.test.ts` 413, `bridge.test.ts` 213. -- Independent verifier: - - Backend verifier reported DONE with the same typecheck, target tests, and - line-count evidence. - - It confirmed `src/bridge.ts` was intentionally not modified; heartbeat is - non-visual because existing bridge code marks activity before the switch and - has no output-producing heartbeat case. diff --git a/devlog/_fin/143_kiro-gateway-parity/160_phase_usage_estimated_diagnostics.md b/devlog/_fin/143_kiro-gateway-parity/160_phase_usage_estimated_diagnostics.md deleted file mode 100644 index 63d389df9..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/160_phase_usage_estimated_diagnostics.md +++ /dev/null @@ -1,165 +0,0 @@ -# Phase 160/170 (P2) - Kiro estimated usage tagging and redacted diagnostics - -## Trigger - -Kiro/CodeWhisperer does not return authoritative token usage. opencodex now -estimates usage so Codex display and auto-compact work, but internal usage logs -still classify those numbers as `reported`. The parity review also asks for -redacted diagnostics that make Kiro auth/region/model/debugging easier without -leaking prompts, tokens, profile ARNs, or local paths. - -## Current state - -- `src/types.ts` `OcxUsage` has token fields but no estimated marker. -- `src/usage-log.ts` already has `UsageStatus = "estimated"`, and - `usage-summary.ts` already counts `estimatedRequests`, but - `usageStatusForFinalLog()` currently returns `reported` for any usage object. -- Kiro emits heuristic usage from `parseKiroStream()` and `parseResponse()` via - `done.usage`. -- Request logging extracts usage from the Responses SSE/JSON that the bridge - sends downstream. That means adapter-only metadata is not preserved unless - request-log finalization marks Kiro usage as estimated by provider. -- `src/debug.ts` has opt-in `OCX_DEBUG_FRAMES=1` diagnostics for dropped frames, - but no structured redacted provider breadcrumb helper. -- `src/usage-debug.ts` already writes redacted JSONL records behind - `OPENCODEX_USAGE_DEBUG=1`. - -## Diff plan - -### MODIFY `src/types.ts` - -- Add optional `estimated?: boolean` to `OcxUsage`. -- This is an internal metadata flag; bridge response usage should keep the - OpenAI-compatible token shape. - -### MODIFY `src/usage-log.ts` - -- Add `usageForFinalLog(provider: string, usage: OcxUsage | undefined)`: - - returns `undefined` when no usage exists. - - returns `{ ...usage, estimated: true }` for provider `kiro`. - - preserves an already-estimated usage object for other providers. -- Change `usageStatusForFinalLog()`: - - no usage -> `unreported` - - usage.estimated -> `estimated` - - otherwise -> `reported` -- Preserve `estimated: true` in `normalizeUsageValue()` while continuing to - strip unknown runtime fields. - -### MODIFY `src/server.ts` - -- In `addFinalRequestLog()`, derive `const finalUsage = - usageForFinalLog(logCtx.provider, logCtx.usage)`. -- Use `finalUsage` for: - - `usageStatusForFinalLog(finalUsage)` - - `usageTotalTokens(finalUsage)` - - persisted/request-log `usage` - - `usage-debug` `extractedUsage` -- Do not expose the estimated flag through `bridge.ts` Responses usage. - -### MODIFY `src/adapters/kiro.ts` - -- Keep Kiro `done.usage` values heuristic, but add `estimated: true` for direct - adapter consumers/tests. -- Add an opt-in redacted diagnostic breadcrumb after `buildKiroPayload()`: - - adapter/provider: `kiro` - - auth/runtime region - - requested model id - - body byte length - - message/tool counts - - booleans only for profile ARN presence and previous response state -- Do not log raw request body, prompt content, image bytes, bearer tokens, or - profile ARN values. - -### MODIFY `src/debug.ts` - -- Add `debugProviderDiagnostic(adapter, event, details)` behind existing - `OCX_DEBUG_FRAMES=1`. -- Redact via `redactSecrets()` before JSON serialization. -- Keep failure-safe behavior: diagnostics must never throw back into request - handling. - -### MODIFY tests - -- `tests/usage-log.test.ts` - - `usageStatusForFinalLog({ estimated: true })` returns `estimated`. - - `usageForFinalLog("kiro", usage)` marks usage estimated. - - persisted usage JSONL preserves only the boolean `estimated` metadata and - still strips unknown fields. -- `tests/request-log.test.ts` - - Kiro deferred SSE logging records `usageStatus: "estimated"` and usage with - `estimated: true`, while the SSE usage payload itself stays standard. -- `tests/kiro-stream.test.ts` - - Kiro done usage includes `estimated: true`. -- `tests/debug.test.ts` - - provider diagnostic helper stays silent by default and redacts secrets when - enabled. -- `tests/usage-debug.test.ts` - - usage-debug record can store `extractedUsage.estimated === true` without - leaking secret fields. - -## Verification - -- `bun x tsc --noEmit` -- `bun test tests/usage-log.test.ts tests/request-log.test.ts tests/kiro-stream.test.ts tests/debug.test.ts tests/usage-debug.test.ts tests/usage-summary.test.ts` -- `wc -l src/types.ts src/usage-log.ts src/server.ts src/adapters/kiro.ts src/debug.ts tests/usage-log.test.ts tests/request-log.test.ts tests/kiro-stream.test.ts tests/debug.test.ts tests/usage-debug.test.ts` - -## Commit - -`fix(kiro): mark heuristic usage as estimated` - -## Explicit non-goals - -- No raw prompt or full Kiro payload logging. -- No full raw AWS eventstream frame capture. -- No public Responses API schema change for usage; estimated status is for - opencodex logs/debugging. - -## Completion evidence - -- Implementation commit: `e50ca23 fix(kiro): mark heuristic usage as estimated`. -- `src/types.ts` now carries internal `OcxUsage.estimated`. -- `src/usage-log.ts` marks provider `kiro` usage as estimated at final log time, - preserves only the boolean metadata in JSONL, and still strips unknown runtime - fields. -- `src/server.ts` uses the final normalized usage for request logs and - `usage-debug` extracted usage, without changing downstream Responses usage. -- `src/adapters/kiro.ts` emits `done.usage.estimated = true` and logs only - opt-in redacted request breadcrumbs through `debugProviderDiagnostic()`. -- `src/debug.ts` keeps provider diagnostics behind `OCX_DEBUG_FRAMES=1`, redacts - secrets/profile ARNs/tokens, and swallows diagnostic failures. -- Local verification passed: - - `bun x tsc --noEmit` - - `bun test tests/usage-log.test.ts tests/request-log.test.ts tests/kiro-stream.test.ts tests/debug.test.ts tests/usage-debug.test.ts tests/usage-summary.test.ts` - - `65 pass, 0 fail` -- Backend verifier returned `DONE`, confirmed public Responses usage shape stays - unchanged, request logs mark Kiro usage `estimated`, usage-debug preserves - `estimated`, provider diagnostics are opt-in/redacted, and `src/adapters/kiro.ts` - is 489 lines. - -## Follow-up: Request Logs should show full-context estimates - -User observed that Kiro request log rows showed only ~100-800 tokens while -ChatGPT rows showed large context-sized totals. Root cause: Kiro's downstream -Responses usage intentionally reports only current-turn input delta so Codex's -own cumulative session accounting does not double count old history. The GUI -Request Logs, however, should answer a different question: approximate context -size/cost for that request. - -Patch plan: - -- Keep public Responses/SSE Kiro usage unchanged (`input_tokens` remains - current-turn delta). -- Add adapter-internal `AdapterRequest.usageLog.inputTokens`. -- Have Kiro fill `usageLog.inputTokens` with a full Codex-context estimate: - system prompt, tools, user/developer messages, assistant text/tool calls, - and tool results. -- Have server request-log finalization use that internal estimate only for - persisted/logged usage totals. -- Keep `estimated: true` on Kiro logs. - -Verification: - -- `bun x tsc --noEmit` -- `bun test tests/kiro-stream.test.ts tests/request-log.test.ts tests/usage-log.test.ts tests/usage-summary.test.ts tests/usage-debug.test.ts` -- `63 pass, 0 fail` -- `src/adapters/kiro.ts` is 494 lines. diff --git a/devlog/_fin/143_kiro-gateway-parity/170_phase_final_gap_audit_push.md b/devlog/_fin/143_kiro-gateway-parity/170_phase_final_gap_audit_push.md deleted file mode 100644 index 53335d4ef..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/170_phase_final_gap_audit_push.md +++ /dev/null @@ -1,145 +0,0 @@ -# Phase 170 - Final Kiro parity gap audit and push - -## Trigger - -All planned Kiro/CodeWhisperer parity hardening slices have landed. The user -asked to continue until final completion, verify whether any functional gaps -remain, and push the branch after the final audit passes. Multi-account failover -remains explicitly out of scope. - -## Plain-language plan - -Review the completed Kiro work against the GPT-Pro gap list, confirm that each -in-scope item has code and test evidence, run the strongest practical local -verification, get a read-only employee audit, then push `feat/kiro-on-dev`. -Do not add new feature code unless the audit finds a blocking in-scope gap. - -## Gap checklist - -### In scope and expected closed - -- Native Kiro image payloads via `userInputMessage.images`. -- Terminal handling for Kiro eventstream exception/error frames. -- Kiro HTTP retry/backoff and first-token retry boundaries. -- OAuth refresh singleflight, SQLite reload/recovery, broader single-account - auth inputs, and API-region/runtime-region split. -- Resume/tool-result payload correctness. -- AWS eventstream decoder bounds/fuzz hardening. -- Tool schema sanitization and tool fallback hardening for long descriptions, - orphaned tool results, and no-tools payloads. -- Model list/resolver updates, versioned aliases, and max effort metadata. -- Actionable Kiro upstream error mapping. -- Thinking-tag exposure fix. -- Truncation detection/fail-closed stream recovery. -- Estimated usage tagging and redacted Kiro diagnostics. - -### Explicitly out of scope - -- Multi-account Kiro failover, sticky account selection, and circuit breakers. -- Live validation against a real Kiro account when local credentials/session are - not available. -- ChatGPT web-session follow-up if the linked ChatGPT page requires login. - -## Diff plan - -### NEW `devlog/_plan/143_kiro-gateway-parity/170_phase_final_gap_audit_push.md` - -This document. - -### MODIFY durable memory - -Save a short session outcome to: - -- `structured/episodes/live/2026-06-29.md` - -### NO CODE CHANGES unless audit fails - -If the final audit finds an in-scope blocker, return to a new PABCD fix phase -instead of pushing. - -## Verification plan - -- `bun x tsc --noEmit` -- `bun test tests/kiro-images.test.ts tests/kiro-stream.test.ts tests/kiro-retry.test.ts tests/kiro-oauth.test.ts tests/eventstream-decoder.test.ts tests/kiro-adapter.test.ts tests/error-fidelity.test.ts tests/usage-log.test.ts tests/request-log.test.ts tests/debug.test.ts tests/usage-debug.test.ts tests/usage-summary.test.ts` -- `git status --short --branch` -- Read-only Backend verifier to compare completed commits/plans against the - GPT-Pro P0/P1/P2 list. -- Push only after verifier status is `DONE`. - -## Commit/push plan - -- Commit this plan/evidence as `docs(kiro): plan final parity audit`. -- If verification passes and no blocker remains, push: - - `git push origin feat/kiro-on-dev` - -## Local final audit evidence - -Local audit after the Kiro Request Logs full-context usage follow-up found no -new in-scope functional gap against the GPT-Pro P0/P1/P2 checklist. The -remaining known non-parity is intentionally out of scope: Kiro multi-account -failover / sticky account selection / circuit breakers, and live Kiro account -validation when no live account is available. - -Verification passed: - -- `bun x tsc --noEmit` -- `bun test tests/kiro-images.test.ts tests/kiro-stream.test.ts tests/kiro-retry.test.ts tests/kiro-oauth.test.ts tests/eventstream-decoder.test.ts tests/kiro-adapter.test.ts tests/error-fidelity.test.ts tests/usage-log.test.ts tests/request-log.test.ts tests/debug.test.ts tests/usage-debug.test.ts tests/usage-summary.test.ts` -- `138 pass, 0 fail` -- Kiro split files are below the 500-line limit; `src/adapters/kiro.ts` is 494 - lines. - -Independent Backend employee verification could not be completed in this -session. Two dispatch attempts returned `Not logged in - Please run /login`. -Therefore no employee `DONE` verdict is claimed. - -Push was not performed in this audit pass because the verification plan says to -push only after employee `DONE`, and the current user turn did not freshly -authorize `git push`. - -## Post-audit follow-up - -Later live Codex Desktop/Kiro Computer Use testing found additional tool -fidelity issues that were not covered by the local final-audit matrix: -namespaced MCP/Computer Use tool names needed full wire-name preservation, -complete JSON tool calls could reach EOF without a Kiro `stop` frame, and -Computer Use screenshot images were dropped from Kiro tool-result -continuations. The follow-up patch and evidence are recorded below. - -Implementation summary: - -- `src/adapters/kiro-tools.ts` now advertises full - `namespacedToolName(namespace, name)` wire names instead of bare tool names, - and no longer truncates tool specification names to 64 characters. -- `src/adapters/kiro.ts` replays assistant `toolUses` with the same full wire - name, so Kiro history matches the current tool definitions and the bridge can - restore MCP namespaces from `toolNsMap`. -- `parseKiroStream()` still fails closed for incomplete tool JSON at EOF, but - recovers complete JSON tool calls when Kiro omits the terminal `stop` frame. -- `src/responses/schema.ts` now explicitly accepts `input_image` blocks in - `function_call_output.output`. -- Kiro tool-result continuations now preserve screenshots by extracting image - parts from structured `toolResult` content and attaching them to the carrier - `userInputMessage.images` alongside structured `toolResults`. - -Regression coverage added: - -- Full MCP/Computer Use wire names are advertised and replayed. -- Long namespaced tool names are preserved rather than truncated. -- Complete JSON EOF recovery emits a normal tool call; partial JSON EOF still - emits the truncation error. -- Responses `function_call_output` image blocks are preserved. -- Kiro carrier user messages receive tool-result screenshot images. - -Local verification passed: - -- `bun x tsc --noEmit` -- `bun test tests/kiro-adapter.test.ts tests/kiro-stream.test.ts tests/bridge.test.ts tests/responses-parser.test.ts` -- Result: 62 pass, 0 fail. -- `src/adapters/kiro.ts` remains at the 500-line limit. - -Remaining work after the local patch is runtime-only: rebuild/restart the local -proxy/app so the patched adapter is loaded, run one live Kiro Computer Use smoke -(`list_apps`, `get_app_state` on Chrome/Finder), and confirm the next Kiro turn -can reason over the screenshot. Whether the conversation UI should render a -visible screenshot thumbnail remains a Codex Desktop UI-rendering question; the -Kiro model-context mapping is now covered locally. diff --git a/devlog/_fin/143_kiro-gateway-parity/20_phase2_retry_backoff.md b/devlog/_fin/143_kiro-gateway-parity/20_phase2_retry_backoff.md deleted file mode 100644 index b32169d21..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/20_phase2_retry_backoff.md +++ /dev/null @@ -1,21 +0,0 @@ -# Phase 20 — Retry / backoff on transient upstream failures - -## Problem -Gateway (network_errors.py) classifies 403/429/5xx as retryable and retries. -opencodex kiro has no retry: a transient 429/503 fails the whole turn. - -## Plan (to be finalized in this phase's P) -- Decide layer: adapter-level wrapper around the upstream fetch vs shared - transport. Prefer the smallest correct layer that the kiro path already owns. -- Classify retryable: 429, 500, 502, 503, 504; honor Retry-After when present; - exponential backoff with jitter; bounded attempts (e.g. 3). -- Do NOT retry non-idempotent partial streams once bytes have been yielded; - retry only pre-first-byte failures to avoid duplicate output. - -## Tests -- 429 then 200 -> succeeds after one retry. -- exhausted retries -> surfaces last error. -- post-first-byte error -> NOT retried (no duplicate stream). - -## Commit -feat(kiro): retry transient 429/5xx with bounded backoff (gateway parity) diff --git a/devlog/_fin/143_kiro-gateway-parity/30_phase3_payload_guard.md b/devlog/_fin/143_kiro-gateway-parity/30_phase3_payload_guard.md deleted file mode 100644 index 4032f35da..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/30_phase3_payload_guard.md +++ /dev/null @@ -1,22 +0,0 @@ -# Phase 30 — Payload size guard + history trimming - -## Problem -Gateway (payload_guards.py) trims oldest history pairs to fit a byte cap and -repairs orphaned tool results. opencodex kiro builds the full history with no -size guard; very long sessions can exceed Kiro's request limit and hard-fail. - -## Plan (finalized in this phase's P) -- After buildKiroPayload, measure serialized byte size; if over cap, trim oldest - history entries in user/assistant pairs, keeping >=2 entries and the current - message intact. -- Preserve toolResult adjacency invariants the adapter already enforces (no - orphaned toolResults after trim). -- Cap value: source from Kiro's documented/observed limit; make it a named const. - -## Tests -- oversized history -> trimmed under cap, current message preserved. -- trim never orphans a toolResult (alternation invariant holds). -- under-cap payload -> untouched (byte-identical). - -## Commit -feat(kiro): trim oldest history to fit payload byte cap (gateway parity) diff --git a/devlog/_fin/143_kiro-gateway-parity/40_phase4_thinking_parseback.md b/devlog/_fin/143_kiro-gateway-parity/40_phase4_thinking_parseback.md deleted file mode 100644 index 74be3d92e..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/40_phase4_thinking_parseback.md +++ /dev/null @@ -1,24 +0,0 @@ -# Phase 40 — Response-side thinking block parse-back - -## Problem -opencodex injects request-side thinking tags (kiro.ts 196-209) but does not -parse // blocks OUT of the response stream. Gateway -(thinking_parser.py) runs an FSM that detects a leading thinking block and emits -it as reasoning_content separate from visible text. - -## Plan (finalized in this phase's P) -- Add a streaming FSM in the kiro parse path: detect a thinking block ONLY at the - start of the response; buffer until close tag; emit AdapterEvent reasoning - deltas (matching how other opencodex adapters surface reasoning), then switch - to normal text_delta for the remainder. -- Handle tag split across SSE chunks (FSM buffers partial tags). -- Once closed, no further thinking detection (later literal tags pass through). - -## Tests -- "plananswer" -> reasoning="plan", text="answer". -- tag split across chunks -> still parsed. -- no thinking block -> all text, no reasoning event. -- thinking tag NOT at start -> treated as text. - -## Commit -feat(kiro): surface response blocks as reasoning (gateway parity) diff --git a/devlog/_fin/143_kiro-gateway-parity/50_phase5_model_normalization.md b/devlog/_fin/143_kiro-gateway-parity/50_phase5_model_normalization.md deleted file mode 100644 index 8ff7c9831..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/50_phase5_model_normalization.md +++ /dev/null @@ -1,24 +0,0 @@ -# Phase 50 — Smart model-name normalization - -## Problem -mapModelId (kiro.ts 56-58) only strips a "kiro-" prefix. Gateway -(model_resolver.py normalize_model_name) maps versioned/dashed slugs -(claude-sonnet-4-5-20250929, claude-3-7-sonnet, claude-4.5-sonnet-high) to -canonical Kiro model ids. opencodex can mis-route versioned slugs. - -## Plan (finalized in this phase's P) -- Add normalizeKiroModelId covering: date-suffix stripping (-YYYYMMDD), - dashed-version -> dotted (4-5 -> 4.5), family reordering where gateway does it, - and effort suffix stripping (-high/-low) since effort is a separate field here. -- Keep "auto"/"kiro-auto" handling. Map to the registry's canonical KIRO_MODELS id. -- Pure function + table-driven; no network. - -## Tests (mirror gateway docstring cases) -- claude-sonnet-4-5-20250929 -> claude-sonnet-4.5 -- claude-3-7-sonnet -> claude-3.7-sonnet (if in catalog) else passthrough -- claude-4.5-sonnet-high -> claude-sonnet-4.5 (effort stripped) -- already-canonical id -> unchanged -- auto -> auto - -## Commit -feat(kiro): normalize versioned model slugs to canonical ids (gateway parity) diff --git a/devlog/_fin/143_kiro-gateway-parity/60_phase6_truncation_recovery.md b/devlog/_fin/143_kiro-gateway-parity/60_phase6_truncation_recovery.md deleted file mode 100644 index 5598f7c09..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/60_phase6_truncation_recovery.md +++ /dev/null @@ -1,22 +0,0 @@ -# Phase 60 — Truncation detection / recovery - -## Problem -Gateway (truncation_recovery.py, Issue #56) detects when Kiro truncates large -tool-call payloads or content mid-stream and injects a synthetic message so the -model adapts. opencodex kiro swallows truncation silently. - -## Plan (finalized in this phase's P) -- Detect truncation signals in the eventstream (incomplete tool_input JSON at - stream end, or an explicit truncation marker if CW sends one). -- On detection, emit a clear AdapterEvent (text or error annotation) so the - turn surfaces "(response truncated upstream)" rather than producing invalid - partial tool JSON. -- Guard: only activate when truncation is actually detected (no false positives - on normal completion). - -## Tests -- tool_input stream ends mid-JSON -> truncation surfaced, no invalid tool call. -- normal completion -> no truncation event. - -## Commit -feat(kiro): detect and surface upstream truncation (gateway parity) diff --git a/devlog/_fin/143_kiro-gateway-parity/70_phase_stream_error_terminal.md b/devlog/_fin/143_kiro-gateway-parity/70_phase_stream_error_terminal.md deleted file mode 100644 index 2a4cbf5a1..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/70_phase_stream_error_terminal.md +++ /dev/null @@ -1,43 +0,0 @@ -# Phase 70 (P0-1) - Stream exception/error is terminal - -## Problem -parseKiroStream (kiro.ts ~407-414) on a CW eventstream frame with -:message-type == "exception"|"error" yields an `error` AdapterEvent then -`continue`s the loop. When the loop ends it ALSO yields `done` with usage. - -Downstream (bridge.ts 351-377): `done` -> response.completed; `error` -> -response.failed. Both set terminated=true. So whichever the bridge sees FIRST -wins, but the generator keeps yielding post-exception content and a trailing -`done`. A failed upstream call can leak partial content and a success-shaped -`done`, and the generator wastes work after termination. - -## Fix -On an exception/error frame: yield the `error` event and `return` immediately -(terminal). Do not parse further frames, do not emit `done`. - -### MODIFY src/adapters/kiro.ts (in parseKiroStream loop) -Before: - if (mt === "exception" || mt === "error") { - yield { type: "error", message: ... }; - continue; - } -After: - if (mt === "exception" || mt === "error") { - if (open) yield { type: "tool_call_end" }; // close any dangling tool call - yield { type: "error", message: ... }; - return; // terminal: no further frames, no done - } - -Rationale for closing `open`: keeps tool-call bracketing balanced for the -bridge's closeCurrentToolCall path even on the error route. - -## Tests (tests/kiro-adapter.test.ts, add cases) -- exception frame mid-stream -> yields error, NO done after it, no further text. -- exception frame before any content -> yields error only, no done. -- (existing) exception-only test updated to assert absence of trailing done. - -## Verify -bun x tsc --noEmit + bun test tests/kiro-adapter.test.ts - -## Commit -fix(kiro): treat upstream exception/error frames as terminal (no trailing done) diff --git a/devlog/_fin/143_kiro-gateway-parity/80_phase_http_retry_backoff.md b/devlog/_fin/143_kiro-gateway-parity/80_phase_http_retry_backoff.md deleted file mode 100644 index d07753a17..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/80_phase_http_retry_backoff.md +++ /dev/null @@ -1,114 +0,0 @@ -# Phase 80 (P0-2) - Kiro HTTP retry/backoff - -## Scope - -Implement the retry surface that can be closed without changing credential -ownership: - -- retry connect/header-timeout errors before any response body exists -- retry HTTP 429/500/502/503/504 before any body is parsed -- honor `Retry-After` when present; otherwise exponential backoff + bounded jitter -- preserve client abort propagation -- apply to both the normal Kiro call path and Kiro `parseResponse` calls used by - the web-search sidecar loop - -Defer 401/403 refresh-once to Phase 90, because provider.apiKey is resolved -outside the adapter and needs OAuth singleflight/reload semantics to be correct. - -## Design - -The current `ProviderAdapter` owns request construction and stream parsing, but -`server.ts` owns fetch. Kiro-specific retry therefore needs one small adapter -extension: - -```ts -fetchResponse?(request, ctx): Promise -``` - -`server.ts` and `web-search/loop.ts` should use `adapter.fetchResponse(...)` -when present; otherwise keep the existing fetch path. This keeps Kiro retry -adapter-local and avoids changing other providers. - -## File changes - -### MODIFY src/adapters/base.ts - -Add exported `AdapterRequest` and optional `fetchResponse`: - -```ts -export interface AdapterRequest { - url: string; - method: string; - headers: Record; - body: string; -} - -export interface AdapterFetchContext { - abortSignal?: AbortSignal; - timeoutMs?: number; -} - -fetchResponse?(request: AdapterRequest, ctx?: AdapterFetchContext): Promise; -``` - -### MODIFY src/adapters/kiro.ts - -Add helper functions: - -- `kiroSleep(ms, signal)` abort-aware delay -- `retryAfterMs(headers)` parse seconds or HTTP date -- `isRetryableKiroStatus(status)` = 429/500/502/503/504 -- `fetchKiroWithRetry(request, ctx)`: - - max attempts: 3 - - per-attempt timeout: ctx.timeoutMs ?? 30_000 - - if fetch throws due to caller abort: rethrow - - if fetch throws connect/header timeout/network error: retry if attempts remain - - if response status retryable: cancel/read body safely, wait, retry - - if final attempt or non-retryable: return response/throw last error - -Expose it through `createKiroAdapter(...).fetchResponse`. - -### MODIFY src/server.ts - -In the routed non-passthrough path, replace direct `fetchWithHeaderTimeout(...)` -with: - -```ts -upstreamResponse = adapter.fetchResponse - ? await adapter.fetchResponse(request, { abortSignal: upstream.signal, timeoutMs: connectMs }) - : await fetchWithHeaderTimeout(...); -``` - -Keep the existing catch/error mapping and `linkAbortSignal` behavior. - -### MODIFY src/web-search/loop.ts - -Use `adapter.fetchResponse?.(request, { abortSignal })` before falling back to -direct `fetch`, so Kiro sidecar iterations get the same transient retry. - -## Tests - -### NEW tests/kiro-retry.test.ts - -Mock `globalThis.fetch` and fake timers lightly: - -- 429 then 200 -> two fetch calls, final response 200 -- 503 with `Retry-After: 0` then 200 -> two calls -- 400 -> no retry, one call -- caller-aborted signal -> no retry - -### Existing checks - -- `bun x tsc --noEmit` -- `bun test tests/kiro-retry.test.ts tests/kiro-adapter.test.ts tests/kiro-images.test.ts` - -## Acceptance - -- Other adapters keep existing direct fetch behavior. -- Kiro does not retry after stream parsing has begun; retry happens only before - the caller receives a Response. -- No 401/403 refresh behavior in this phase. - -## Commit - -feat(kiro): retry transient HTTP failures before stream parse diff --git a/devlog/_fin/143_kiro-gateway-parity/85_phase_thinking_exposure.md b/devlog/_fin/143_kiro-gateway-parity/85_phase_thinking_exposure.md deleted file mode 100644 index e8d0d9a35..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/85_phase_thinking_exposure.md +++ /dev/null @@ -1,64 +0,0 @@ -# Phase 85 (P0 user-visible leak) - Hide Kiro blocks - -## Trigger - -The user observed a Kiro response exposing a raw leading block: - - ... - -This is a user-visible leak. It must be routed as reasoning output, not normal -assistant text. - -## Root cause - -Kiro emits fake-thinking content as ordinary CodeWhisperer `content` chunks. -`parseKiroStream` currently forwards every content chunk as `text_delta`, so -the bridge emits it as `response.output_text.delta` and the UI displays the raw -tags and internal reasoning text. - -The bridge already supports hidden/raw reasoning through AdapterEvent -`reasoning_raw_delta`, which becomes `response.reasoning_text.delta`. - -## File changes - -### ADD src/adapters/kiro-thinking.ts - -Add a small streaming FSM: - -- detect only a leading ``, ``, or `` block -- tolerate opening/closing tags split across chunks -- emit content inside the block as `reasoning_raw_delta` -- emit text after the closing tag as `text_delta` -- if no leading thinking tag appears, flush buffered prefix as normal text -- if stream ends while inside a thinking block, flush buffered content as raw - reasoning (no raw tags) -- after the first text is emitted, do not parse later literal tags - -### MODIFY src/adapters/kiro.ts - -- import `KiroThinkingParser` -- instantiate it once per `parseKiroStream` -- for `content` events, feed text into the parser and yield parser-produced - events instead of always yielding `text_delta` -- on stream end, flush parser-finalized events before usage `done` -- count output usage from visible text + reasoning text, but never include raw - tags - -### MODIFY tests/kiro-adapter.test.ts - -Add parseStream regression tests: - -- leading `rawanswer` -> `reasoning_raw_delta:raw`, - `text_delta:answer`, no text event containing `` -- opening and closing tags split across chunks are parsed -- non-leading literal `` inside normal answer remains visible text -- unterminated leading thinking block flushes as reasoning at stream end - -## Verify - -- bun x tsc --noEmit -- bun test tests/kiro-adapter.test.ts tests/kiro-images.test.ts tests/bridge.test.ts - -## Commit - -fix(kiro): route leading thinking blocks as reasoning diff --git a/devlog/_fin/143_kiro-gateway-parity/90_phase_oauth_singleflight_reload.md b/devlog/_fin/143_kiro-gateway-parity/90_phase_oauth_singleflight_reload.md deleted file mode 100644 index be2021145..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/90_phase_oauth_singleflight_reload.md +++ /dev/null @@ -1,72 +0,0 @@ -# Phase 90 (P0-3) - OAuth refresh singleflight + Kiro SQLite reload - -## Security boundary - -Trust boundary: local credential store (`OPENCODEX_HOME/auth.json`) + imported -Kiro CLI SQLite token cache -> outbound Kiro runtime Authorization header. - -Protecting: OAuth access/refresh tokens and correct persisted credential state. -Main failure mode: concurrent near-expiry requests refresh with the same old -refresh token, race writes to auth.json, or miss a fresher Kiro CLI token. - -## Scope - -Implement the hardening that is correct inside current opencodex ownership: - -- general per-provider singleflight around `getValidAccessToken` refresh work -- Kiro-only SQLite reload before refreshing: if installed Kiro CLI has a fresh - token, persist/use that instead of hitting the desktop refresh endpoint -- Kiro-only SQLite reload after refresh failure: if external Kiro CLI refreshed - during/after our failed attempt, recover by importing it - -Out of scope: - -- AWS SSO OIDC/device-registration refresh path -- multi-account failover -- changing auth-source precedence for login - -## File changes - -### MODIFY src/oauth/index.ts - -1. Import `readKiroCliSqlite` in addition to login/refresh. -2. Add module-level `const tokenRefreshes = new Map>();` -3. Factor refresh path into `refreshAndPersistAccessToken(provider, def, cred)`: - - for `provider === "kiro"`, call `readKiroCliSqlite()` - - if imported token is valid beyond `REFRESH_SKEW_MS`, saveCredential(provider, imported) and return imported.access - - otherwise call `def.refresh(cred.refresh)`, saveCredential, return access - - if refresh throws and provider is Kiro, re-read SQLite and use it if now valid; otherwise rethrow -4. Update `getValidAccessToken`: - - still returns immediately when existing credential is valid - - if refresh needed and a provider refresh promise exists, return it - - otherwise create promise, store in map, delete in finally - -### Tests - -NEW tests/oauth-refresh.test.ts: - -- concurrent expired Kiro calls share one refresh request and both return same access -- fresh Kiro SQLite token is imported before refresh endpoint is called -- failed refresh recovers if SQLite now has a valid token -- non-expired stored credential still returns without refresh - -Use isolated `OPENCODEX_HOME` and `HOME` temp dirs. Seed auth.json via -`saveCredential`, seed Kiro SQLite with the same schema used in kiro-oauth tests, -and mock `globalThis.fetch`. - -## Verification - -- bun x tsc --noEmit -- bun test tests/oauth-refresh.test.ts tests/kiro-oauth.test.ts - -## Acceptance - -- No tokens are logged. -- Singleflight map always clears after success/failure. -- Existing valid credentials remain fast-path and do not touch SQLite/fetch. -- Kiro reload path never weakens credential precedence at login time; it only - prevents stale refresh races once an opencodex Kiro credential already exists. - -## Commit - -fix(oauth): singleflight refresh and reload Kiro CLI tokens diff --git a/devlog/_fin/143_kiro-gateway-parity/95_phase_auth_input_hardening.md b/devlog/_fin/143_kiro-gateway-parity/95_phase_auth_input_hardening.md deleted file mode 100644 index 727628e60..000000000 --- a/devlog/_fin/143_kiro-gateway-parity/95_phase_auth_input_hardening.md +++ /dev/null @@ -1,235 +0,0 @@ -# Phase 95 (P0 residual) - Kiro auth input hardening without multi-account - -## Trigger - -The external code review says Phase 90 closed refresh singleflight and -SQLite reload, but not the broader single-account Kiro auth surface: - -- JSON credential file import. -- AWS SSO OIDC refresh through `clientId`/`clientSecret`. -- Device registration keys in kiro-cli SQLite. -- API-region/runtime-region split (`KIRO_API_REGION`) without conflating it - with SSO/auth refresh region (`KIRO_REGION`). -- Broader SQLite path coverage and clearer diagnostics. - -User scope decision: account failover / multi-account routing is not needed. -Do not implement account pool, circuit breaker, sticky account selection, or -per-request account failover in this phase. - -## Current state - -- `src/oauth/index.ts` already has per-provider singleflight refresh. -- `src/oauth/index.ts` already reloads fresh Kiro CLI SQLite tokens before - calling the refresh endpoint and after refresh failure. -- `src/oauth/kiro.ts` still reads only two hardcoded SQLite paths, ignores - device-registration keys, has no JSON credential file import, and uses - `KIRO_REGION` for both auth refresh and runtime API. -- `src/adapters/kiro.ts` calls `resolveKiroRegion()` for the runtime URL. - -## Diff plan - -### ADD `src/oauth/kiro-credentials.ts` - -Add a small parser module so `src/oauth/kiro.ts` stays below the 500-line -limit: - -- Export `ImportedKiroCredential`, `KiroAuthType`, and helpers: - - `readImportedKiroCredential(opts?)` - - `readKiroCliSqliteCredential()` - - `inferRegionFromProfileArn(arn)` -- Source precedence: - 1. `KIRO_CREDS_FILE` or `KIRO_CREDENTIALS_FILE` JSON. - 2. `KIRO_CLI_DB_FILE` SQLite override. - 3. Known SQLite paths: - - `~/Library/Application Support/kiro-cli/data.sqlite3` - - `~/.local/share/kiro-cli/data.sqlite3` - - `~/.local/share/amazon-q/data.sqlite3` - - `~/.kiro/sso/cache.db` - 4. Existing env-token fallback stays in `loginKiro()`. -- JSON fields: - - `accessToken` / `access_token` - - `refreshToken` / `refresh_token` - - `expiresAt` / `expires_at` - - `profileArn` / `profile_arn` - - `region` - - `apiRegion` / `api_region` - - `clientId` / `client_id` - - `clientSecret` / `client_secret` - - `clientIdHash`, loading `~/.aws/sso/cache/{clientIdHash}.json`. -- SQLite fields: - - Existing token keys: `kirocli:social:token`, - `kirocli:odic:token`, `codewhisperer:odic:token`. - - Device-registration keys: `kirocli:odic:device-registration`, - `codewhisperer:odic:device-registration`. - - `state` table profile row: `api.codewhisperer.profile`. -- Apply `PRAGMA busy_timeout = 5000` on opened SQLite handles. -- Preserve read-only behavior: this phase only reads external Kiro stores. -- Diagnostics must be secret-free. Return only source labels/status codes such - as `missing`, `invalid_json`, `token_found`, `registration_found`, and - `schema_mismatch`; never include token values, refresh tokens, client secrets, - profile ARNs, raw JSON payloads, or absolute user paths in diagnostic objects, - progress strings, thrown messages, or tests. - -### MODIFY `src/oauth/kiro.ts` - -- Replace inline SQLite scanning with the new helper. -- Keep the public `readKiroCliSqlite()` export shape for existing tests. -- Preserve and adapt the public `inspectKiroCliSqlite()` export introduced by - security commit `68b079f`; it must continue returning - `{ token, diagnostics }` with token values only in `token`, never in - diagnostics. -- `loginKiro()` source order becomes imported credential JSON/SQLite -> - `KIRO_ACCESS_TOKEN` -> manual paste. -- Add `resolveKiroApiRegion()`: - - `KIRO_API_REGION` - - imported `apiRegion` - - imported `profileArn` ARN region - - imported SSO `region` - - `KIRO_REGION` - - `us-east-1` -- Keep `resolveKiroRegion()` as auth/SSO refresh region: - - imported SSO `region` - - `KIRO_REGION` - - `us-east-1` -- Keep `resolveKiroProfileArn()` env-first, then imported credential. -- `refreshKiroToken()` chooses: - - AWS SSO OIDC endpoint when imported current credential has - `clientId` + `clientSecret`. - - Kiro Desktop refresh endpoint otherwise. -- OIDC request: - - URL `https://oidc.{region}.amazonaws.com/token`. - - JSON body with camelCase: - `{ grantType: "refresh_token", clientId, clientSecret, refreshToken }`. - - Header `Content-Type: application/json`. - - Rationale: AWS IAM Identity Center OIDC `CreateToken` is an AWS JSON API, - not a generic form-urlencoded OAuth endpoint. The official AWS docs list - `Content-type: application/json` and camelCase request fields, and - `kiro-gateway`'s current implementation sends this same JSON/camelCase - payload. - - On a `400` with a new imported SQLite refresh token, retry once with the - reloaded token. - -### MODIFY `src/adapters/kiro.ts` - -- Use `resolveKiroApiRegion()` for `https://runtime.{region}.kiro.dev/`. -- Continue using `resolveKiroProfileArn()` for payload/header profile ARN. - -### MODIFY `tests/kiro-oauth.test.ts` - -Add regression coverage for: - -- JSON credential import from `KIRO_CREDS_FILE`. -- Enterprise `clientIdHash` loading from `~/.aws/sso/cache/{hash}.json`. -- AWS SSO OIDC refresh URL/body selection. -- `KIRO_API_REGION` runtime override separate from `KIRO_REGION`. -- SQLite device-registration client credentials and `state` table profile ARN - region detection. - -### MODIFY `tests/oauth-refresh.test.ts` - -- Keep existing singleflight/reload tests green. -- Extend temporary env cleanup for new Kiro env vars if needed. - -## Verification - -- `bun x tsc --noEmit` -- `bun test tests/kiro-oauth.test.ts tests/oauth-refresh.test.ts tests/kiro-adapter.test.ts` -- `wc -l src/oauth/kiro.ts src/oauth/kiro-credentials.ts src/adapters/kiro.ts` - -## Commit - -`fix(oauth): broaden single-account Kiro credential inputs` - -## Explicit non-goals - -- No opencodex Kiro account pool. -- No account circuit breaker. -- No sticky account routing. -- No write-back to Kiro CLI SQLite in this phase. - -## Build record - -Files changed: - -- ADD `src/oauth/kiro-credentials.ts`: centralized Kiro credential import from - JSON credential files, SQLite overrides, macOS/Linux/Amazon-Q SQLite paths, - device registration rows, profile state rows, and AWS SSO cache references. -- MODIFY `src/oauth/kiro.ts`: replaced inline SQLite scanning with the helper, - preserved public SQLite import APIs, added `resolveKiroApiRegion()`, kept - `resolveKiroRegion()` for auth/SSO refresh, and selected AWS SSO OIDC refresh - when imported credentials carry client registration data. -- MODIFY `src/adapters/kiro.ts`: runtime requests now use - `resolveKiroApiRegion()` instead of conflating runtime API region with auth - region. -- MODIFY `src/oauth/types.ts` and `src/oauth/store.ts`: added and persisted the - safe `credential-file` credential source label through the existing OAuth - store allowlist. -- MODIFY `tests/kiro-oauth.test.ts`: added JSON credential import, - `clientIdHash` AWS SSO cache loading, AWS SSO OIDC body/URL selection, - SQLite device-registration/profile-state import, no-secret diagnostics, and - API-region resolver coverage. -- MODIFY `tests/kiro-adapter.test.ts`: added runtime URL coverage for - `KIRO_API_REGION`. -- MODIFY `tests/oauth-status-privacy.test.ts`: confirmed `credential-file` - survives the credential-store allowlist without allowing arbitrary metadata. - -Verification: - -- `bun test tests/kiro-oauth.test.ts tests/oauth-refresh.test.ts tests/kiro-adapter.test.ts tests/oauth-status-privacy.test.ts` - -> 59 pass, 0 fail. -- `bun x tsc --noEmit` -> exit 0, no diagnostics. -- `wc -l src/oauth/kiro.ts src/oauth/kiro-credentials.ts src/adapters/kiro.ts` - -> 161 / 242 / 496 lines, all under the 500-line project limit. - -Security notes: - -- Diagnostics remain label/status-only and intentionally exclude access tokens, - refresh tokens, client secrets, profile ARNs, raw JSON payloads, and absolute - user paths. -- The phase stays single-account. It broadens credential inputs and refresh - compatibility but does not introduce failover, account pools, or write-back to - external Kiro stores. - -## Independent audit follow-up - -Read-only audit (`Bacon`) initially returned FAIL after commit `4c97e0d`. - -Blocking findings: - -- Region strings from `KIRO_REGION`, `KIRO_API_REGION`, and imported - credential metadata were interpolated into Kiro runtime/auth URLs without a - central validator. -- `clientIdHash` was joined directly into the AWS SSO cache path, allowing - traversal outside the intended cache directory. -- Kiro eventstream exception/error payloads and stream parser catch errors could - surface raw upstream JSON, tokens, client secrets, profile ARNs, or local - absolute paths. - -Follow-up fixes: - -- Commit `931847b fix(kiro): sanitize region and upstream error details` - added: - - central Kiro region normalization/rejection in - `src/oauth/kiro-credentials.ts` and `src/oauth/kiro.ts`; - - `clientIdHash` basename-safe allowlisting; - - client-secret redaction in `src/redact.ts`; - - host-injection, traversal, and error-leak regression tests. -- Commit `e95338e fix(kiro): include safe error formatter` added the tracked - `src/adapters/kiro-errors.ts` helper so `src/adapters/kiro.ts` stays under - the project 500-line limit. - -Re-verification: - -- `bun test tests/redact.test.ts tests/kiro-oauth.test.ts tests/kiro-adapter.test.ts tests/oauth-status-privacy.test.ts` - -> 67 pass, 0 fail. -- `bun test tests/redact.test.ts tests/crash-guard.test.ts tests/usage-debug.test.ts tests/request-log.test.ts tests/server-auth.test.ts tests/error-fidelity.test.ts tests/usage-log.test.ts tests/usage-summary.test.ts tests/oauth-status-privacy.test.ts tests/kiro-oauth.test.ts tests/oauth-refresh.test.ts tests/config.test.ts tests/kiro-adapter.test.ts` - -> 179 pass, 0 fail, 604 expect calls. -- `bun x tsc --noEmit` -> exit 0, no diagnostics. -- `wc -l src/oauth/kiro.ts src/oauth/kiro-credentials.ts src/adapters/kiro.ts src/adapters/kiro-errors.ts` - -> 164 / 256 / 497 / 40 lines. - -Re-audit: - -- `Bacon` returned PASS. The audit confirmed validated region interpolation, - `clientIdHash` traversal blocking, safe Kiro upstream error formatting, - regression coverage, and file-size compliance. diff --git a/devlog/_fin/145_common-security-hardening/00_plan.md b/devlog/_fin/145_common-security-hardening/00_plan.md deleted file mode 100644 index 8793cde4b..000000000 --- a/devlog/_fin/145_common-security-hardening/00_plan.md +++ /dev/null @@ -1,81 +0,0 @@ -# 145 — Common Security Hardening (outside Kiro parity) - -Goal: harden OpenCodex's shared security surfaces on `feat/kiro-on-dev` while -Kiro adapter parity continues separately in `143_kiro-gateway-parity`. - -This plan intentionally excludes Kiro-specific parity work such as CodeWhisperer -retry semantics, payload trimming, tool-schema compatibility, model resolution, -and eventstream adapter behavior. Those stay in plan 143. This plan covers the -common proxy surfaces that can leak secrets, expose local control APIs, or -persist more diagnostic data than needed. - -## Context read - -- `README.md`: product shape, localhost dashboard, provider config, account pool. -- `structure/01_runtime.md`: `src/server.ts` owns `/v1/responses`, `/v1/models`, - static GUI, and `/api/*`; adapter events stay internal until bridge conversion. -- `structure/05_gui-and-management-api.md`: management API, logs, usage summary, - `usage.jsonl`, and `usage-debug.jsonl` invariants. -- `structure/06_docs-and-release.md`: runtime quality gate commands and CI scope. -- `devlog/_plan/143_kiro-gateway-parity/*`: Kiro-specific PABCD stream, to avoid - overlapping user-owned parity work. -- GPT Pro Q2 security review summary: confirmed common risks are secret logging, - local API exposure, usage/debug privacy, credential import safety, and config - input validation. - -## Threat model - -| Asset | Boundary | Attacker | Failure impact | -| --- | --- | --- | --- | -| Provider API keys and OAuth access/refresh tokens | Browser / local app -> proxy -> config/logs | Malicious local webpage, LAN host, compromised shell, bug report leak | Provider account and quota compromise | -| Local management API and WebSocket routes | Browser Origin / host binding / API key boundary | Malicious webpage, DNS rebinding, LAN host when non-loopback bound | Config mutation, request driving, log/usage disclosure | -| Usage and debug artifacts | Runtime -> `~/.opencodex/*.jsonl` -> GUI/API | Local user, synced backup, support bundle | Prompt, account, project, or secret metadata disclosure | -| Provider config URLs and headers | User config -> outbound fetch | Malicious config or UI input | SSRF, private network probing, credential exfiltration | - -## Work-phase map - -Each work-phase is one full PABCD cycle with its own focused tests and atomic -commit. Phase 0 is this documentation-only cycle. - -| Phase | Priority | Surface | Outcome | -| --- | --- | --- | --- | -| 00 | P0 | Plan and threat model | Scope frozen; phase stubs created; Kiro parity excluded | -| 10 | P0 | Secret redaction foundation | Shared redactor for logs/diagnostics; tests prove token/header/body masking | -| 20 | P0 | Crash/request/usage debug sinks | Existing diagnostic writers use the redactor; no bearer/refresh/profile leaks | -| 30 | P0 | Local HTTP/WS boundary | Origin/CORS/API-key behavior verified and patched where missing | -| 40 | P1 | Usage privacy minimization | Usage/debug records store numeric/coarse metadata only; tests lock shape | -| 50 | P1 | Credential import safeguards | Imported credentials have explicit source/safety metadata and no silent leak paths | -| 60 | P1 | Config URL/header input validation | Provider URL/header validation blocks dangerous local/private/protocol input where applicable | -| 90 | P0 | Final security review | Independent review + full relevant test/typecheck evidence | - -## Phase dependencies - -Phase 10 must land before phases 20 and 50 so every sink can reuse one redaction -policy. Phase 30 can run in parallel conceptually, but will be done as its own -PABCD pass to avoid mixing server-boundary changes with logging changes. Phase -60 waits until the existing server/config tests are understood so validation does -not break legitimate local provider use cases such as Ollama. - -## Verification baseline - -- Targeted tests per phase under the existing `bun test tests/...` harness. -- `bun x tsc --noEmit` after code phases. -- For secret-handling changes, add negative tests containing realistic marker - strings such as bearer tokens, refresh tokens, cookies, profile ARNs, and API - keys, then assert the exact values never appear in stored/output records. -- For server-boundary changes, use the existing `tests/server-auth.test.ts` and - focused additions instead of new harnesses. - -## Commit discipline - -- Commit Phase 0 docs separately. -- Commit each implementation phase separately. -- Do not push unless the user explicitly asks in the same turn. -- Leave unrelated `.opencode/` untracked state untouched. - -## Completion criteria - -The goal is complete when phases 00/10/20/30/40/50/60/90 have each passed PABCD -with devlog evidence, atomic commits, targeted tests, typecheck for code phases, -and final independent review confirming common OpenCodex security surfaces are -hardened without taking over Kiro adapter parity work. diff --git a/devlog/_fin/145_common-security-hardening/10_phase1_redaction-foundation.md b/devlog/_fin/145_common-security-hardening/10_phase1_redaction-foundation.md deleted file mode 100644 index a7a4f948c..000000000 --- a/devlog/_fin/145_common-security-hardening/10_phase1_redaction-foundation.md +++ /dev/null @@ -1,87 +0,0 @@ -# 10 — Phase 1: Secret redaction foundation - -Purpose: introduce or consolidate a shared redaction policy that can be reused by -request logs, crash diagnostics, usage debug records, OAuth/import diagnostics, -and server error surfaces. - -Planned surfaces: - -- `src/redact.ts` or existing nearest owner if one already exists. -- Tests proving redaction of: - - `Authorization: Bearer ...` - - `apiKey`, `accessToken`, `refreshToken` - - cookies and `Set-Cookie` - - Kiro `profileArn` - - bearer-like strings embedded in nested objects and strings - -Non-goals: - -- Do not change Kiro adapter parity behavior. -- Do not change provider routing. - -Verification: - -- Focused redaction unit tests. -- Typecheck. - -## Diff-level plan - -NEW `src/redact.ts` - -- Export `REDACTED_SECRET = "[REDACTED]"`. -- Export `redactSecretString(value: string): string`. - - Redact bearer-like strings: `Bearer `, `sk-...`, `api_key=...`, - `access_token=...`, `refresh_token=...`, `refreshToken=...`. - - Redact Kiro/AWS profile ARNs as a stable sensitive identifier class. - - Preserve non-secret text so diagnostics stay useful. -- Export `redactSecrets(value: unknown): unknown`. - - Recursively redact arrays and plain objects. - - Redact values whose keys are sensitive (`authorization`, `cookie`, - `set-cookie`, `apiKey`, `accessToken`, `refreshToken`, `token`, - `profileArn`, `x-api-key`, `x-goog-api-key`, `x-amz-security-token`). - - Redact string values by pattern even when the key is not sensitive. - - Leave numbers, booleans, null, undefined, and Dates safe. -- Export `redactHeaders(headers: Headers | Record): Record`. - - Normalize header keys to lower-case in the returned diagnostic object. - - Mask sensitive headers by key. - - Pattern-redact non-sensitive header values. -- Export `redactUrlForLog(url: string): string`. - - Remove credentials and query/hash values. - - Pattern-redact invalid URL strings best-effort. - -NEW `tests/redact.test.ts` - -- Verify `redactSecretString` masks bearer/API/access/refresh/profile values. -- Verify recursive `redactSecrets` masks nested objects and arrays without - mutating primitive non-secrets. -- Verify `redactHeaders` masks `authorization`, `cookie`, `set-cookie`, - `x-api-key`, and preserves safe metadata like `content-type`. -- Verify `redactUrlForLog` strips credentials/query/hash and keeps - `protocol//host/path`. - -MODIFY `devlog/_plan/145_common-security-hardening/10_phase1_redaction-foundation.md` - -- Record the exact changed files, verification command, and commit after build. - -Out of scope for Phase 10: - -- Do not wire the redactor into `src/crash-guard.ts`, `src/server.ts`, or - `src/usage-debug.ts` yet. Those are Phase 20 sink changes. -- Do not modify Kiro adapter request/stream semantics. - -## Build record - -Files changed: - -- NEW `src/redact.ts`: shared redaction helpers for strings, nested objects, - headers, and URLs. -- NEW `tests/redact.test.ts`: focused regression coverage for bearer/API token, - access/refresh token, cookie/header, Kiro/AWS profile ARN, nested object, and - URL redaction. -- MODIFY `devlog/_plan/145_common-security-hardening/10_phase1_redaction-foundation.md`: - this build/verification record. - -Verification: - -- `bun test tests/redact.test.ts` -> 8 pass, 0 fail. -- `bun x tsc --noEmit` -> exit 0, no diagnostics. diff --git a/devlog/_fin/145_common-security-hardening/20_phase2_diagnostic-sinks.md b/devlog/_fin/145_common-security-hardening/20_phase2_diagnostic-sinks.md deleted file mode 100644 index f9122da0f..000000000 --- a/devlog/_fin/145_common-security-hardening/20_phase2_diagnostic-sinks.md +++ /dev/null @@ -1,82 +0,0 @@ -# 20 — Phase 2: Diagnostic sink redaction - -Purpose: route existing crash-guard, request-log, and usage-debug diagnostic -sinks through the shared redactor before any data is stored or returned through -the GUI/API. - -Planned surfaces: - -- `src/crash-guard.ts` -- `src/server.ts` request log helpers -- `src/usage-debug.ts` -- Existing tests near `tests/crash-guard.test.ts`, `tests/request-log.test.ts`, - and `tests/usage-debug.test.ts` - -Verification: - -- Tests assert marker secrets never appear in diagnostic output. -- Existing request-log filtering still works. -- Typecheck. - -## Diff-level plan - -MODIFY `src/crash-guard.ts` - -- Import `redactSecretString` and `redactUrlForLog` from `./redact`. -- Redact error messages, stacks, causes, codes, inspect output, promise render, - and fetch rejection strings before formatting crash entries. -- Replace local `redactUrl()` with shared `redactUrlForLog()`. - -MODIFY `src/usage-debug.ts` - -- Import `redactSecretString` and `redactSecrets` from `./redact`. -- Make `truncateForDebug()` redact before truncation so truncated samples cannot - preserve the beginning of a token. -- Make `appendUsageDebug()` sanitize the full record before JSONL write. - -MODIFY `tests/crash-guard.test.ts` - -- Add a regression test proving `formatCrashEntry()` does not include bearer - tokens, refresh tokens, API keys, cookies, or Kiro/AWS profile ARNs from error - message/stack/cause/code paths. -- Extend the recent-fetch test to prove query, credentials, and bearer-like - invalid URL strings are redacted. - -MODIFY `tests/usage-debug.test.ts` - -- Add a test proving `truncateForDebug()` redacts before applying the byte cap. -- Add a test proving `appendUsageDebug()` writes redacted body samples and - extracted usage metadata only. - -MODIFY `devlog/_plan/145_common-security-hardening/20_phase2_diagnostic-sinks.md` - -- Record build evidence and verification commands. - -Out of scope: - -- Do not change the shape of `RequestLogEntry` unless tests reveal an actual - secret-bearing field. -- Do not alter Kiro adapter stream/retry behavior. - -## Build record - -Files changed: - -- MODIFY `src/crash-guard.ts`: crash detail, stack, cause, code, inspect output, - promise render, fetch rejection strings, and fetch URLs are now passed through - shared redaction helpers before formatting. -- MODIFY `src/usage-debug.ts`: debug body samples are redacted before truncation, - and the full debug record is redacted before JSONL append. -- MODIFY `tests/crash-guard.test.ts`: added secret-leak regression coverage for - crash entry details and diagnostics. -- MODIFY `tests/usage-debug.test.ts`: added redaction-before-truncation and JSONL - body sample redaction coverage. -- MODIFY `devlog/_plan/145_common-security-hardening/20_phase2_diagnostic-sinks.md`: - this build/verification record. - -Verification: - -- `bun test tests/crash-guard.test.ts tests/usage-debug.test.ts tests/redact.test.ts` - -> 29 pass, 0 fail. -- `bun test tests/request-log.test.ts` -> 8 pass, 0 fail. -- `bun x tsc --noEmit` -> exit 0, no diagnostics. diff --git a/devlog/_fin/145_common-security-hardening/30_phase3_local-server-boundary.md b/devlog/_fin/145_common-security-hardening/30_phase3_local-server-boundary.md deleted file mode 100644 index af06b56a0..000000000 --- a/devlog/_fin/145_common-security-hardening/30_phase3_local-server-boundary.md +++ /dev/null @@ -1,74 +0,0 @@ -# 30 — Phase 3: Local HTTP/WS boundary - -Purpose: verify and harden local server exposure for `/api/*`, `/v1/models`, -`/v1/responses`, and WebSocket upgrades. - -Planned surfaces: - -- `src/server.ts` -- `src/ws-bridge.ts` only if server tests reveal a WebSocket boundary gap. -- `tests/server-auth.test.ts` -- `tests/ws-endpoint.test.ts` if needed. - -Checks: - -- Non-loopback binding requires configured API auth for API/model/response - surfaces. -- Non-local `Origin` is rejected for management and WebSocket paths. -- CORS does not use wildcard credentials behavior. -- WebSocket upgrade inherits the same local-origin and auth boundary. - -Verification: - -- Focused server-auth tests. -- Typecheck. - -## Diff-level plan - -MODIFY `tests/server-auth.test.ts` - -- Add an `OPTIONS` preflight regression test: - - loopback/default config rejects non-loopback `Origin` with 403. - - loopback/default config accepts matching loopback `Origin` with 204. -- Add a WebSocket upgrade regression test for non-loopback bindings: - - valid `X-OpenCodex-API-Key` is not enough when `Origin` is hostile. - - response is 403 with `origin_rejected` / cross-origin rejection shape. -- Reuse existing `startServer`, `saveConfig`, and `config()` test helpers. - -MODIFY `src/server.ts` only if the new tests expose an actual boundary gap. - -MODIFY `devlog/_plan/145_common-security-hardening/30_phase3_local-server-boundary.md` - -- Record whether this phase was test-only or required a server patch. -- Record verification commands and commit. - -Out of scope: - -- Do not change Kiro adapter parity files. -- Do not broaden CORS to support arbitrary browser apps. -- Do not introduce a new auth scheme; use the existing local API auth behavior. - -## Build record - -Files changed: - -- MODIFY `tests/server-auth.test.ts`: added `OPTIONS` hostile-origin regression - coverage and WebSocket hostile-origin coverage with a valid local API token. -- MODIFY `src/errors.ts`: preserve explicit `origin_rejected` error code before - generic 401/403 authentication mapping. -- MODIFY `tests/error-fidelity.test.ts`: locked `origin_rejected` classification. -- MODIFY `devlog/_plan/145_common-security-hardening/30_phase3_local-server-boundary.md`: - this build/verification record. - -Implementation note: - -- The planned test-only slice exposed one real bug: WebSocket hostile-Origin - responses had the right status/message but were classified as `invalid_api_key`. - The fix keeps the boundary behavior and only corrects the machine-readable - error code. - -Verification: - -- `bun test tests/server-auth.test.ts tests/error-fidelity.test.ts` -> 38 pass, - 0 fail. -- `bun x tsc --noEmit` -> exit 0, no diagnostics. diff --git a/devlog/_fin/145_common-security-hardening/40_phase4_usage-privacy.md b/devlog/_fin/145_common-security-hardening/40_phase4_usage-privacy.md deleted file mode 100644 index b6519b03a..000000000 --- a/devlog/_fin/145_common-security-hardening/40_phase4_usage-privacy.md +++ /dev/null @@ -1,68 +0,0 @@ -# 40 — Phase 4: Usage privacy minimization - -Purpose: ensure persistent usage accounting and debug summaries stay numeric and -coarse, without prompts, tool inputs, profile ARNs, raw upstream bodies, or -credential-derived identifiers. - -Planned surfaces: - -- `src/usage-log.ts` -- `src/usage-summary.ts` -- `src/usage-debug.ts` -- `tests/usage-log.test.ts` -- `tests/usage-summary.test.ts` -- `tests/usage-debug.test.ts` - -Verification: - -- Usage records use mode `0o600`. -- Stored records contain provider/model/status/counts but not prompt/tool text. -- Debug body samples are redacted and size-capped. -- Typecheck. - -## Diff-level plan - -MODIFY `src/usage-log.ts` - -- Add an internal `normalizeUsageEntry(entry: PersistedUsageEntry): PersistedUsageEntry`. -- `appendUsageEntry()` writes only the normalized allowlisted fields: - `requestId`, `timestamp`, `provider`, `model`, optional `resolvedModel`, - `status`, `durationMs`, `usageStatus`, optional numeric `usage`, and optional - `totalTokens`. -- Ignore any runtime extra keys such as `prompt`, `input`, `messages`, - `headers`, `authorization`, `accessToken`, `refreshToken`, `profileArn`, or - tool payloads even if a caller passes a widened object. - -MODIFY `tests/usage-log.test.ts` - -- Add a regression test that passes an object with secret-bearing extra keys via - a widened cast and proves the persisted JSONL line omits them. -- Keep existing mode `0o600` and malformed-line tests. - -MODIFY `devlog/_plan/145_common-security-hardening/40_phase4_usage-privacy.md` - -- Record changed files, verification commands, and commit. - -Out of scope: - -- Do not change `/api/usage` response shape unless this phase reveals a leak. -- Do not change usage aggregation semantics. -- Do not change `usage-debug` again; diagnostic debug sink redaction was Phase 20. - -## Build record - -Files changed: - -- MODIFY `src/usage-log.ts`: `appendUsageEntry()` now writes a normalized - allowlisted `PersistedUsageEntry` so runtime extra fields cannot persist into - `usage.jsonl`. -- MODIFY `tests/usage-log.test.ts`: added widened-object regression coverage for - prompt/message/header/token/profile extra fields. -- MODIFY `devlog/_plan/145_common-security-hardening/40_phase4_usage-privacy.md`: - this build/verification record. - -Verification: - -- `bun test tests/usage-log.test.ts tests/usage-summary.test.ts` -> 15 pass, - 0 fail. -- `bun x tsc --noEmit` -> exit 0, no diagnostics. diff --git a/devlog/_fin/145_common-security-hardening/50_phase5_credential-import-safeguards.md b/devlog/_fin/145_common-security-hardening/50_phase5_credential-import-safeguards.md deleted file mode 100644 index 4baa7f509..000000000 --- a/devlog/_fin/145_common-security-hardening/50_phase5_credential-import-safeguards.md +++ /dev/null @@ -1,129 +0,0 @@ -# 50 — Phase 5: Non-Kiro credential import safeguards - -Purpose: make common OAuth credential persistence auditable and safe without -claiming or changing Kiro-specific behavior. Kiro adapter, Kiro OAuth, and -Kiro tests are out of scope for this common-security track. - -## Planned non-Kiro surfaces - -- `src/oauth/types.ts` -- `src/oauth/store.ts` -- `src/oauth/index.ts` only for provider-neutral status/source handling -- `src/oauth/local-token-detect.ts` -- `src/oauth/xai.ts` -- `src/oauth/anthropic.ts` -- `tests/oauth-status-privacy.test.ts` - -## Checks - -- Imported non-Kiro local credentials have clear safe source metadata. -- `auth.json` rewrites normalize the whole store, not just the provider being - saved, so legacy extra fields cannot survive a later write. -- `getLoginStatus()` can expose only allowlisted source labels and masked email - metadata; it never returns access tokens, refresh tokens, arbitrary legacy - `source` strings, prompts, headers, ID tokens, or diagnostics. -- Refresh-token persistence remains intentionally unchanged; replacing it with - memory-only imports or OS keychain storage is a product decision outside this - slice. - -## Diff-level plan - -MODIFY `src/oauth/types.ts` - -- Add a small `OAuthCredentialSource` union: - `oauth | local-cli | credential-file | environment | manual`. -- Add optional `source` metadata to persisted OAuth credentials. - -MODIFY `src/oauth/store.ts` - -- Normalize loaded credentials before any caller observes them. -- Normalize the entire persisted store before writing `auth.json`. -- Persist only `access`, `refresh`, `expires`, optional masked-status metadata - (`email`, `accountId`, `source`), and drop accidental extra fields such as - prompt text, headers, ID tokens, or diagnostics. - -MODIFY `src/oauth/index.ts` - -- Default ordinary OAuth logins to source `oauth`. -- Preserve existing source metadata when a provider refresh response does not - provide a new source. -- Expose only safe source metadata through `getLoginStatus()`, never access or - refresh token values. - -MODIFY `src/oauth/local-token-detect.ts` - -- Mark Grok CLI and Claude Code local imports as `local-cli` at detection time. - -MODIFY `src/oauth/xai.ts` and `src/oauth/anthropic.ts` - -- Preserve `local-cli` source when an imported local token is refreshed before - persistence. - -MODIFY tests - -- `tests/oauth-status-privacy.test.ts`: status source is safe, invalid legacy - source strings are dropped, and credential persistence allowlists known fields - across the whole store. - -Out of scope: - -- Kiro adapter parity, Kiro OAuth import semantics, Kiro diagnostics, and - `tests/kiro*.test.ts`. - -## Build record - -Files changed for the non-Kiro common track: - -- MODIFY `src/oauth/types.ts`: added `OAuthCredentialSource` and optional - credential `source`. -- MODIFY `src/oauth/store.ts`: credential loading/writing now allowlists known - fields and normalizes all providers before writing `auth.json`. -- MODIFY `src/oauth/index.ts`: `runLogin()` defaults to `oauth` source, - refresh persistence preserves existing source metadata, and - `getLoginStatus()` exposes only safe source metadata. -- MODIFY `src/oauth/local-token-detect.ts`: Grok CLI and Claude Code imports are - tagged `local-cli`. -- MODIFY `src/oauth/xai.ts` and `src/oauth/anthropic.ts`: refreshed local - imports keep `local-cli`. -- MODIFY `tests/oauth-status-privacy.test.ts`: added status-source, whole-store - credential allowlist, invalid-source, and malformed-store regression coverage. - -Verification: - -- `bun test tests/oauth-status-privacy.test.ts` - -> 5 pass, 0 fail. -- `bun x tsc --noEmit` - -> exit 0, no diagnostics at the time of the non-Kiro slice. - -## Independent verification follow-up - -Read-only sub-agent audit (`Gauss`) returned FAIL after the first credential -metadata commit. - -Non-Kiro findings: - -- Existing providers in `auth.json` were not re-normalized when saving one - provider, so legacy extra fields could survive rewrites. -- xAI/Anthropic local-token imports could be labeled as `oauth`. -- Refreshed credentials could lose existing `source` metadata. -- Legacy arbitrary `source` strings could be reflected by `getLoginStatus()`. - -Follow-up changes: - -- MODIFY `src/oauth/store.ts`: `loadAuthStore()` now normalizes the whole store; - invalid credential sources and extra fields are dropped for all providers. -- MODIFY `src/oauth/index.ts`: refresh persistence preserves existing source - metadata when refresh responses do not provide one. -- MODIFY `src/oauth/local-token-detect.ts`: Grok CLI and Claude Code imports are - tagged `local-cli` at detection time. -- MODIFY `src/oauth/xai.ts` and `src/oauth/anthropic.ts`: refreshed local imports - keep `local-cli`. -- MODIFY `tests/oauth-status-privacy.test.ts`: added whole-store normalization - and invalid-source reflection coverage. - -Re-verification: - -- `bun test tests/oauth-status-privacy.test.ts` - -> 5 pass, 0 fail. -- `bun x tsc --noEmit` - -> exit 0, no diagnostics at the time of the non-Kiro slice. diff --git a/devlog/_fin/145_common-security-hardening/60_phase6_config-url-validation.md b/devlog/_fin/145_common-security-hardening/60_phase6_config-url-validation.md deleted file mode 100644 index f0eaf3f86..000000000 --- a/devlog/_fin/145_common-security-hardening/60_phase6_config-url-validation.md +++ /dev/null @@ -1,77 +0,0 @@ -# 60 — Phase 6: Config URL and header validation - -Purpose: reduce SSRF and credential forwarding risk from provider configuration -while preserving legitimate local providers such as Ollama, LM Studio, and vLLM. - -Planned surfaces: - -- `src/config.ts` -- `src/server.ts` provider create/update validation path -- `src/oauth/key-providers.ts` only if provider validation is centralized there. -- Existing provider/config/server tests. - -Checks: - -- Reject unsupported protocols for provider base URLs. -- Keep local/private provider URLs allowed only where the product intentionally - supports local model servers. -- Prevent user-defined sensitive headers from being reflected through management - APIs or logs. - -Verification: - -- Focused config/provider API tests. -- Typecheck. - -## Diff-level plan - -MODIFY `src/config.ts` - -- Centralize provider base URL validation as `providerBaseUrlConfigError()`: - http/https only, no embedded credentials, no query strings, no fragments. -- Add `providerHeadersConfigError()`: - - headers must be a plain object with valid HTTP token names. - - values must be strings without CR/LF. - - sensitive credential headers (`Authorization`, `Cookie`, `Set-Cookie`, - `Proxy-Authorization`, `x-api-key`, `x-goog-api-key`, - `x-amz-security-token`) are rejected; callers must use `apiKey`/`authMode`. -- Reuse the same validation in config-file schema refinement so unsafe manual - config does not load silently. - -MODIFY `src/server.ts` - -- Reuse config-owned provider URL/header validation in `/api/providers` POST. -- Keep the existing built-in ChatGPT `authMode: "forward"` exception unchanged. - -MODIFY tests - -- `tests/server-auth.test.ts`: provider management rejects sensitive or - injectable headers. -- `tests/config.test.ts`: config diagnostics/load reject unsafe provider URLs - and sensitive/injectable headers. - -Out of scope: - -- Do not block `http://127.0.0.1`, `localhost`, or private-network provider - URLs because local Ollama/LM Studio/vLLM are documented supported use cases. -- Do not add DNS/IP resolution or private-address SSRF blocking in this phase. - -## Build record - -Files changed: - -- MODIFY `src/config.ts`: added shared provider base URL and header validation; - config-file load now rejects unsafe provider URL/header shapes. -- MODIFY `src/server.ts`: `/api/providers` now uses the shared URL/header - validation before persisting provider config. -- MODIFY `tests/server-auth.test.ts`: added API regression coverage for - sensitive provider headers and CR/LF header injection. -- MODIFY `tests/config.test.ts`: added diagnostics/load coverage for unsafe - URLs and sensitive/injectable headers. -- MODIFY `devlog/_plan/145_common-security-hardening/60_phase6_config-url-validation.md`: - this build/verification record. - -Verification: - -- `bun test tests/config.test.ts tests/server-auth.test.ts` -> 56 pass, 0 fail. -- `bun x tsc --noEmit` -> exit 0, no diagnostics. diff --git a/devlog/_fin/145_common-security-hardening/90_final-review.md b/devlog/_fin/145_common-security-hardening/90_final-review.md deleted file mode 100644 index c772775eb..000000000 --- a/devlog/_fin/145_common-security-hardening/90_final-review.md +++ /dev/null @@ -1,57 +0,0 @@ -# 90 — Final review - -Purpose: close the common-security hardening passes with concrete evidence. -This review covers only non-Kiro phases 10 through 60 in -`devlog/_plan/145_common-security-hardening/`. Kiro adapter, Kiro OAuth, Kiro -parity, and `tests/kiro*.test.ts` evidence are intentionally excluded because -that work is owned separately. - -## Implemented phase evidence - -- Phase 10 redaction foundation: - - Commit `46f2e21 feat(security): add shared secret redactor`. - - Main files: `src/redact.ts`, `tests/redact.test.ts`. -- Phase 20 diagnostic sinks: - - Commit `3d91af0 fix(security): redact diagnostic sinks`. - - Main files: `src/crash-guard.ts`, `src/usage-debug.ts`, - `tests/crash-guard.test.ts`, `tests/usage-debug.test.ts`. -- Phase 30 local boundary: - - Commit `4210d49 fix(security): preserve origin rejection errors`. - - Main files: `src/errors.ts`, `tests/server-auth.test.ts`, - `tests/error-fidelity.test.ts`. -- Phase 40 usage privacy: - - Commit `9d29a31 fix(security): allowlist usage log records`. - - Main files: `src/usage-log.ts`, `tests/usage-log.test.ts`. -- Phase 50 credential safeguards: - - Commits `68b079f fix(security): record OAuth credential source safely` and - `4566b11 fix(security): normalize OAuth credential store`, counted here - only for their non-Kiro common OAuth changes. - - Main files: `src/oauth/store.ts`, `src/oauth/index.ts`, - `src/oauth/local-token-detect.ts`, `src/oauth/xai.ts`, - `src/oauth/anthropic.ts`, `tests/oauth-status-privacy.test.ts`. -- Phase 60 provider config validation: - - Commit `96a60a2 fix(security): validate provider URLs and headers`. - - Main files: `src/config.ts`, `src/server.ts`, `tests/config.test.ts`, - `tests/server-auth.test.ts`. - -## Independent review evidence - -- Phase 50 re-audit (`Gauss`): PASS. Confirmed whole-store OAuth - normalization, credential-source allowlist, local-cli source tagging, - and refresh-source preservation for the common OAuth path. -- Phase 60 audit (`James`): PASS. Confirmed scoped URL/header validation, - management DTO redaction, preserved local/private HTTP provider support, and - focused config/server tests. - -## Final verification bundle - -- `bun test tests/redact.test.ts tests/crash-guard.test.ts tests/usage-debug.test.ts tests/request-log.test.ts tests/server-auth.test.ts tests/error-fidelity.test.ts tests/usage-log.test.ts tests/usage-summary.test.ts tests/oauth-status-privacy.test.ts tests/config.test.ts` - -> 120 pass, 0 fail, 418 expect calls. -- `bun x tsc --noEmit` - -> exit 0, no diagnostics. - -## Completion decision - -Common-security phases 10 through 60 are implemented, committed, independently -reviewed, and covered by a focused non-Kiro regression bundle. Kiro-specific -functional hardening is outside this goal scope. diff --git a/devlog/_fin/150_cross-platform-ci-release-gate/00_plan.md b/devlog/_fin/150_cross-platform-ci-release-gate/00_plan.md deleted file mode 100644 index 70e3521da..000000000 --- a/devlog/_fin/150_cross-platform-ci-release-gate/00_plan.md +++ /dev/null @@ -1,189 +0,0 @@ -# 150.00 — Plan: Cross-Platform CI and Release Gate - -## Goal - -Add the smallest useful CI surface for opencodex: - -- run typecheck and the existing Bun test suite on Linux and Windows; -- keep the Release workflow manual and publish-focused; -- require a successful Cross-platform CI run for the exact commit being released; -- keep `scripts/release.ts` compatible by waiting for that CI run after pushing a version bump; -- document the workflow change and verify it before push. - -This phase is C4 because it changes release governance and npm publishing gates. The implementation -must stay intentionally small: no coverage, no docs build, no GUI build in normal CI, no macOS matrix, -and no remote Ubuntu/RDP smoke in CI. - -## Sources Checked - -- `structure/06_docs-and-release.md` - - npm release is managed by `scripts/release.ts` and `.github/workflows/release.yml`. - - docs deploy is separate from npm release publishing. -- `.github/workflows/release.yml` - - manual `workflow_dispatch`; - - version/package check; - - npm Trusted Publishing through OIDC; - - post-publish npm smoke; - - GitHub release creation. -- `.github/workflows/deploy-docs.yml` - - docs-only GitHub Pages workflow, intentionally separate. -- `scripts/release.ts` - - clean main + typecheck; - - bump package.json; - - commit/push; - - immediately dispatch Release and watch. -- `devlog/70_windows-linux-support/00_overview.md` - - Windows/Linux support is an explicit project concern. -- `devlog/80_windows-codex-path-hardening/00_overview.md` - - Codex path handling and service behavior need cross-platform protection. -- `devlog/mvp/65_npm-publish-ci/00_plan.md` - - release flow is jawcode-style, manual dispatch plus OIDC Trusted Publishing. - -## PABCD Cycle Map - -### P — Plan - -New: - -- `.github/workflows/ci.yml` -- `devlog/150_cross-platform-ci-release-gate/00_plan.md` - -Modify: - -- `.github/workflows/release.yml` -- `scripts/release.ts` -- `structure/06_docs-and-release.md` -- commit note: `devlog/` is ignored by `.gitignore`, so this plan file must be staged with - `git add -f devlog/150_cross-platform-ci-release-gate/00_plan.md` unless the ignore policy is - deliberately changed. This phase will not change `.gitignore`. - -Non-goals: - -- no release workflow test rerun; -- no macOS CI initially; -- no GUI/docs build in normal CI; -- no coverage or E2E gates; -- no npm publish dry-run in normal CI. - -### A — Plan Audit - -Use a read-only auditor to check: - -- the planned files exist where expected, except the new CI/devlog files; -- `scripts/release.ts` can safely call GitHub CLI and parse run status; -- release workflow can use `GH_TOKEN` and read Actions metadata; -- the Release workflow will not publish without successful CI for `GITHUB_SHA`; -- the helper script will not race by dispatching Release before CI has completed. - -### B — Build - -#### New `.github/workflows/ci.yml` - -Create a single cross-platform CI workflow: - -- name: `Cross-platform CI`; -- triggers: - - `pull_request` to `main`; - - `push` to `main`; - - `workflow_dispatch`; -- path filters for code/test/package/workflow files only; -- permissions: `contents: read`; -- concurrency: cancel superseded runs per ref; -- matrix: - - `ubuntu-latest`; - - `windows-latest`; -- timeout: 8 minutes; -- steps: - - checkout; - - setup Bun latest; - - `bun install --frozen-lockfile`; - - `bun x tsc --noEmit`; - - `bun test tests`; - - `bun build scripts/release.ts --target=bun --outdir=.tmp/ci-release-script-check`; - - `bun run src/cli.ts help` CLI entrypoint smoke. - -#### Modify `.github/workflows/release.yml` - -Add a pre-publish job/step that: - -- runs before npm publish; -- rejects non-`main` refs; -- grants `actions: read` because the explicit workflow token permissions otherwise cannot read - workflow run metadata; -- sets `GH_TOKEN: ${{ github.token }}` on the gate step because the later GitHub release step's - environment is step-local and is not inherited; -- uses `gh run list --workflow ci.yml --commit "$GITHUB_SHA" --status success`; -- requires at least one successful `Cross-platform CI` run for exactly `GITHUB_SHA`; -- prints the matching CI run URL for auditability; -- does not run tests itself. - -Keep the existing: - -- manual inputs; -- version/package match check; -- npm Trusted Publishing; -- npm registry smoke; -- GitHub release creation. - -#### Modify `scripts/release.ts` - -Keep the existing local `bun x tsc --noEmit` preflight. After pushing the release commit: - -- resolve `HEAD`; -- poll GitHub Actions for `ci.yml` runs on that SHA; -- wait until success before dispatching Release; -- fail if any matching CI run completes with a non-success conclusion; -- before dispatch, verify `origin/main` still points to the same release SHA; -- dispatch with `gh workflow run release.yml --ref main ...` only after that SHA check; -- fail after a bounded timeout instead of dispatching an unsafe Release. - -This keeps the existing helper usable while preserving the invariant that Release only publishes a -CI-passed commit. - -#### Modify `structure/06_docs-and-release.md` - -Record the maintained rule: - -- Cross-platform CI is the ordinary quality gate; -- Release is manual and publish-only; -- Release requires a successful Cross-platform CI run for the release commit; -- docs deploy remains separate. - -### C — Check - -Local checks: - -- `bun x tsc --noEmit` -- `bun test tests` -- `bun build scripts/release.ts --target=bun --outdir=.tmp/ci-release-script-check` -- `bun run src/cli.ts help` -- YAML presence/structure review: - - `.github/workflows/ci.yml` - - `.github/workflows/release.yml` -- release helper syntax/bundling compatibility through the Bun build smoke above. Root - `bun x tsc --noEmit` intentionally checks `src/` only and does not typecheck `scripts/`. - -Remote checks after push: - -- push `main`; -- confirm GitHub `Cross-platform CI` run starts for the pushed commit; -- watch it to success; -- verify `Release` is still manually dispatchable and now has the CI gate. - -If the Cross-platform CI run is still running when local work completes, register it as a -server-owned `cli-jaw bgtask` instead of leaving an in-flight process attached to the turn. - -### D — Done - -Record: - -- files changed; -- local verification output; -- GitHub Actions run URL/result; -- final git status; -- any residual limitation. - -Expected residual limitation: - -- macOS is intentionally not in CI yet. Add it only if a future macOS-only break escapes local - development or the project starts shipping native macOS-specific behavior. diff --git a/devlog/_fin/150_cross-platform-ci-release-gate/10_verification.md b/devlog/_fin/150_cross-platform-ci-release-gate/10_verification.md deleted file mode 100644 index f24070313..000000000 --- a/devlog/_fin/150_cross-platform-ci-release-gate/10_verification.md +++ /dev/null @@ -1,68 +0,0 @@ -# 150.10 — Verification: Cross-Platform CI and Release Gate - -## Local Verification - -Ran: - -```bash -bun install --frozen-lockfile -bun x tsc --noEmit -bun test tests -bun build scripts/release.ts --target=bun --outdir=.tmp/ci-release-script-check -bun run src/cli.ts help -ruby -e 'require "yaml"; ARGV.each { |p| YAML.load_file(p); puts "ok #{p}" }' .github/workflows/ci.yml .github/workflows/release.yml -``` - -Results: - -- dependency check: pass, no lockfile changes; -- typecheck: pass; -- test suite: pass, 92 tests; -- release helper build smoke: pass; -- CLI help smoke: pass; -- workflow YAML parse: pass. - -## Plan Audit - -Initial read-only audit found three blocking risks: - -- Release workflow needed `actions: read` for `gh run list` with explicit token permissions. -- `scripts/release.ts` needed to ensure the SHA it waited on was the SHA it dispatched. -- Root `tsc` does not include `scripts/`, so release helper verification needed a separate smoke. - -The plan was revised to: - -- add `actions: read`; -- set `GH_TOKEN` on the release-gate step; -- wait for `ci.yml` success on the pushed release SHA; -- verify `origin/main` still equals the release SHA before `gh workflow run release.yml --ref main`; -- verify `scripts/release.ts` with `bun build`. - -Final read-only audit verdict: PASS. - -## Build Verification - -Read-only build verification confirmed: - -- `.github/workflows/ci.yml` contains Linux/Windows matrix and the intended short command set. -- `.github/workflows/release.yml` gates publish on a successful `ci.yml` run for `GITHUB_SHA`. -- `scripts/release.ts` waits for CI, checks `origin/main`, then dispatches Release on `main`. -- `structure/06_docs-and-release.md` documents the CI/release split. - -The verifier returned `NEEDS_FIX` only because the new workflow/devlog files had not yet been staged, -committed, pushed, or remote-CI-verified at that point. The remaining steps are commit, push, and -GitHub Actions verification. - -## Remote Verification Plan - -After push: - -1. Confirm the new `Cross-platform CI` workflow starts for the pushed head SHA. -2. Watch the run to completion. -3. Record run URL/result here or in the final task summary. - -## Residual Limitations - -- macOS is intentionally not in the matrix. -- The release workflow still performs `prepublishOnly` during dry-run/publish because npm publish - semantics and GUI packaging need that check; ordinary CI remains short. diff --git a/devlog/_fin/160_dashboard-redesign-and-media-models/00_overview.md b/devlog/_fin/160_dashboard-redesign-and-media-models/00_overview.md deleted file mode 100644 index 95cdefdb1..000000000 --- a/devlog/_fin/160_dashboard-redesign-and-media-models/00_overview.md +++ /dev/null @@ -1,81 +0,0 @@ -# 160.00 — Overview / MOC: Dashboard Redesign + Media Models - -- **Status:** Planning (scaffold) — not yet started -- **Date:** 2026-06-20 -- **Work class:** C3 (cross-domain: `gui/` frontend redesign + possible `src/` registry/metadata/API change) -- **Owner:** boss (frontend redesign), pending direction confirmation - -## Goal (plain language) - -Two related pieces of work, tracked together because they touch the same surface — the -page the proxy serves at `http://localhost:10100`: - -1. **Redesign the 10100 dashboard.** The current GUI works but the user finds it "too ugly" - (구리다). Refresh the visual design of the management dashboard (`gui/`). -2. **Surface Grok's image/video models.** The user states Grok now has "image and video - models on in the registry," and the dashboard should reflect that. Today the dashboard - shows only a model id + provider — no capability/modality at all. - -## Two workstreams - -| # | Workstream | Primary surface | Risk | -|---|------------|-----------------|------| -| A | Dashboard visual redesign | `gui/src/**` (pages + `styles.css`) | C2–C3, reversible (prototype-friendly) | -| B | Model capability / media-model surfacing | `src/generated/jawcode-model-metadata.ts`, `src/server.ts` (`/api/models`), `gui/src/pages/{Models,Dashboard}.tsx` | C3 — touches a generated metadata contract; treat with extra care | - -Workstream A can ship independently of B. B depends on resolving the open question below. - -## Current state (evidence) - -- The proxy serves the GUI at `GET /` → `src/server.ts:692` (`GET / → GUI dashboard`); - static files via `serveGuiFile()` / `src/server.ts:58-67`. -- GUI is a Vite + React app: `gui/src/` — `App.tsx` (shell + 5-item sidebar nav, - `App.tsx:13-19`), pages `Dashboard / Providers / Models / Subagents / Logs`, shared - primitives in `ui.tsx`, design system in `styles.css` (242 lines, dark "terminal devtool" - theme, single purple accent `#7c5cff`). Full audit → [01_current-design-audit.md](01_current-design-audit.md). -- Grok (`xai`) registry entry lists **text + vision chat** models only — - `src/providers/registry.ts:41-54` (`grok-4.3`, `grok-4.20-*`, `grok-build-0.1`, - `grok-composer-2.5-fast`; with `noVisionModels`). -- Model metadata schema tracks **input** modality only: `input?: ("text" | "image")[]` - (`src/generated/jawcode-model-metadata.ts:9`). Several xai rows are `text,image` - (vision input). There is **no `video` modality and no image/video generation concept** - anywhere in `src/` (`grep -rn "video" src` → 0 hits). -- `/api/models` returns `{ provider, id, namespaced, disabled }` only — no modality field - (`src/server.ts:419-426`). So the GUI cannot display capability today even though the - metadata partially has it. Full analysis → [02_grok-media-models-and-registry-gap.md](02_grok-media-models-and-registry-gap.md). - -## Open questions (must resolve before Workstream B) - -1. **What does "image and video models on in the registry" mean concretely?** - - (a) Vision = image **input** capability (already partly modeled as `text,image`), just - not surfaced in the UI → pure surfacing task, no schema change; or - - (b) Add Grok image-**generation** and video-**generation** models (e.g. Aurora / - image / video endpoints) to the registry → needs a new modality concept + metadata - schema extension + adapter support; or - - (c) "Registry" refers to an upstream/jawcode bundle list, not `registry.ts`. - Current code matches (a) partially; (b)/(c) are not implemented. **Do not assume — confirm.** -2. **Redesign direction** — keep the dark devtool theme and refine it, or change direction - (e.g. light/dense, Linear-like, etc.)? See Design Read in 01. - -## Document index - -- [01_current-design-audit.md](01_current-design-audit.md) — current GUI inventory, design - tokens, per-page audit, Design Read + dials, redesign opportunities. -- [02_grok-media-models-and-registry-gap.md](02_grok-media-models-and-registry-gap.md) — - what the registry/metadata actually contain, the exact gap vs the request, options. -- `10_*` Phase 1 (redesign implementation) — **to be written in P after direction is confirmed.** -- `20_*` Phase 2 (media-model surfacing) — **to be written in P after Q1 is resolved.** - -## Source-of-truth references (read, reused) - -- `structure/05_gui-and-management-api.md` — existing SOT for the GUI + management API. -- `structure/03_catalog-and-subagents.md` — catalog/model routing. -- Prior catalog work: `devlog/130_provider-catalog-single-source/`, - `devlog/140_remaining-provider-ports/`. - -## Next steps - -1. Confirm answers to the two open questions (design direction + media-model meaning). -2. `cli-jaw orchestrate P` → write `10_*` (redesign) and, if Q1 = option (b), `20_*` - (schema + API + UI) diff-level phase docs. -3. Build redesign first (independent, low risk); media-model surfacing second. diff --git a/devlog/_fin/160_dashboard-redesign-and-media-models/01_current-design-audit.md b/devlog/_fin/160_dashboard-redesign-and-media-models/01_current-design-audit.md deleted file mode 100644 index 9100e9dde..000000000 --- a/devlog/_fin/160_dashboard-redesign-and-media-models/01_current-design-audit.md +++ /dev/null @@ -1,119 +0,0 @@ -# 160.01 — Current Design Audit (the "10100" dashboard) - -Honest audit of what the proxy serves at `http://localhost:10100`. Evidence-based; every -claim points at a file. Goal: separate **what is fine** (keep) from **what reads as dated / -generic** (the actual target of "너무 구리다"). - -## What the 10100 page is - -A Vite + React SPA under `gui/`, served as static files by the proxy (`src/server.ts:692`, -`serveGuiFile` at `src/server.ts:58-67`). Build output goes to `gui/dist` and is bundled -into the npm package. - -``` -gui/src/ - App.tsx # shell: 232px sidebar + main, 5 nav items (58 lines) - main.tsx # React root (10) - ui.tsx # Switch, Notice, EmptyState primitives (31) - icons.tsx # 20 inline Lucide-style SVG icons, no dependency (29) - styles.css # the whole design system, CSS custom properties (242) - pages/ - Dashboard.tsx # status stats + provider table + model cards (114) - Providers.tsx # OAuth login panel + provider cards + JSON edit (240) - Models.tsx # per-provider collapsible model toggles (122) - Subagents.tsx # pick/order up to 5 spawn_agent models (149) - Logs.tsx # request log table, auto-refresh (73) - components/ - AddProviderModal.tsx (282) -``` - -## Design tokens (current — `gui/src/styles.css:7-43`) - -| Token | Value | Note | -|-------|-------|------| -| `--bg` | `#0b0b0f` | off-black canvas (good — not pure `#000`) | -| `--surface` / `--raised` | `#14141a` / `#1c1c25` | flat dark greys | -| `--accent` | `#7c5cff` (purple) | single accent + `--accent-hover` `#9077ff` | -| status | green `#34d399`, red `#f87171`, amber `#fbbf24` | only used for status dots / log codes | -| `--radius` | 12 / 8 / 6px | rounded | -| fonts | system sans + `ui-monospace` stack | mono for ids/urls | -| bg accent | one faint radial purple glow top-left | the only "decoration" | - -## What is already good (keep) - -- **No AI-slop tells:** real inline SVG icons (`icons.tsx`), **no emoji as UI** (dev-frontend §5 STRICT ✓), - all colors are CSS custom properties (theme-ready), off-black not pure black. -- **Accessibility baseline mostly present:** semantic `