From c64b5ca877acfbcfcb6163c8fe85550441d5375e Mon Sep 17 00:00:00 2001 From: Eldar <112889004+e1daru@users.noreply.github.com> Date: Tue, 29 Sep 2026 13:33:03 +0100 Subject: [PATCH 1/4] ci(leakcheck): scan every commit a push publishes, and read dotfiles (main) Published form of #26: the rebuilt bundles without their inline sourcemap line, and the docs, skills and manifests that ship. Sources and tests stay on pre-main. Co-Authored-By: Claude Opus 5.5 --- .githooks/pre-push | 45 +++++++++++++++++------ .github/scripts/leakcheck.mjs | 3 +- .github/scripts/leakcheck.selftest.mjs | 51 ++++++++++++++++++++++++++ 3 files changed, 87 insertions(+), 12 deletions(-) diff --git a/.githooks/pre-push b/.githooks/pre-push index 392365a..ea4b754 100755 --- a/.githooks/pre-push +++ b/.githooks/pre-push @@ -6,8 +6,9 @@ # workflow starts, GitHub is already serving the commit, and a later force-push does not # un-publish a blob that anyone can still fetch by SHA. CI here is a detector. This is the gate. # -# It scans **the commits being pushed**, not the working tree, because pushing a branch you are -# not standing on is ordinary and scanning the wrong tree would pass. +# It scans **every commit being pushed**, not the working tree and not only the tip: pushing a +# branch you are not standing on is ordinary, and a commit that adds a leak stays fetchable by +# SHA after a later one deletes it. # # Install (once per clone; it covers every worktree): # @@ -42,20 +43,42 @@ fi status=0 checked='' +# Files a commit adds or changes relative to its first parent; every file for a root commit. +changed_files() { + git diff --name-only -z --diff-filter=AMRC "$1^" "$1" 2>/dev/null || git ls-tree -r -z --name-only "$1" +} + while read -r local_ref local_sha remote_ref remote_sha; do [ -n "${local_sha:-}" ] || continue [ "$local_sha" = "$ZERO" ] && continue # deleting a remote branch - # Several refs in one push often share a tip; scan each commit once. - case " $checked " in *" $local_sha "*) continue ;; esac - checked="$checked $local_sha" - - short=$(git rev-parse --short "$local_sha" 2>/dev/null || echo "$local_sha") - printf 'leakcheck: scanning %s (%s)\n' "$short" "${local_ref##*/}" >&2 - - if ! node "$SCANNER" --strict --quiet --rev "$local_sha" --no-annotations >&2; then - status=1 + # Every commit the remote does not have yet is published, not only the tip: a leak added in + # one commit and deleted in the next is still fetchable by SHA. A remote tip we cannot resolve + # (someone else pushed) falls back to "not on any remote-tracking branch". + if [ "$remote_sha" = "$ZERO" ] || ! commits=$(git rev-list "$remote_sha..$local_sha" 2>/dev/null); then + commits=$(git rev-list "$local_sha" --not --remotes 2>/dev/null) fi + [ -n "$commits" ] || commits=$local_sha + + for sha in $commits; do + # Several refs in one push often share commits; scan each one once. + case " $checked " in *" $sha "*) continue ;; esac + checked="$checked $sha" + + short=$(git rev-parse --short "$sha" 2>/dev/null || echo "$sha") + + # The tip gets the whole-tree scan. An earlier commit can only add a leak in a file it + # touched, so it is scanned on those files alone — a full scan per commit is ~20 s. + if [ "$sha" = "$local_sha" ]; then + printf 'leakcheck: scanning %s (%s)\n' "$short" "${local_ref##*/}" >&2 + node "$SCANNER" --strict --quiet --rev "$sha" --no-annotations >&2 || status=1 + else + [ -n "$(changed_files "$sha" | tr -d '\0')" ] || continue + printf 'leakcheck: scanning %s (%s, changed files)\n' "$short" "${local_ref##*/}" >&2 + changed_files "$sha" | xargs -0 node "$SCANNER" --strict --quiet --no-annotations --rev "$sha" --paths >&2 \ + || status=1 + fi + done done if [ "$status" -ne 0 ]; then diff --git a/.github/scripts/leakcheck.mjs b/.github/scripts/leakcheck.mjs index 5f8919e..2159cd6 100644 --- a/.github/scripts/leakcheck.mjs +++ b/.github/scripts/leakcheck.mjs @@ -267,7 +267,8 @@ function isTextPath(path) { const ext = extname(path).toLowerCase(); if (CONFIG.textExtensions.includes(ext)) return true; // Extensionless files that are conventionally text. - return ['LICENSE', 'README', 'Makefile', 'Dockerfile'].includes(path.split('/').pop() || ''); + return ['LICENSE', 'README', 'Makefile', 'Dockerfile', '.gitignore', '.gitattributes', '.npmignore', + '.npmrc', '.editorconfig', '.nvmrc'].includes(path.split('/').pop() || ''); } function ruleApplies(rule, path) { diff --git a/.github/scripts/leakcheck.selftest.mjs b/.github/scripts/leakcheck.selftest.mjs index 65fe717..92d0fc3 100644 --- a/.github/scripts/leakcheck.selftest.mjs +++ b/.github/scripts/leakcheck.selftest.mjs @@ -229,6 +229,12 @@ const CASES = [ ].join('\n'), expect: ['tenancy-collapse', 'isolation-defect-disclosure'], }, + { + // Extensionless dotfiles are text, and they are published like everything else. + path: 'b/.gitignore', + body: '# bundles are committed artifacts (build-guide section 11)\n*.local\n', + expect: ['internal-doc-reference'], + }, ]; let failures = 0; @@ -315,6 +321,51 @@ try { rmSync(root, { recursive: true, force: true }); } +/** + * A push publishes every commit it carries, not only the tip, and a blob stays fetchable by SHA + * after a later commit deletes it. So the hook must refuse a branch whose leak lives only in an + * intermediate commit — and must still let a clean branch through. + */ +function checkPrePushScansEveryCommit() { + const repo = mkdtempSync(join(tmpdir(), 'leakcheck-prepush-')); + const g = (...a) => execFileSync('git', ['-C', repo, ...a], { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }); + const push = (sha) => { + try { + execFileSync('sh', [join(repo, '.githooks/pre-push'), 'origin', 'https://example.invalid/r.git'], { + cwd: repo, input: `refs/heads/x ${sha} refs/heads/x ${'0'.repeat(40)}\n`, stdio: ['pipe', 'pipe', 'pipe'], + }); + return 0; + } catch (e) { + return /** @type {any} */ (e).status ?? 1; + } + }; + try { + g('init', '-q'); + g('config', 'user.email', 'selftest@example.com'); + g('config', 'user.name', 'selftest'); + cpSync(SOURCE_GITHUB, join(repo, '.github'), { recursive: true }); + rmSync(join(repo, '.github/leakcheck/baseline.json'), { force: true }); + cpSync(join(SOURCE_GITHUB, '..', '.githooks'), join(repo, '.githooks'), { recursive: true }); + writeFileSync(join(repo, 'README.md'), 'A clean tree.\n'); + g('add', '-A'); + g('commit', '-qm', 'clean'); + const clean = g('rev-parse', 'HEAD').trim(); + if (push(clean) !== 0) fail('pre-push refused a branch with no findings'); + + writeFileSync(join(repo, 'leak.mjs'), '// see crates/control/src/overlay.rs\nexport {};\n'); + g('add', '-A'); + g('commit', '-qm', 'leak'); + rmSync(join(repo, 'leak.mjs')); + g('add', '-A'); + g('commit', '-qm', 'remove it again'); + const tip = g('rev-parse', 'HEAD').trim(); + if (push(tip) === 0) fail('pre-push passed a branch whose leak is only in an intermediate commit'); + } finally { + rmSync(repo, { recursive: true, force: true }); + } +} +checkPrePushScansEveryCommit(); + if (failures) { process.stdout.write(`\nleakcheck.selftest: ${failures} failure(s) — the gate is not seeing what it claims to.\n`); process.exit(1); From f892d581d785e13668f809baf50d8375606fad4a Mon Sep 17 00:00:00 2001 From: Eldar <112889004+e1daru@users.noreply.github.com> Date: Tue, 29 Sep 2026 13:33:03 +0100 Subject: [PATCH 2/4] feat(scorecard): session log, memory handles, settings and the card renderer (main) Published form of #27: the rebuilt bundles without their inline sourcemap line, and the docs, skills and manifests that ship. Sources and tests stay on pre-main. Co-Authored-By: Claude Opus 5.5 --- README.md | 6 +- .../claude-code/.claude-plugin/plugin.json | 4 + integrations/claude-code/README.md | 2 + integrations/claude-code/bin/activity.mjs | 14 ++- integrations/claude-code/bin/admin.mjs | 14 ++- integrations/claude-code/bin/dashboard.mjs | 14 ++- integrations/claude-code/bin/handoff.mjs | 14 ++- .../claude-code/bin/impl/statusline.mjs | 14 ++- integrations/claude-code/bin/import.mjs | 14 ++- integrations/claude-code/bin/pin.mjs | 14 ++- integrations/claude-code/docs/user-guide.md | 105 +++++++++++++++++- .../claude-code/hooks/dist/impl/capture.mjs | 14 ++- .../hooks/dist/impl/checkpoint.mjs | 14 ++- .../hooks/dist/impl/cwd-changed.mjs | 14 ++- .../claude-code/hooks/dist/impl/drain.mjs | 17 ++- .../claude-code/hooks/dist/impl/pre-tool.mjs | 14 ++- .../hooks/dist/impl/prompt-recall.mjs | 14 ++- .../hooks/dist/impl/recall-refresh.mjs | 14 ++- .../hooks/dist/impl/session-end.mjs | 17 ++- .../hooks/dist/impl/session-resume.mjs | 14 ++- .../hooks/dist/impl/session-start.mjs | 14 ++- .../hooks/dist/impl/stage-prompt.mjs | 14 ++- .../hooks/dist/impl/subagent-start.mjs | 14 ++- integrations/claude-code/mcp/dist/index.js | 14 ++- integrations/codex/README.md | 2 + integrations/codex/bin/activity.mjs | 14 ++- integrations/codex/bin/admin.mjs | 14 ++- integrations/codex/bin/dashboard.mjs | 14 ++- integrations/codex/bin/handoff.mjs | 14 ++- integrations/codex/bin/import.mjs | 14 ++- integrations/codex/bin/pin.mjs | 14 ++- integrations/codex/docs/user-guide.md | 13 +++ .../codex/hooks/dist/impl/capture.mjs | 14 ++- .../codex/hooks/dist/impl/checkpoint.mjs | 14 ++- integrations/codex/hooks/dist/impl/drain.mjs | 17 ++- .../codex/hooks/dist/impl/pre-tool.mjs | 14 ++- .../codex/hooks/dist/impl/prompt-recall.mjs | 14 ++- .../codex/hooks/dist/impl/recall-refresh.mjs | 14 ++- .../codex/hooks/dist/impl/session-end.mjs | 17 ++- .../codex/hooks/dist/impl/session-resume.mjs | 14 ++- .../codex/hooks/dist/impl/session-start.mjs | 14 ++- .../codex/hooks/dist/impl/stage-prompt.mjs | 14 ++- .../codex/hooks/dist/impl/subagent-start.mjs | 14 ++- integrations/codex/mcp/dist/index.js | 14 ++- 44 files changed, 635 insertions(+), 41 deletions(-) diff --git a/README.md b/README.md index 38d2a92..5c221fb 100644 --- a/README.md +++ b/README.md @@ -359,8 +359,10 @@ the default. The ones worth knowing about: | `recallAsync` | `false` | Never make a prompt wait on recall, at the cost of one turn of staleness. | | `preToolWarnings` | `false` | Show the model a stored rule just before an `rm` or `git push`. It only ever warns. | | `capture` / `recall` | `true` | Turn either half off. | +| `sessionScore` | `full` | The memory scorecard printed under each reply that showed a lesson. `compact` for one line, `off` to hide it. | +| `outcomeReview` | `stop` | Asks Claude once per turn to credit the lessons that helped or misled it. `nudge` keeps only a one-line ask. | -The [Claude Code guide](integrations/claude-code/README.md#configuration) documents all 25. +The [Claude Code guide](integrations/claude-code/README.md#configuration) documents all 27. ## When something looks wrong @@ -409,7 +411,7 @@ Both hosts execute these directories as fetched, with no build step, which is wh ## Documentation and support - **Guides** — [Claude Code](integrations/claude-code/README.md) · - [Codex CLI](integrations/codex/README.md). Install, all 25 options, and troubleshooting. + [Codex CLI](integrations/codex/README.md). Install, all 27 options, and troubleshooting. - **Reference** — [docs.mubit.ai](https://docs.mubit.ai) for the Mubit API, SDKs and console. - **Keys and instances** — the [Mubit console](https://console.mubit.ai). - **Bugs** — [open an issue](https://github.com/mubit-ai/plugins/issues). What to put in diff --git a/integrations/claude-code/.claude-plugin/plugin.json b/integrations/claude-code/.claude-plugin/plugin.json index 23c46af..7489553 100644 --- a/integrations/claude-code/.claude-plugin/plugin.json +++ b/integrations/claude-code/.claude-plugin/plugin.json @@ -45,6 +45,10 @@ "description": "Let the end-of-session drain and reflection finish in a detached process. The host cancels the session-end hook about a second in when a session is torn down, and everything still running inside it dies with it — including the only call that promotes a lesson beyond its own run. Turning this off keeps that work in the hook, where a teardown can cut it short." }, "outcomeMode": { "type": "string", "title": "Outcome attribution", "default": "implicit", "description": "How turn outcomes are attributed back to recalled memories: off, implicit, or explicit." }, + "sessionScore": { "type": "string", "title": "Session memory scorecard", "default": "full", + "description": "After each turn that showed a lesson, print a scorecard under Claude's reply: lessons shown this session, how many the replies used, and whether those turns worked, failed or are waiting on your reply. 'full' is a short tree, 'compact' one line, 'off' nothing. Local only: it reads a log on disk and makes no network call." }, + "outcomeReview": { "type": "string", "title": "Outcome review", "default": "stop", + "description": "How Claude is asked to credit the memory it used. Every injected memory line starts with a short id like [m7k2q] that mubit_outcome accepts. 'nudge' adds one sentence asking Claude to credit what helped or misled it before finishing, and keeps mubit_outcome and mubit_learned loaded. 'stop' also has the Stop hook ask once per turn for a short review of that turn's lessons; it costs one extra short step, and Claude Code labels that step 'Stop hook error occurred' although nothing failed. 'off' does neither." }, "statusLine": { "type": "boolean", "title": "Status line", "default": true, "description": "Show Mubit connection and capture stats in the status line." }, "preToolWarnings": { "type": "boolean", "title": "Warn before matching tool calls", "default": false, diff --git a/integrations/claude-code/README.md b/integrations/claude-code/README.md index 5fe77e0..5cd82a4 100644 --- a/integrations/claude-code/README.md +++ b/integrations/claude-code/README.md @@ -319,6 +319,8 @@ that cache, and writing credentials invalidates it immediately rather than after | `reflectOnEnd` | `true` | `MUBIT_CC_REFLECT_ON_END` | Reflect at `SessionEnd`. This is the only path that promotes a lesson beyond its own run, so turning it off to save a few seconds trades away cross-session memory entirely. See below. | | `sessionEndDetach` | `true` | `MUBIT_CC_SESSION_END_DETACH` | Let the end-of-session drain and reflection finish in a detached process. The host cancels the `SessionEnd` hook about a second into a teardown — under `--print` it always does — and anything still running inside the hook dies with it, including the reflect above. On, the hook stamps the marker `detached`, hands the work over and returns in milliseconds; the child reports a terminal `reflect.status` when it is done, usually a few seconds after the CLI has exited. Turn it off only where background processes are forbidden — the work then runs inline, where a teardown can cut it short. | | `outcomeMode` | `implicit` | `MUBIT_CC_OUTCOME_MODE` | `implicit`: a turn whose reply carried the recalled memory's own vocabulary is attributed to those memories; a turn that carried none of it is recorded as `neutral` against the run and attributed to no entry, so an injection nobody used is counted rather than being invisible. `explicit`: only the model's own `mubit_outcome` calls count. `off`: no attribution, and no measurement of it either. | +| `sessionScore` | `full` (`off` under Codex) | `MUBIT_CC_SESSION_SCORE` | After each turn that showed a lesson, print a scorecard under the reply: lessons shown this session, how many replies used, and whether those turns worked, failed or are waiting on your reply. `full` is a short tree, `compact` one line, `off` nothing. It folds a local log (`scorecard/.jsonl`, 7 days) and makes no network call; nothing is shown while `capture` is off. See [the session scorecard](docs/user-guide.md#the-session-scorecard). | +| `outcomeReview` | `stop` (`nudge` under Codex) | `MUBIT_CC_OUTCOME_REVIEW` | How hard Claude is asked to credit the memory it used. Every injected memory line starts with a short id such as `[m7k2q]`, which `mubit_outcome` accepts and the plugin maps back to the entry's reference id. `nudge` adds one sentence asking Claude to credit what helped or misled it, and keeps `mubit_outcome` and `mubit_learned` loaded rather than deferred behind tool search. `stop` also has the Stop hook ask Claude once per turn to review that turn's lessons — one extra short step, which Claude Code labels "Stop hook error occurred" although nothing failed. `off` does neither. No review runs under `outcomeMode: off`. See [crediting memory by id](docs/user-guide.md#crediting-memory-by-id-the-outcome-review). | | `statusLine` | `true` | `MUBIT_CC_STATUSLINE` | Render the status line. When false it prints an empty line and exits 0 rather than erroring per frame. | | `preToolWarnings` | `false` | `MUBIT_CC_PRE_TOOL_WARNINGS` | Show the model a matching stored `rule` just before an `rm` or `git push` runs. Warnings only — it never blocks, rewrites or asks about a tool call, and the filter that decides when it runs at all is best-effort, so treat it as a reminder and use Claude Code's permission system for anything that has to hold. Off by default: this is the one setting that can put text in front of a tool call. | | `resumeBlock` | `true` | `MUBIT_CC_RESUME_BLOCK` | Open a session with a briefing on where earlier work left off. `SessionStart` spawns a detached child that asks `/v2/control/context` for a sections block about this run, and the first substantive prompt of the session renders it above the ordinary recall block. **The one opt-in feature here that ships on**, because its cost is per *session* and not per prompt: one background process and **2 LLM calls once**, against the prompt where the model knows least about what it is walking into — nothing waits for it, and no prompt after the first pays anything. Only `startup` and `resume` sessions get one: `/clear` starts a fresh run with no history, and a compaction or a fork is already re-anchored. It renders as `` and says, in the block, that it is a briefing and not a task list. **How much it can describe depends on `runStrategy`.** `/v2/control/context` is *mostly* run-scoped — activity, working memory, rules and archived blocks all come from the run id you give it — but lessons also reach across runs, through linked runs and a session/global lesson lane. So under the default `per-directory` the block summarises everything this project has ever done; under `per-conversation`, where every session is its own run, a new session's own run is empty and the block falls back to whatever cross-run lessons apply — thinner, but not nothing. Set `MUBIT_CC_RESUME_TOKENS` to change its 1000-token ceiling. | diff --git a/integrations/claude-code/bin/activity.mjs b/integrations/claude-code/bin/activity.mjs index 0cbc132..470c95f 100644 --- a/integrations/claude-code/bin/activity.mjs +++ b/integrations/claude-code/bin/activity.mjs @@ -424,7 +424,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -544,6 +544,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -617,6 +627,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/bin/admin.mjs b/integrations/claude-code/bin/admin.mjs index 3aca8dd..013d0c1 100755 --- a/integrations/claude-code/bin/admin.mjs +++ b/integrations/claude-code/bin/admin.mjs @@ -429,7 +429,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -549,6 +549,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -622,6 +632,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/bin/dashboard.mjs b/integrations/claude-code/bin/dashboard.mjs index 40dbbba..d5fd4d9 100644 --- a/integrations/claude-code/bin/dashboard.mjs +++ b/integrations/claude-code/bin/dashboard.mjs @@ -444,7 +444,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -564,6 +564,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -637,6 +647,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/bin/handoff.mjs b/integrations/claude-code/bin/handoff.mjs index 58906bc..79a6ffa 100644 --- a/integrations/claude-code/bin/handoff.mjs +++ b/integrations/claude-code/bin/handoff.mjs @@ -182,7 +182,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -302,6 +302,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -375,6 +385,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/bin/impl/statusline.mjs b/integrations/claude-code/bin/impl/statusline.mjs index 4d0f801..fa07f2d 100644 --- a/integrations/claude-code/bin/impl/statusline.mjs +++ b/integrations/claude-code/bin/impl/statusline.mjs @@ -276,7 +276,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -382,6 +382,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -455,6 +465,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/bin/import.mjs b/integrations/claude-code/bin/import.mjs index 012666c..cf61f44 100644 --- a/integrations/claude-code/bin/import.mjs +++ b/integrations/claude-code/bin/import.mjs @@ -492,7 +492,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; var MAX_ENV_TAGS = 8; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); @@ -676,6 +676,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -749,6 +759,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/bin/pin.mjs b/integrations/claude-code/bin/pin.mjs index e4452ab..a5d189a 100644 --- a/integrations/claude-code/bin/pin.mjs +++ b/integrations/claude-code/bin/pin.mjs @@ -192,7 +192,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -312,6 +312,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -385,6 +395,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/docs/user-guide.md b/integrations/claude-code/docs/user-guide.md index 37f9da4..9de6a76 100644 --- a/integrations/claude-code/docs/user-guide.md +++ b/integrations/claude-code/docs/user-guide.md @@ -243,7 +243,7 @@ What happens on its own: | Every prompt you send | Queries memory and injects what is relevant, within a 1500 ms budget and a 1500-token cap. **Zero LLM calls** — assembly is local | | Every tool call | Redacts and spools it. Zero network on the hot path | | Every tool failure | Captured — these produce the most useful lessons | -| Every turn ends | Writes the `Q: … / A: …` pair, flushes the spool, and attributes the turn's success or failure back to the memories that were injected for it | +| Every turn ends | Writes the `Q: … / A: …` pair, flushes the spool, credits the memories the reply actually used, and — on turns that showed a lesson — asks Claude to review those lessons and prints the [session scorecard](#the-session-scorecard) | | Before a compact | Snapshots the last 200 KB of transcript before the host throws it away | | Session ends | Drains, flushes outcomes, then reflects — extracting lessons that outlive this run | @@ -268,6 +268,97 @@ costs you a memory, never a turn. Groups are hidden while still zero. It reads two local JSON files and never touches the network, so a dead server can never freeze your terminal. +### The session scorecard + +After each turn that showed Claude at least one lesson, a short card appears under the reply. +Claude Code prefixes it with `Stop says:`. + +``` +mubit · this session · lessons on 6 of 8 prompts · +2 learned · memory added 6.4k tok + 9 lessons shown + ├ 5 used 3 worked · 1 failed · 1 waiting on your reply + └ 4 not used + this turn: used "run vitest with --pool=forks" + review: "use npm ci in CI" failed on prompt 5 · 2 lessons shown 3+ times and never used +``` + +The header counts the prompts in this session that carried a lesson, the lessons Claude saved +with `mubit_learned`, and the tokens memory added to the context (standing lessons plus every +per-prompt injection). Parts that would read zero are left out. + +The tree counts **lessons**: distinct entries of type `lesson` shown in this session. Rules, +facts and past work are injected too, but they are not counted here. Every lesson lands in +exactly one row, so the rows always add up to the total. + +| Row | Meaning | +| --- | --- | +| shown | Injected in full, or repeated as a `(seen earlier)` pointer. Standing lessons count on the first prompt after each session start. | +| used | The lesson's own distinctive words appear in Claude's reply, or Claude named it in `mubit_outcome`. Words shared with another entry shown that turn, or that you typed in your prompt, do not count. A lesson needs at least two of its words, or a quarter of them if that is more; when it has two or fewer, it needs all of them. | +| unknown | Could not be checked: the lesson had no distinctive words, there was no reply, the API failed, or you interrupted the turn. Shown only when non-zero. | +| not used | Everything else. This is not a penalty: a lesson like "never force-push" is followed by *not* doing something, which a word match cannot see. The outcome review below is what catches those. | + +A used lesson gets a verdict for the turn it was used in. The first rule that applies wins: + +1. **Claude's own verdict** through `mubit_outcome`: success or partial means worked, failure + means failed. +2. **Your next prompt corrects Claude** — "no, that's wrong", "still failing", "revert + that" — means failed. A correction never counts across `/clear` or for a slash command, and + a bare "no" answering a question Claude asked is an answer, not a correction. +3. **The turn's last tool call that changed something failed** means failed. Reads and + searches do not count, so a `grep` that matched nothing is not a failure. +4. **You sent another prompt** means worked. +5. Otherwise the lesson is **waiting on your reply**. + +A lesson used on several turns shows as failed if any of them failed, then waiting if any is +still waiting, and worked otherwise. `this turn` names up to two lessons this reply used. The +`review` line appears only when something failed, or when a lesson has been shown three or more +times and never used — a candidate for `/mubit-memory:forget`. + +The card is built from a local log (`scorecard/.jsonl` under the plugin data +directory, kept for 7 days) and costs no network call. Set `sessionScore` to `compact` for a +single line or `off` to hide it. Nothing is measured, and no card is shown, while `capture` is +off. + +### Crediting memory by id: the outcome review + +Mubit learns which memories help from outcomes. The plugin posts one automatically at the end +of every turn, but the model's own judgement — "this lesson helped", "this one was wrong" — +is the stronger signal, and it used to be almost impossible for Claude to give: memory arrived +as plain lines with no id to name. + +Now every injected memory line starts with a short id: + +``` +## Lessons +- [m7k2q] run vitest with --pool=forks; the thread pool hangs on the native module +- (seen earlier) [m3jd9] — use npm ci in CI… +``` + +Claude passes those ids to `mubit_outcome`, and the plugin maps each one back to the entry's +real reference id before the call leaves your machine. `outcomeReview` decides how hard Claude +is asked to do it: + +| Value | What happens | +| --- | --- | +| `stop` (default in Claude Code) | Everything `nudge` does, plus: when a turn showed lessons that were new to Claude or that its reply appeared to use, the Stop hook asks Claude once to review them. Claude credits the ones that helped (`success`), flags the ones that misled it (`failure`), saves a corrected lesson with `mubit_learned` when one was wrong, and ends with a one-line `Memory review:` summary. | +| `nudge` (default in Codex) | One sentence in the memory block asks Claude to credit what helped or misled it before finishing. In Claude Code, `mubit_outcome` and `mubit_learned` are also kept loaded rather than deferred behind tool search, so reporting costs one call instead of two. | +| `off` | No sentence, no review, and the two tools are deferred like the rest. The ids stay on the lines. | + +The review costs one extra short model step on the turns that trigger it. Claude Code labels any +continuation a Stop hook asks for as **"Stop hook error occurred"**, even though nothing failed +— that label is the review running. Set `outcomeReview` to `nudge` if you would rather not see +it. With `outcomeMode: off` there is no review at all. Codex defaults to `nudge` because a Stop +continuation has not been verified there. + +What reaches the server: + +- The automatic outcome now reinforces only the entries the reply actually used, instead of + everything that was injected that turn. +- An entry Claude credited or blamed itself is left to its verdict, so it is never counted + twice for one turn. +- When your next prompt corrects Claude, a failure (−0.3) is posted against the entries the + previous reply used. + --- ## Part 4 — Prove it is actually working @@ -652,12 +743,24 @@ on it — the plugin denies nothing at any exit it controls, and it exits 0 on e including the path where its own internal deadline fires — but if you write your own pre-tool hook, do not carry the assumption forward. +### The scorecard and the outcome review + +`sessionScore`, default **`full`** (`off` in Codex): `full` prints the card described in +[The session scorecard](#the-session-scorecard) under every reply that showed a lesson, +`compact` prints its first line only, `off` prints nothing. + +`outcomeReview`, default **`stop`** (`nudge` in Codex): how hard Claude is asked to credit the +memory it used. See [Crediting memory by id](#crediting-memory-by-id-the-outcome-review) for +what each value does and what the review costs. + ### Quieting it temporarily ```bash MUBIT_CC_CAPTURE=0 claude # stop capturing, keep recall MUBIT_CC_RECALL=0 claude # stop injecting, keep capturing MUBIT_CC_PRE_TOOL_WARNINGS=0 claude # stop the pre-command reminders (already the default) +MUBIT_CC_SESSION_SCORE=off claude # hide the scorecard under each reply +MUBIT_CC_OUTCOME_REVIEW=nudge claude # no end-of-turn review step, keep the one-line ask ``` ### Fewer MCP tools diff --git a/integrations/claude-code/hooks/dist/impl/capture.mjs b/integrations/claude-code/hooks/dist/impl/capture.mjs index f0f4bd8..3e70af3 100755 --- a/integrations/claude-code/hooks/dist/impl/capture.mjs +++ b/integrations/claude-code/hooks/dist/impl/capture.mjs @@ -239,7 +239,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE2 = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION2 = 3; +var CACHE_VERSION2 = 4; var MAX_ENV_TAGS = 8; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); @@ -409,6 +409,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -482,6 +492,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/checkpoint.mjs b/integrations/claude-code/hooks/dist/impl/checkpoint.mjs index 66adb8c..776b8d0 100755 --- a/integrations/claude-code/hooks/dist/impl/checkpoint.mjs +++ b/integrations/claude-code/hooks/dist/impl/checkpoint.mjs @@ -279,7 +279,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE2 = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION2 = 3; +var CACHE_VERSION2 = 4; var MAX_ENV_TAGS = 8; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); @@ -463,6 +463,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -536,6 +546,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/cwd-changed.mjs b/integrations/claude-code/hooks/dist/impl/cwd-changed.mjs index 7bb31fc..3de5a0e 100755 --- a/integrations/claude-code/hooks/dist/impl/cwd-changed.mjs +++ b/integrations/claude-code/hooks/dist/impl/cwd-changed.mjs @@ -183,7 +183,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -289,6 +289,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -362,6 +372,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/drain.mjs b/integrations/claude-code/hooks/dist/impl/drain.mjs index f9ece6c..752db05 100755 --- a/integrations/claude-code/hooks/dist/impl/drain.mjs +++ b/integrations/claude-code/hooks/dist/impl/drain.mjs @@ -159,6 +159,9 @@ function pruneStale(cfg = {}) { for (const name of jsonFiles(join(root, "import"))) { expire(join(root, "import", name), 30 * DAY); } + for (const e of dirEntries(join(root, "scorecard"))) { + if (e.isFile() && e.name.endsWith(".jsonl")) expire(join(root, "scorecard", e.name), 7 * DAY); + } for (const e of dirEntries(join(root, "tmp"))) { if (e.isFile()) expire(join(root, "tmp", e.name), 1 * HOUR); } @@ -639,7 +642,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE2 = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION2 = 3; +var CACHE_VERSION2 = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -759,6 +762,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -832,6 +845,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/pre-tool.mjs b/integrations/claude-code/hooks/dist/impl/pre-tool.mjs index 6f70886..cbac4d5 100755 --- a/integrations/claude-code/hooks/dist/impl/pre-tool.mjs +++ b/integrations/claude-code/hooks/dist/impl/pre-tool.mjs @@ -183,7 +183,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -289,6 +289,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -362,6 +372,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/prompt-recall.mjs b/integrations/claude-code/hooks/dist/impl/prompt-recall.mjs index 83a0200..53e18ee 100755 --- a/integrations/claude-code/hooks/dist/impl/prompt-recall.mjs +++ b/integrations/claude-code/hooks/dist/impl/prompt-recall.mjs @@ -714,7 +714,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; var MAX_ENV_TAGS = 8; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); @@ -898,6 +898,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -971,6 +981,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/recall-refresh.mjs b/integrations/claude-code/hooks/dist/impl/recall-refresh.mjs index 62a7ca6..57abda7 100755 --- a/integrations/claude-code/hooks/dist/impl/recall-refresh.mjs +++ b/integrations/claude-code/hooks/dist/impl/recall-refresh.mjs @@ -192,7 +192,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; var MAX_ENV_TAGS = 8; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); @@ -376,6 +376,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -449,6 +459,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/session-end.mjs b/integrations/claude-code/hooks/dist/impl/session-end.mjs index 3da064c..900e415 100755 --- a/integrations/claude-code/hooks/dist/impl/session-end.mjs +++ b/integrations/claude-code/hooks/dist/impl/session-end.mjs @@ -158,6 +158,9 @@ function pruneStale(cfg = {}) { for (const name of jsonFiles(join(root, "import"))) { expire(join(root, "import", name), 30 * DAY); } + for (const e of dirEntries(join(root, "scorecard"))) { + if (e.isFile() && e.name.endsWith(".jsonl")) expire(join(root, "scorecard", e.name), 7 * DAY); + } for (const e of dirEntries(join(root, "tmp"))) { if (e.isFile()) expire(join(root, "tmp", e.name), 1 * HOUR); } @@ -537,7 +540,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -657,6 +660,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -730,6 +743,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/session-resume.mjs b/integrations/claude-code/hooks/dist/impl/session-resume.mjs index 9894483..1a93c89 100755 --- a/integrations/claude-code/hooks/dist/impl/session-resume.mjs +++ b/integrations/claude-code/hooks/dist/impl/session-resume.mjs @@ -183,7 +183,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -303,6 +303,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -376,6 +386,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/session-start.mjs b/integrations/claude-code/hooks/dist/impl/session-start.mjs index e42bcf5..5d9c1fb 100644 --- a/integrations/claude-code/hooks/dist/impl/session-start.mjs +++ b/integrations/claude-code/hooks/dist/impl/session-start.mjs @@ -439,7 +439,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -559,6 +559,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -632,6 +642,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs b/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs index e8c1134..a414b93 100644 --- a/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs +++ b/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs @@ -195,7 +195,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -301,6 +301,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -374,6 +384,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/hooks/dist/impl/subagent-start.mjs b/integrations/claude-code/hooks/dist/impl/subagent-start.mjs index 221fc71..7c4d362 100755 --- a/integrations/claude-code/hooks/dist/impl/subagent-start.mjs +++ b/integrations/claude-code/hooks/dist/impl/subagent-start.mjs @@ -195,7 +195,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; var MAX_ENV_TAGS = 8; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); @@ -379,6 +379,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -452,6 +462,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/claude-code/mcp/dist/index.js b/integrations/claude-code/mcp/dist/index.js index 530d12b..1740d86 100644 --- a/integrations/claude-code/mcp/dist/index.js +++ b/integrations/claude-code/mcp/dist/index.js @@ -182,7 +182,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -302,6 +302,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -375,6 +385,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, diff --git a/integrations/codex/README.md b/integrations/codex/README.md index 6e61831..52b6b62 100644 --- a/integrations/codex/README.md +++ b/integrations/codex/README.md @@ -109,6 +109,8 @@ The settings worth knowing, all `MUBIT_CC_*` unless noted: | `MUBIT_CC_SESSION_END_DETACH` | `1` | Finish the end-of-session flush in a detached process. **Leave this on under Codex** — see below. | | `MUBIT_CC_PRE_TOOL_WARNINGS` | `0` | Show a stored rule before a matching tool call. It only ever warns. | | `MUBIT_CC_PINS` | `1` | Render the constraints pinned with the `pin` skill above the recalled block on every prompt of the run — including the prompts recall skips — and above the block a subagent is given at `SubagentStart`. Capped at five pins, 200 characters each and 240 tokens (96 for a subagent); costs no extra request on the prompt path. Off restores the injected block exactly. | +| `MUBIT_CC_SESSION_SCORE` | `off` here | The memory scorecard the Stop hook can print under each reply that showed a lesson (`full`, `compact` or `off`). Off by default under Codex, where the card has not been verified; see [the session scorecard](../claude-code/docs/user-guide.md#the-session-scorecard). | +| `MUBIT_CC_OUTCOME_REVIEW` | `nudge` here | How hard the model is asked to credit memory by the short id (`[m7k2q]`) now printed on every injected line. `nudge` is one sentence in the memory block; `stop` also asks for a review from the Stop hook, which has not been verified under Codex; `off` does neither. See [crediting memory by id](../claude-code/docs/user-guide.md#crediting-memory-by-id-the-outcome-review). | | `MUBIT_CC_DATA_DIR` | — | Overrides where state lives. Highest precedence of any data-dir input. | | `MUBIT_CC_STATUSLINE` | `0` here | Defaults **off** under Codex, whose status line is a fixed list of built-in item ids with nothing scriptable in it. | | `MUBIT_MCP_TOOLS` (no `_CC`) | — | Which MCP tools to register, comma-separated. Blank means the seven below. A list you supply is used **verbatim**, not unioned with that default, so it is also how you reach the other eight. | diff --git a/integrations/codex/bin/activity.mjs b/integrations/codex/bin/activity.mjs index e64cef2..e1ead94 100644 --- a/integrations/codex/bin/activity.mjs +++ b/integrations/codex/bin/activity.mjs @@ -551,6 +551,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -624,6 +634,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -769,7 +781,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/bin/admin.mjs b/integrations/codex/bin/admin.mjs index b08d433..e1c747a 100755 --- a/integrations/codex/bin/admin.mjs +++ b/integrations/codex/bin/admin.mjs @@ -556,6 +556,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -629,6 +639,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -774,7 +786,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/bin/dashboard.mjs b/integrations/codex/bin/dashboard.mjs index b401295..58456d3 100644 --- a/integrations/codex/bin/dashboard.mjs +++ b/integrations/codex/bin/dashboard.mjs @@ -569,6 +569,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -642,6 +652,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -787,7 +799,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/bin/handoff.mjs b/integrations/codex/bin/handoff.mjs index cd10f2d..5125880 100644 --- a/integrations/codex/bin/handoff.mjs +++ b/integrations/codex/bin/handoff.mjs @@ -306,6 +306,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -379,6 +389,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -524,7 +536,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/bin/import.mjs b/integrations/codex/bin/import.mjs index ef85e4e..5f11e4c 100644 --- a/integrations/codex/bin/import.mjs +++ b/integrations/codex/bin/import.mjs @@ -670,6 +670,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -743,6 +753,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -888,7 +900,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MAX_ENV_TAGS = 8; MODE = "hosted"; LANG_FILES = [ diff --git a/integrations/codex/bin/pin.mjs b/integrations/codex/bin/pin.mjs index 62951e1..cd3eca9 100644 --- a/integrations/codex/bin/pin.mjs +++ b/integrations/codex/bin/pin.mjs @@ -316,6 +316,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -389,6 +399,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -534,7 +546,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/docs/user-guide.md b/integrations/codex/docs/user-guide.md index 7d7a57a..e218cd0 100644 --- a/integrations/codex/docs/user-guide.md +++ b/integrations/codex/docs/user-guide.md @@ -417,6 +417,19 @@ command. It only ever warns — it never allows, denies or rewrites, on any path whether the feature is on or not, which is why `setup` leaves the registration out unless you pass `--with-pre-tool`. +### Crediting memory, and the scorecard — mostly off here + +Every injected memory line now starts with a short id such as `[m7k2q]`, which `mubit_outcome` +accepts and the plugin maps back to the entry's reference id. `MUBIT_CC_OUTCOME_REVIEW`, +default `nudge` under Codex, adds one sentence to the memory block asking the model to credit +what helped or misled it before finishing; `off` drops it. `stop` also has the Stop hook ask +for a short review once per turn — the default in Claude Code, but a Stop continuation has not +been verified under Codex. + +`MUBIT_CC_SESSION_SCORE`, default `off` under Codex, prints the memory scorecard under each +reply that showed a lesson (`full` or `compact`). The +[Claude Code guide](../../claude-code/docs/user-guide.md#the-session-scorecard) explains both. + ### Quieting it temporarily ```bash diff --git a/integrations/codex/hooks/dist/impl/capture.mjs b/integrations/codex/hooks/dist/impl/capture.mjs index c8296b3..f060dde 100755 --- a/integrations/codex/hooks/dist/impl/capture.mjs +++ b/integrations/codex/hooks/dist/impl/capture.mjs @@ -406,6 +406,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -479,6 +489,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -624,7 +636,7 @@ var init_config = __esm({ ]; CACHE_FILE2 = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION2 = 3; + CACHE_VERSION2 = 4; MAX_ENV_TAGS = 8; MODE = "hosted"; LANG_FILES = [ diff --git a/integrations/codex/hooks/dist/impl/checkpoint.mjs b/integrations/codex/hooks/dist/impl/checkpoint.mjs index c665704..bb0765b 100755 --- a/integrations/codex/hooks/dist/impl/checkpoint.mjs +++ b/integrations/codex/hooks/dist/impl/checkpoint.mjs @@ -470,6 +470,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -543,6 +553,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -688,7 +700,7 @@ var init_config = __esm({ ]; CACHE_FILE2 = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION2 = 3; + CACHE_VERSION2 = 4; MAX_ENV_TAGS = 8; MODE = "hosted"; LANG_FILES = [ diff --git a/integrations/codex/hooks/dist/impl/drain.mjs b/integrations/codex/hooks/dist/impl/drain.mjs index 3e43f01..e3879a9 100755 --- a/integrations/codex/hooks/dist/impl/drain.mjs +++ b/integrations/codex/hooks/dist/impl/drain.mjs @@ -153,6 +153,9 @@ function pruneStale(cfg = {}) { for (const name of jsonFiles(join2(root, "import"))) { expire(join2(root, "import", name), 30 * DAY); } + for (const e of dirEntries(join2(root, "scorecard"))) { + if (e.isFile() && e.name.endsWith(".jsonl")) expire(join2(root, "scorecard", e.name), 7 * DAY); + } for (const e of dirEntries(join2(root, "tmp"))) { if (e.isFile()) expire(join2(root, "tmp", e.name), 1 * HOUR); } @@ -768,6 +771,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -841,6 +854,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -986,7 +1001,7 @@ var init_config = __esm({ ]; CACHE_FILE2 = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION2 = 3; + CACHE_VERSION2 = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/hooks/dist/impl/pre-tool.mjs b/integrations/codex/hooks/dist/impl/pre-tool.mjs index 2cb8777..f005d39 100755 --- a/integrations/codex/hooks/dist/impl/pre-tool.mjs +++ b/integrations/codex/hooks/dist/impl/pre-tool.mjs @@ -292,6 +292,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -365,6 +375,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -510,7 +522,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/hooks/dist/impl/prompt-recall.mjs b/integrations/codex/hooks/dist/impl/prompt-recall.mjs index 62410b7..dd4ce06 100755 --- a/integrations/codex/hooks/dist/impl/prompt-recall.mjs +++ b/integrations/codex/hooks/dist/impl/prompt-recall.mjs @@ -904,6 +904,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -977,6 +987,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -1122,7 +1134,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MAX_ENV_TAGS = 8; MODE = "hosted"; LANG_FILES = [ diff --git a/integrations/codex/hooks/dist/impl/recall-refresh.mjs b/integrations/codex/hooks/dist/impl/recall-refresh.mjs index e5d077c..2143133 100755 --- a/integrations/codex/hooks/dist/impl/recall-refresh.mjs +++ b/integrations/codex/hooks/dist/impl/recall-refresh.mjs @@ -370,6 +370,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -443,6 +453,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -588,7 +600,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MAX_ENV_TAGS = 8; MODE = "hosted"; LANG_FILES = [ diff --git a/integrations/codex/hooks/dist/impl/session-end.mjs b/integrations/codex/hooks/dist/impl/session-end.mjs index b0b28a2..6188ab6 100755 --- a/integrations/codex/hooks/dist/impl/session-end.mjs +++ b/integrations/codex/hooks/dist/impl/session-end.mjs @@ -153,6 +153,9 @@ function pruneStale(cfg = {}) { for (const name of jsonFiles(join2(root, "import"))) { expire(join2(root, "import", name), 30 * DAY); } + for (const e of dirEntries(join2(root, "scorecard"))) { + if (e.isFile() && e.name.endsWith(".jsonl")) expire(join2(root, "scorecard", e.name), 7 * DAY); + } for (const e of dirEntries(join2(root, "tmp"))) { if (e.isFile()) expire(join2(root, "tmp", e.name), 1 * HOUR); } @@ -660,6 +663,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -733,6 +746,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -878,7 +893,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/hooks/dist/impl/session-resume.mjs b/integrations/codex/hooks/dist/impl/session-resume.mjs index 375bf5d..252f080 100755 --- a/integrations/codex/hooks/dist/impl/session-resume.mjs +++ b/integrations/codex/hooks/dist/impl/session-resume.mjs @@ -306,6 +306,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -379,6 +389,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -524,7 +536,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/hooks/dist/impl/session-start.mjs b/integrations/codex/hooks/dist/impl/session-start.mjs index 960fba1..5bf7858 100755 --- a/integrations/codex/hooks/dist/impl/session-start.mjs +++ b/integrations/codex/hooks/dist/impl/session-start.mjs @@ -564,6 +564,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -637,6 +647,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -782,7 +794,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/hooks/dist/impl/stage-prompt.mjs b/integrations/codex/hooks/dist/impl/stage-prompt.mjs index f6ed860..a05b575 100755 --- a/integrations/codex/hooks/dist/impl/stage-prompt.mjs +++ b/integrations/codex/hooks/dist/impl/stage-prompt.mjs @@ -300,6 +300,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -373,6 +383,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -518,7 +530,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MODE = "hosted"; } }); diff --git a/integrations/codex/hooks/dist/impl/subagent-start.mjs b/integrations/codex/hooks/dist/impl/subagent-start.mjs index 5630fa2..deb321d 100755 --- a/integrations/codex/hooks/dist/impl/subagent-start.mjs +++ b/integrations/codex/hooks/dist/impl/subagent-start.mjs @@ -370,6 +370,16 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -443,6 +453,8 @@ function resolveAll(e, userFile, creds, projectDir2, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, @@ -588,7 +600,7 @@ var init_config = __esm({ ]; CACHE_FILE = "config.json"; CACHE_TTL_MS = 300 * 1e3; - CACHE_VERSION = 3; + CACHE_VERSION = 4; MAX_ENV_TAGS = 8; MODE = "hosted"; LANG_FILES = [ diff --git a/integrations/codex/mcp/dist/index.js b/integrations/codex/mcp/dist/index.js index e47a20f..d02c894 100644 --- a/integrations/codex/mcp/dist/index.js +++ b/integrations/codex/mcp/dist/index.js @@ -182,7 +182,7 @@ var DEFAULT_MCP_TOOLS = [ ]; var CACHE_FILE = "config.json"; var CACHE_TTL_MS = 300 * 1e3; -var CACHE_VERSION = 3; +var CACHE_VERSION = 4; function screaming(key) { return String(key).replace(/([a-z0-9])([A-Z])/g, "$1_$2").toUpperCase(); } @@ -302,6 +302,16 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { ["off", "implicit", "explicit"], "implicit" ); + const sessionScore = enumOf( + pick("sessionScore", "MUBIT_CC_SESSION_SCORE"), + ["off", "compact", "full"], + host(e) === "codex" ? "off" : "full" + ); + const outcomeReview = enumOf( + pick("outcomeReview", "MUBIT_CC_OUTCOME_REVIEW"), + ["off", "nudge", "stop"], + host(e) === "codex" ? "nudge" : "stop" + ); const statusLine = bool(pick("statusLine", "MUBIT_CC_STATUSLINE"), host(e) !== "codex"); const preToolWarnings = bool(pick("preToolWarnings", "MUBIT_CC_PRE_TOOL_WARNINGS"), false); const resumeBlock = bool(pick("resumeBlock", "MUBIT_CC_RESUME_BLOCK"), true); @@ -375,6 +385,8 @@ function resolveAll(e, userFile, creds, projectDir, dataDir2) { resumeTokenBudget, policyTtlMs, outcomeMode, + sessionScore, + outcomeReview, reflectOnEnd, sessionEndDetach, statusLine, From 8df438adf84a1113f2e850e3f59df805fff95f29 Mon Sep 17 00:00:00 2001 From: Eldar <112889004+e1daru@users.noreply.github.com> Date: Tue, 29 Sep 2026 13:33:03 +0100 Subject: [PATCH 3/4] feat(scorecard): correction signal, tool intent and per-entry outcome credit (main) Published form of #28: the rebuilt bundles without their inline sourcemap line, and the docs, skills and manifests that ship. Sources and tests stay on pre-main. Co-Authored-By: Claude Opus 5.5 --- integrations/claude-code/README.md | 2 +- integrations/claude-code/bin/dashboard.mjs | 69 ++++- integrations/claude-code/docs/user-guide.md | 2 +- .../claude-code/hooks/dist/impl/capture.mjs | 69 ++++- .../claude-code/hooks/dist/impl/drain.mjs | 130 +++++++++- .../hooks/dist/impl/session-end.mjs | 69 ++++- .../hooks/dist/impl/stage-prompt.mjs | 217 ++++++++++++++-- integrations/codex/bin/dashboard.mjs | 69 ++++- .../codex/hooks/dist/impl/capture.mjs | 69 ++++- integrations/codex/hooks/dist/impl/drain.mjs | 130 +++++++++- .../codex/hooks/dist/impl/session-end.mjs | 69 ++++- .../codex/hooks/dist/impl/stage-prompt.mjs | 238 ++++++++++++++++-- 12 files changed, 1022 insertions(+), 111 deletions(-) diff --git a/integrations/claude-code/README.md b/integrations/claude-code/README.md index 5cd82a4..0a2ad25 100644 --- a/integrations/claude-code/README.md +++ b/integrations/claude-code/README.md @@ -318,7 +318,7 @@ that cache, and writing credentials invalidates it immediately rather than after | `recallAsync` | `false` | `MUBIT_CC_RECALL_ASYNC` | Never make a prompt wait on recall. On, `UserPromptSubmit` injects the block that a **detached refresh retrieved just after the previous prompt** and returns without dialling — so the hook's wall clock is a file read, however slow the endpoint is, and `MUBIT_CC_RECALL_BUDGET_MS` stops being something you have to discover and tune. It costs one turn of staleness (the block says so, in the block) and the first prompt of a session gets no recalled memory — `SessionStart`'s standing lessons still land, so the session is not memoryless. Attribution is unaffected: the ids are staged against the turn that received the block. Off by default. | | `reflectOnEnd` | `true` | `MUBIT_CC_REFLECT_ON_END` | Reflect at `SessionEnd`. This is the only path that promotes a lesson beyond its own run, so turning it off to save a few seconds trades away cross-session memory entirely. See below. | | `sessionEndDetach` | `true` | `MUBIT_CC_SESSION_END_DETACH` | Let the end-of-session drain and reflection finish in a detached process. The host cancels the `SessionEnd` hook about a second into a teardown — under `--print` it always does — and anything still running inside the hook dies with it, including the reflect above. On, the hook stamps the marker `detached`, hands the work over and returns in milliseconds; the child reports a terminal `reflect.status` when it is done, usually a few seconds after the CLI has exited. Turn it off only where background processes are forbidden — the work then runs inline, where a teardown can cut it short. | -| `outcomeMode` | `implicit` | `MUBIT_CC_OUTCOME_MODE` | `implicit`: a turn whose reply carried the recalled memory's own vocabulary is attributed to those memories; a turn that carried none of it is recorded as `neutral` against the run and attributed to no entry, so an injection nobody used is counted rather than being invisible. `explicit`: only the model's own `mubit_outcome` calls count. `off`: no attribution, and no measurement of it either. | +| `outcomeMode` | `implicit` | `MUBIT_CC_OUTCOME_MODE` | `implicit`: each injected entry is checked on its own against the reply, and only the entries whose own vocabulary the reply carried are credited; when none were, the turn is recorded as `neutral` against the run and attributed to no entry, so an injection nobody used is counted rather than being invisible. Where there is no per-entry data (server-assembled recall, older turns) the whole turn is judged as before. If your next prompt corrects Claude ("no, that's wrong", "still failing"), a failure is posted against the entries the previous reply used. Entries Claude reported on itself with `mubit_outcome` are left to that report. `explicit`: only the model's own `mubit_outcome` calls count. `off`: no attribution, and no measurement of it either. | | `sessionScore` | `full` (`off` under Codex) | `MUBIT_CC_SESSION_SCORE` | After each turn that showed a lesson, print a scorecard under the reply: lessons shown this session, how many replies used, and whether those turns worked, failed or are waiting on your reply. `full` is a short tree, `compact` one line, `off` nothing. It folds a local log (`scorecard/.jsonl`, 7 days) and makes no network call; nothing is shown while `capture` is off. See [the session scorecard](docs/user-guide.md#the-session-scorecard). | | `outcomeReview` | `stop` (`nudge` under Codex) | `MUBIT_CC_OUTCOME_REVIEW` | How hard Claude is asked to credit the memory it used. Every injected memory line starts with a short id such as `[m7k2q]`, which `mubit_outcome` accepts and the plugin maps back to the entry's reference id. `nudge` adds one sentence asking Claude to credit what helped or misled it, and keeps `mubit_outcome` and `mubit_learned` loaded rather than deferred behind tool search. `stop` also has the Stop hook ask Claude once per turn to review that turn's lessons — one extra short step, which Claude Code labels "Stop hook error occurred" although nothing failed. `off` does neither. No review runs under `outcomeMode: off`. See [crediting memory by id](docs/user-guide.md#crediting-memory-by-id-the-outcome-review). | | `statusLine` | `true` | `MUBIT_CC_STATUSLINE` | Render the status line. When false it prints an empty line and exits 0 rather than erroring per frame. | diff --git a/integrations/claude-code/bin/dashboard.mjs b/integrations/claude-code/bin/dashboard.mjs index d5fd4d9..5ba5b3c 100644 --- a/integrations/claude-code/bin/dashboard.mjs +++ b/integrations/claude-code/bin/dashboard.mjs @@ -2097,10 +2097,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str4(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str4(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str4(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -2111,21 +2141,44 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str4(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + } + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str4(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject(v) { return !!v && typeof v === "object" && !Array.isArray(v); diff --git a/integrations/claude-code/docs/user-guide.md b/integrations/claude-code/docs/user-guide.md index 9de6a76..2d8e5fc 100644 --- a/integrations/claude-code/docs/user-guide.md +++ b/integrations/claude-code/docs/user-guide.md @@ -747,7 +747,7 @@ hook, do not carry the assumption forward. `sessionScore`, default **`full`** (`off` in Codex): `full` prints the card described in [The session scorecard](#the-session-scorecard) under every reply that showed a lesson, -`compact` prints its first line only, `off` prints nothing. +`compact` prints a one-line summary instead, `off` prints nothing. `outcomeReview`, default **`stop`** (`nudge` in Codex): how hard Claude is asked to credit the memory it used. See [Crediting memory by id](#crediting-memory-by-id-the-outcome-review) for diff --git a/integrations/claude-code/hooks/dist/impl/capture.mjs b/integrations/claude-code/hooks/dist/impl/capture.mjs index 3e70af3..a2fb30e 100755 --- a/integrations/claude-code/hooks/dist/impl/capture.mjs +++ b/integrations/claude-code/hooks/dist/impl/capture.mjs @@ -1768,10 +1768,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str3(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject4(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str3(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str3(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -1782,21 +1812,44 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str3(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str3(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; + } + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject4(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject4(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject4(v) { return !!v && typeof v === "object" && !Array.isArray(v); diff --git a/integrations/claude-code/hooks/dist/impl/drain.mjs b/integrations/claude-code/hooks/dist/impl/drain.mjs index 752db05..9b0340b 100755 --- a/integrations/claude-code/hooks/dist/impl/drain.mjs +++ b/integrations/claude-code/hooks/dist/impl/drain.mjs @@ -1016,10 +1016,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str2(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject2(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str2(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str2(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -1030,8 +1060,35 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) + }; +} +function decideCorrection(turn) { + if (!isObject2(turn)) return { post: false, reason: "not_a_turn" }; + if (numOr(turn.correction_sent_at, 0) > 0) return { post: false, reason: "already_sent" }; + if (str2(turn[API_ERROR_KEY])) return { post: false, reason: "api_failed" }; + const ev = isObject2(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (!entries) return { post: false, reason: "nothing_used" }; + const explicit = new Set(explicitIdsOf(turn)); + const ids = Object.keys(entries).filter((r) => entries[r].used === true && !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "nothing_used" }; + return { + post: true, + outcome: OUTCOME_FAILURE, + signal: SIGNAL_FAILURE, + entryIds: ids, + rationale: `The user's next prompt corrected this Claude Code turn; the reply had used ${ids.length} ${ids.length === 1 ? "memory" : "memories"} (memory-term-echo/v2-entry).` + }; +} +function correctionIdempotencyKey(runId, promptId) { + return `cc-correction-${str2(runId)}-${str2(promptId)}`; +} +function correctionRequest(o) { + return { + ...outcomeRequest(o), + idempotency_key: correctionIdempotencyKey(o.runId, o.promptId) }; } function outcomeIdempotencyKey(runId, promptId) { @@ -1050,17 +1107,40 @@ function outcomeRequest(o) { idempotency_key: outcomeIdempotencyKey(o.runId, o.promptId) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str2(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + } + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str2(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject2(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject2(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject2(v) { return !!v && typeof v === "object" && !Array.isArray(v); @@ -2663,6 +2743,7 @@ async function main() { const outcomeArg = flagValue(argv, "--with-outcome"); const wantsOutcome = argv.includes("--with-outcome"); const pinnedRun = flagValue(argv, "--run"); + const correctArg = str3(flagValue(argv, "--correct")); const payload = await readPayload(payloadPath); const cfg = loadConfig(process.env); cfgRef = cfg; @@ -2683,7 +2764,7 @@ async function main() { } const agentId = deriveAgentId(payload); const promptId = str3(outcomeArg) || turnKey(payload); - const lock = await acquireConfirmed(cfg, runId, wantsOutcome, started); + const lock = await acquireConfirmed(cfg, runId, wantsOutcome || !!correctArg, started); if (!lock) { log(cfg, "debug", "drain: another drainer holds the lock; standing down", { run_id: runId }); return; @@ -2700,6 +2781,7 @@ async function main() { try { const drained = await drainSpool(cfg, runId, agentId, promptId, started); await flushOutcome(cfg, runId, agentId, promptId, wantsOutcome); + if (correctArg && !breakerOpen(cfg)) await sendCorrection(cfg, runId, agentId, correctArg); log(cfg, "info", `drain: ${drained.sent} item(s) in ${drained.batches} batch(es)`, { run_id: runId, rejected: drained.rejected, @@ -2926,6 +3008,36 @@ async function sendOutcome(cfg, runId, agentId, promptId) { log(cfg, "warn", `drain: outcome skipped \u2014 ${messageOf3(err)}`, { run_id: runId }); } } +async function sendCorrection(cfg, runId, agentId, promptId) { + try { + if (!implicitOutcomesEnabled(cfg)) return; + const p = join12(runDir(cfg, runId), "turns", `${safeSegment(promptId)}.json`); + const turn = readJson(p, null); + const decision = decideCorrection(turn); + if (!decision.post) { + log(cfg, "debug", `drain: no correction to post (${decision.reason})`, { run_id: runId, prompt_id: promptId }); + return; + } + const res = await postOutcome( + cfg, + correctionRequest({ runId, agentId, promptId, decision }), + { timeoutMs: numOr2(cfg.timeoutMs, 4e3) } + ); + if (!res.ok) { + log(cfg, "warn", `drain: correction post failed (${res.state})`, { run_id: runId, prompt_id: promptId }); + return; + } + const fresh2 = readJson(p, turn); + writeJsonAtomic(p, { ...fresh2 && typeof fresh2 === "object" ? fresh2 : turn, correction_sent_at: Date.now() }); + appendLedger( + resolveDataDir(cfg), + runId, + { ...outcomeLedgerRow(runId, promptId, decision, 1), correction: true } + ); + } catch (err) { + log(cfg, "warn", `drain: correction skipped \u2014 ${messageOf3(err)}`, { run_id: runId }); + } +} function outcomeLedgerRow(runId, promptId, decision, attempts) { return { v: 1, diff --git a/integrations/claude-code/hooks/dist/impl/session-end.mjs b/integrations/claude-code/hooks/dist/impl/session-end.mjs index 900e415..8c4f4ac 100755 --- a/integrations/claude-code/hooks/dist/impl/session-end.mjs +++ b/integrations/claude-code/hooks/dist/impl/session-end.mjs @@ -914,10 +914,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str2(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str2(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str2(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -928,8 +958,8 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) }; } function outcomeIdempotencyKey(runId, promptId) { @@ -948,17 +978,40 @@ function outcomeRequest(o) { idempotency_key: outcomeIdempotencyKey(o.runId, o.promptId) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str2(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + } + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str2(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject(v) { return !!v && typeof v === "object" && !Array.isArray(v); diff --git a/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs b/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs index a414b93..7c856de 100644 --- a/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs +++ b/integrations/claude-code/hooks/dist/impl/stage-prompt.mjs @@ -2,8 +2,8 @@ // hooks/src/stage-prompt.mjs import { randomUUID } from "node:crypto"; -import { readdirSync as readdirSync4 } from "node:fs"; -import { join as join8 } from "node:path"; +import { readdirSync as readdirSync5 } from "node:fs"; +import { join as join9 } from "node:path"; // lib/config.mjs import { createHash } from "node:crypto"; @@ -516,6 +516,39 @@ function safeCwd() { } } +// lib/correction.mjs +var SCAN_CHARS = 300; +var LEADING_NO = /^(?:no|nope)(?:\s*[,.!;:—–-]|\s*$)/; +var POLITE_NO = /^(?:no|nope)[\s,.!]*(?:problem|worries|thanks|thank you|need|rush|biggie)\b/; +var BARE_NO = /^(?:no|nope)[\s.!]*$/; +var PHRASES = [ + /^(?:wrong|incorrect)\b/, + /\b(?:that'?s|that is|this is|it'?s|it is)\s+(?:wrong|incorrect|not right|not it|not what i (?:asked|wanted|meant))\b/, + /\bnot what i (?:asked|wanted|meant)\b/, + /\b(?:you|it|this|that) broke\b/, + /\bstill (?:failing|fails|broken|erroring|crashing|not working|(?:doesn'?t|does not|isn'?t|is not) work(?:ing)?)\b/, + /\b(?:doesn'?t|does not|didn'?t|did not) work\b/, + /\b(?:revert|undo) (?:that|this|it|the (?:last|previous))\b/, + /\broll back (?:that|this|it|the)\b/, + /\byou misunderstood\b/, + /\byou(?:'ve| have)? got it wrong\b/ +]; +function isCorrection(prompt, opts = {}) { + try { + if (typeof prompt !== "string") return false; + const trimmed = prompt.trim(); + if (!trimmed || trimmed.startsWith("/")) return false; + const text = trimmed.replace(/```[\s\S]*?(?:```|$)/g, " ").replace(/"[^"\n]*"|“[^”\n]*”|`[^`\n]*`/g, " ").replace(/[‘’]/g, "'").trim().slice(0, SCAN_CHARS).toLowerCase(); + if (!text) return false; + const asked = opts?.lastReplyEndedWithQuestion === true; + if (BARE_NO.test(text)) return !asked; + if (!asked && LEADING_NO.test(text) && !POLITE_NO.test(text)) return true; + return PHRASES.some((re) => re.test(text)); + } catch { + return false; + } +} + // lib/hook.mjs import { spawn } from "node:child_process"; import { @@ -640,9 +673,9 @@ function scrub(text, count) { } function entropy(s) { if (s === null || s === void 0) return 0; - const str2 = typeof s === "string" ? s : String(s); - if (str2.length === 0) return 0; - const buf = Buffer.from(str2, "utf8"); + const str3 = typeof s === "string" ? s : String(s); + if (str3.length === 0) return 0; + const buf = Buffer.from(str3, "utf8"); const n = buf.length; if (n === 0) return 0; const counts = new Uint32Array(256); @@ -1115,6 +1148,19 @@ function safeConfig() { } } +// lib/outcome.mjs +var SILENCED_MODES = /* @__PURE__ */ new Set(["off", "explicit"]); +function implicitOutcomesEnabled(cfg) { + const mode = str2(cfg && typeof cfg === "object" ? ( + /** @type {any} */ + cfg.outcomeMode + ) : "").toLowerCase(); + return !SILENCED_MODES.has(mode); +} +function str2(v) { + return typeof v === "string" ? v.trim() : ""; +} + // lib/runid.mjs import { spawnSync } from "node:child_process"; import { createHash as createHash2 } from "node:crypto"; @@ -1417,23 +1463,115 @@ function safeCwd2() { } } -// lib/spool.mjs +// lib/scorecard-log.mjs import { closeSync as closeSync2, - existsSync as existsSync6, - linkSync, + fstatSync, openSync as openSync2, readdirSync as readdirSync3, readFileSync as readFileSync6, - renameSync as renameSync3, + readSync, statSync as statSync5, + writeSync as writeSync3 +} from "node:fs"; +import { dirname as dirname5, join as join7 } from "node:path"; +var SCORE_LOG_VERSION = 1; +var SCORE_DIR = "scorecard"; +var SCORE_LOG_TTL_MS = 7 * 24 * 60 * 60 * 1e3; +var MAX_READ_BYTES = 4 * 1024 * 1024; +var MAX_ID = 128; +function scorecardPath(cfg, sessionId) { + const id = safeSegment(typeof sessionId === "string" ? sessionId.trim() : "", MAX_ID); + if (!id) return ""; + return join7(resolveDataDir(cfg), SCORE_DIR, `${id}.jsonl`); +} +function appendScoreRow(cfg, sessionId, row) { + try { + if (!row || typeof row !== "object" || Array.isArray(row)) return false; + if (typeof row.kind !== "string" || !row.kind) return false; + const p = scorecardPath(cfg, sessionId); + if (!p || !ensureDir(dirname5(p))) return false; + const line = `${JSON.stringify({ v: SCORE_LOG_VERSION, at: Date.now(), ...row })} +`; + const fd = openSync2(p, "a+"); + try { + const st = fstatSync(fd); + let prefix = ""; + if (st.size > 0) { + const last = Buffer.alloc(1); + readSync(fd, last, 0, 1, st.size - 1); + if (last[0] !== 10) prefix = "\n"; + } + writeSync3(fd, prefix + line); + } finally { + closeSync2(fd); + } + return true; + } catch { + return false; + } +} +function readScoreRows(cfg, sessionId, opts = {}) { + const p = scorecardPath(cfg, sessionId); + if (!p) return []; + return readRowsAt(p, opts); +} +function readRowsAt(p, opts = {}) { + try { + const size = statSync5(p).size; + const tail = Number(opts?.tailBytes); + const want = Number.isFinite(tail) && tail > 0 ? Math.min(tail, MAX_READ_BYTES) : MAX_READ_BYTES; + let text; + let partialHead = false; + if (size > want) { + const fd = openSync2(p, "r"); + try { + const buf = Buffer.alloc(want); + readSync(fd, buf, 0, want, size - want); + text = buf.toString("utf8"); + } finally { + closeSync2(fd); + } + partialHead = true; + } else { + text = readFileSync6(p, "utf8"); + } + const lines = text.split("\n"); + if (partialHead) lines.shift(); + const out = []; + for (const line of lines) { + if (!line.trim()) continue; + try { + const row = JSON.parse(line); + if (row && typeof row === "object" && !Array.isArray(row) && typeof row.kind === "string") { + out.push(row); + } + } catch { + } + } + return out; + } catch { + return []; + } +} + +// lib/spool.mjs +import { + closeSync as closeSync3, + existsSync as existsSync6, + linkSync, + openSync as openSync3, + readdirSync as readdirSync4, + readFileSync as readFileSync7, + renameSync as renameSync3, + statSync as statSync6, unlinkSync as unlinkSync4, writeFileSync as writeFileSync3, - writeSync as writeSync3 + writeSync as writeSync4 } from "node:fs"; -import { join as join7 } from "node:path"; +import { join as join8 } from "node:path"; function spoolDir(cfg, runId) { - return join7(runDir(cfg, runId), "spool"); + return join8(runDir(cfg, runId), "spool"); } function stampOf(dir, name) { const m = /^(\d{10,})-/.exec(name); @@ -1442,7 +1580,7 @@ function stampOf(dir, name) { if (Number.isFinite(n)) return n; } try { - return statSync5(join7(dir, name)).mtimeMs; + return statSync6(join8(dir, name)).mtimeMs; } catch { return 0; } @@ -1452,7 +1590,7 @@ function spoolStats(cfg, runId) { const dir = spoolDir(cfg, runId); let entries; try { - entries = readdirSync3(dir, { withFileTypes: true }); + entries = readdirSync4(dir, { withFileTypes: true }); } catch { return { count: 0, oldestMs: 0 }; } @@ -1474,7 +1612,8 @@ function spoolStats(cfg, runId) { // hooks/src/stage-prompt.mjs var BUDGET_MS = 250; var MAX_PROMPT_BYTES = 64 * 1024; -var MAX_ID = 128; +var MAX_ID2 = 128; +var LOG_TAIL_BYTES = 64 * 1024; await runHook("stage-prompt", { budgetMs: BUDGET_MS, body: async (payload) => { @@ -1488,17 +1627,18 @@ await runHook("stage-prompt", { return { suppressOutput: true }; } stageTurn(cfg, runId, payload); + if (cfg.capture !== false) scorePrompt(cfg, payload); if (cfg.capture) maybeDrain(cfg, runId, payload); return { suppressOutput: true }; } }); function stageTurn(cfg, runId, payload) { try { - const promptId = safeSegment(turnKey(payload), MAX_ID); + const promptId = safeSegment(turnKey(payload), MAX_ID2); if (!promptId) return false; - const dir = join8(runDir(cfg, runId), "turns"); + const dir = join9(runDir(cfg, runId), "turns"); if (!ensureDir(dir)) return false; - const file = join8(dir, `${promptId}.json`); + const file = join9(dir, `${promptId}.json`); const prev = readJson(file, null); const base = isObject3(prev) ? prev : {}; const { text, truncated } = clampPrompt(payload?.prompt); @@ -1521,12 +1661,47 @@ function stageTurn(cfg, runId, payload) { } function ordinalFor(dir, mine) { try { - const n = readdirSync4(dir).filter((f) => f.endsWith(".json")).length; + const n = readdirSync5(dir).filter((f) => f.endsWith(".json")).length; return Math.max(1, mine ? n : n + 1); } catch { return 1; } } +function scorePrompt(cfg, payload) { + try { + const sessionId = typeof payload?.session_id === "string" ? payload.session_id : ""; + const promptId = safeSegment(turnKey(payload), MAX_ID2); + if (!sessionId || !promptId) return; + const prompt = typeof payload?.prompt === "string" ? payload.prompt : ""; + const slash = prompt.trim().startsWith("/"); + const rows = readScoreRows(cfg, sessionId, { tailBytes: LOG_TAIL_BYTES }); + let prev = null; + let afterClear = false; + for (let i = rows.length - 1; i >= 0; i--) { + const r = rows[i]; + if (r.kind === "start" && r.source === "clear") afterClear = true; + if (r.kind === "turn" && typeof r.prompt_id === "string" && r.prompt_id && r.prompt_id !== promptId) { + prev = r; + break; + } + } + const correction = !slash && !!prev && !afterClear && isCorrection(prompt, { lastReplyEndedWithQuestion: prev?.ended_with_question === true }); + appendScoreRow(cfg, sessionId, { kind: "prompt", prompt_id: promptId, correction, slash }); + if (!correction || !prev || !implicitOutcomesEnabled(cfg)) return; + const used = Array.isArray(prev.used_refs) ? prev.used_refs.filter((v) => typeof v === "string" && v.trim()) : []; + const prevRun = typeof prev.run_id === "string" ? prev.run_id : ""; + if (!used.length || !prevRun) return; + spawnDetached( + cfg, + "drain", + ["--correct", String(prev.prompt_id), "--run", prevRun], + writePayload(cfg, payload) + ); + log(cfg, "debug", "stage-prompt: correction of the previous turn", { prompt_id: String(prev.prompt_id) }); + } catch (err) { + log(cfg, "warn", `stage-prompt: could not score the prompt (${messageOf(err)})`); + } +} function maybeDrain(cfg, runId, payload) { try { const { count, oldestMs } = spoolStats(cfg, runId); @@ -1542,8 +1717,8 @@ function maybeDrain(cfg, runId, payload) { } } function writePayload(cfg, payload) { - const dir = join8(resolveDataDir(cfg), "tmp"); - const file = join8(dir, `${randomUUID()}.json`); + const dir = join9(resolveDataDir(cfg), "tmp"); + const file = join9(dir, `${randomUUID()}.json`); ensureDir(dir); writeJsonAtomic(file, isObject3(payload) ? payload : {}); return file; diff --git a/integrations/codex/bin/dashboard.mjs b/integrations/codex/bin/dashboard.mjs index 58456d3..b644f5a 100644 --- a/integrations/codex/bin/dashboard.mjs +++ b/integrations/codex/bin/dashboard.mjs @@ -2119,10 +2119,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str4(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str4(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str4(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -2133,21 +2163,44 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str4(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + } + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str4(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject(v) { return !!v && typeof v === "object" && !Array.isArray(v); diff --git a/integrations/codex/hooks/dist/impl/capture.mjs b/integrations/codex/hooks/dist/impl/capture.mjs index f060dde..0222a64 100755 --- a/integrations/codex/hooks/dist/impl/capture.mjs +++ b/integrations/codex/hooks/dist/impl/capture.mjs @@ -1796,10 +1796,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str3(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject4(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str3(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str3(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -1810,21 +1840,44 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str3(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str3(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; + } + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject4(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject4(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject4(v) { return !!v && typeof v === "object" && !Array.isArray(v); diff --git a/integrations/codex/hooks/dist/impl/drain.mjs b/integrations/codex/hooks/dist/impl/drain.mjs index e3879a9..00c2c17 100755 --- a/integrations/codex/hooks/dist/impl/drain.mjs +++ b/integrations/codex/hooks/dist/impl/drain.mjs @@ -1021,10 +1021,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str2(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject2(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str2(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str2(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -1035,8 +1065,35 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) + }; +} +function decideCorrection(turn) { + if (!isObject2(turn)) return { post: false, reason: "not_a_turn" }; + if (numOr(turn.correction_sent_at, 0) > 0) return { post: false, reason: "already_sent" }; + if (str2(turn[API_ERROR_KEY])) return { post: false, reason: "api_failed" }; + const ev = isObject2(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (!entries) return { post: false, reason: "nothing_used" }; + const explicit = new Set(explicitIdsOf(turn)); + const ids = Object.keys(entries).filter((r) => entries[r].used === true && !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "nothing_used" }; + return { + post: true, + outcome: OUTCOME_FAILURE, + signal: SIGNAL_FAILURE, + entryIds: ids, + rationale: `The user's next prompt corrected this Claude Code turn; the reply had used ${ids.length} ${ids.length === 1 ? "memory" : "memories"} (memory-term-echo/v2-entry).` + }; +} +function correctionIdempotencyKey(runId, promptId) { + return `cc-correction-${str2(runId)}-${str2(promptId)}`; +} +function correctionRequest(o) { + return { + ...outcomeRequest(o), + idempotency_key: correctionIdempotencyKey(o.runId, o.promptId) }; } function outcomeIdempotencyKey(runId, promptId) { @@ -1055,17 +1112,40 @@ function outcomeRequest(o) { idempotency_key: outcomeIdempotencyKey(o.runId, o.promptId) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str2(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + } + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str2(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject2(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject2(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject2(v) { return !!v && typeof v === "object" && !Array.isArray(v); @@ -2735,6 +2815,7 @@ async function main() { const outcomeArg = flagValue(argv, "--with-outcome"); const wantsOutcome = argv.includes("--with-outcome"); const pinnedRun = flagValue(argv, "--run"); + const correctArg = str3(flagValue(argv, "--correct")); const payload = await readPayload(payloadPath); const cfg = loadConfig(process.env); cfgRef = cfg; @@ -2755,7 +2836,7 @@ async function main() { } const agentId = deriveAgentId(payload); const promptId = str3(outcomeArg) || turnKey(payload); - const lock = await acquireConfirmed(cfg, runId, wantsOutcome, started); + const lock = await acquireConfirmed(cfg, runId, wantsOutcome || !!correctArg, started); if (!lock) { log(cfg, "debug", "drain: another drainer holds the lock; standing down", { run_id: runId }); return; @@ -2772,6 +2853,7 @@ async function main() { try { const drained = await drainSpool(cfg, runId, agentId, promptId, started); await flushOutcome(cfg, runId, agentId, promptId, wantsOutcome); + if (correctArg && !breakerOpen(cfg)) await sendCorrection(cfg, runId, agentId, correctArg); log(cfg, "info", `drain: ${drained.sent} item(s) in ${drained.batches} batch(es)`, { run_id: runId, rejected: drained.rejected, @@ -2998,6 +3080,36 @@ async function sendOutcome(cfg, runId, agentId, promptId) { log(cfg, "warn", `drain: outcome skipped \u2014 ${messageOf3(err)}`, { run_id: runId }); } } +async function sendCorrection(cfg, runId, agentId, promptId) { + try { + if (!implicitOutcomesEnabled(cfg)) return; + const p = join13(runDir(cfg, runId), "turns", `${safeSegment(promptId)}.json`); + const turn = readJson(p, null); + const decision = decideCorrection(turn); + if (!decision.post) { + log(cfg, "debug", `drain: no correction to post (${decision.reason})`, { run_id: runId, prompt_id: promptId }); + return; + } + const res = await postOutcome( + cfg, + correctionRequest({ runId, agentId, promptId, decision }), + { timeoutMs: numOr2(cfg.timeoutMs, 4e3) } + ); + if (!res.ok) { + log(cfg, "warn", `drain: correction post failed (${res.state})`, { run_id: runId, prompt_id: promptId }); + return; + } + const fresh2 = readJson(p, turn); + writeJsonAtomic(p, { ...fresh2 && typeof fresh2 === "object" ? fresh2 : turn, correction_sent_at: Date.now() }); + appendLedger( + resolveDataDir(cfg), + runId, + { ...outcomeLedgerRow(runId, promptId, decision, 1), correction: true } + ); + } catch (err) { + log(cfg, "warn", `drain: correction skipped \u2014 ${messageOf3(err)}`, { run_id: runId }); + } +} function outcomeLedgerRow(runId, promptId, decision, attempts) { return { v: 1, diff --git a/integrations/codex/hooks/dist/impl/session-end.mjs b/integrations/codex/hooks/dist/impl/session-end.mjs index 6188ab6..cb5c985 100755 --- a/integrations/codex/hooks/dist/impl/session-end.mjs +++ b/integrations/codex/hooks/dist/impl/session-end.mjs @@ -913,10 +913,40 @@ function decideOutcome(turn) { if (numOr(turn.outcome_attempts, 0) >= MAX_OUTCOME_ATTEMPTS) { return { post: false, reason: "attempts_exhausted" }; } - const entryIds = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; - if (entryIds.length === 0) return { post: false, reason: "nothing_injected" }; - const failed = str2(turn.outcome).toLowerCase() === "failure"; + const recalled = Array.isArray(turn.recalled) ? turn.recalled.filter((v) => typeof v === "string" && v.trim()) : []; const ev = isObject(turn.used_evidence) ? turn.used_evidence : {}; + const entries = entriesOf(ev); + if (recalled.length === 0 && !entries) return { post: false, reason: "nothing_injected" }; + const failed = str2(turn.outcome).toLowerCase() === "failure"; + const toolFailure = failed && str2(turn.failure_reason) === "tool_failure"; + if (entries) { + const refs = Object.keys(entries); + const used = refs.filter((r) => entries[r].used === true); + const measured = refs.some((r) => entries[r].used === false); + if (used.length > 0) { + const explicit = new Set(explicitIdsOf(turn)); + const ids = used.filter((r) => !explicit.has(r)); + if (ids.length === 0) return { post: false, reason: "explicit_only" }; + return { + post: true, + outcome: failed ? OUTCOME_FAILURE : OUTCOME_SUCCESS, + signal: failed ? SIGNAL_FAILURE : SIGNAL_SUCCESS, + entryIds: ids, + rationale: entryRationale(ev, used.length, refs.length, failed, toolFailure) + }; + } + if (measured && explicitIdsOf(turn).length > 0) return { post: false, reason: "explicit_only" }; + if (measured) { + return { + post: true, + outcome: OUTCOME_UNUSED, + signal: SIGNAL_UNUSED, + entryIds: [], + rationale: entryRationale(ev, 0, refs.length, failed, toolFailure) + }; + } + if (recalled.length === 0) return { post: false, reason: "nothing_injected" }; + } const unused = ev.used === false; return { post: true, @@ -927,8 +957,8 @@ function decideOutcome(turn) { // The cost is that the record says a turn was injected-and-unused // without saying which entries were ignored — a real limitation, and the honest side of // the trade. - entryIds: unused ? [] : entryIds, - rationale: rationaleFor(ev, unused, failed, entryIds.length) + entryIds: unused ? [] : recalled, + rationale: rationaleFor(ev, unused, failed, recalled.length, toolFailure) }; } function outcomeIdempotencyKey(runId, promptId) { @@ -947,17 +977,40 @@ function outcomeRequest(o) { idempotency_key: outcomeIdempotencyKey(o.runId, o.promptId) }; } -function rationaleFor(ev, unused, failed, n) { +function rationaleFor(ev, unused, failed, n, toolFailure = false) { const method = str2(ev.method); const by = method ? ` (${method})` : ""; const counts = `${numOr(ev.matched, 0)} of ${numOr(ev.candidates, 0)} injected memory terms`; if (unused) { return `Claude Code injected ${n} ${n === 1 ? "memory" : "memories"} and the reply carried none of their vocabulary \u2014 ${counts}${by}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } + const ended = toolFailure ? "Claude Code turn ended on a failed tool call" : "Claude Code turn ended in failure"; if (ev.used === true) { - return failed ? `Claude Code turn ended in failure; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + return failed ? `${ended}; the reply carried ${counts}${by}.` : `Claude Code turn completed and the reply carried ${counts}${by}.`; + } + return failed ? `${ended} after these memories were injected.` : "Claude Code turn completed after these memories were injected."; +} +function entryRationale(ev, used, of, failed, toolFailure) { + const method = str2(ev.entry_method) || "memory-term-echo/v2-entry"; + const counts = `the reply used ${used} of ${of} injected ${of === 1 ? "memory" : "memories"} (${method})`; + if (used === 0) { + return `Claude Code ${counts}. Recorded, not penalised: this method cannot see memory the model followed without quoting it.`; } - return failed ? "Claude Code turn ended in failure after these memories were injected." : "Claude Code turn completed after these memories were injected."; + if (!failed) return `Claude Code turn completed; ${counts}.`; + return toolFailure ? `Claude Code turn ended on a failed tool call; ${counts}.` : `Claude Code turn ended in failure; ${counts}.`; +} +function entriesOf(ev) { + const e = ev.entries; + if (!isObject(e)) return null; + const out = {}; + for (const [ref, v] of Object.entries(e)) { + if (ref.trim() && isObject(v)) out[ref] = /** @type {any} */ + v; + } + return Object.keys(out).length ? out : null; +} +function explicitIdsOf(turn) { + return Array.isArray(turn.explicit_ids) ? turn.explicit_ids.filter((v) => typeof v === "string" && v.trim()) : []; } function isObject(v) { return !!v && typeof v === "object" && !Array.isArray(v); diff --git a/integrations/codex/hooks/dist/impl/stage-prompt.mjs b/integrations/codex/hooks/dist/impl/stage-prompt.mjs index a05b575..3b5c340 100755 --- a/integrations/codex/hooks/dist/impl/stage-prompt.mjs +++ b/integrations/codex/hooks/dist/impl/stage-prompt.mjs @@ -535,6 +535,44 @@ var init_config = __esm({ } }); +// ../claude-code/lib/correction.mjs +function isCorrection(prompt, opts = {}) { + try { + if (typeof prompt !== "string") return false; + const trimmed = prompt.trim(); + if (!trimmed || trimmed.startsWith("/")) return false; + const text = trimmed.replace(/```[\s\S]*?(?:```|$)/g, " ").replace(/"[^"\n]*"|“[^”\n]*”|`[^`\n]*`/g, " ").replace(/[‘’]/g, "'").trim().slice(0, SCAN_CHARS).toLowerCase(); + if (!text) return false; + const asked = opts?.lastReplyEndedWithQuestion === true; + if (BARE_NO.test(text)) return !asked; + if (!asked && LEADING_NO.test(text) && !POLITE_NO.test(text)) return true; + return PHRASES.some((re) => re.test(text)); + } catch { + return false; + } +} +var SCAN_CHARS, LEADING_NO, POLITE_NO, BARE_NO, PHRASES; +var init_correction = __esm({ + "../claude-code/lib/correction.mjs"() { + SCAN_CHARS = 300; + LEADING_NO = /^(?:no|nope)(?:\s*[,.!;:—–-]|\s*$)/; + POLITE_NO = /^(?:no|nope)[\s,.!]*(?:problem|worries|thanks|thank you|need|rush|biggie)\b/; + BARE_NO = /^(?:no|nope)[\s.!]*$/; + PHRASES = [ + /^(?:wrong|incorrect)\b/, + /\b(?:that'?s|that is|this is|it'?s|it is)\s+(?:wrong|incorrect|not right|not it|not what i (?:asked|wanted|meant))\b/, + /\bnot what i (?:asked|wanted|meant)\b/, + /\b(?:you|it|this|that) broke\b/, + /\bstill (?:failing|fails|broken|erroring|crashing|not working|(?:doesn'?t|does not|isn'?t|is not) work(?:ing)?)\b/, + /\b(?:doesn'?t|does not|didn'?t|did not) work\b/, + /\b(?:revert|undo) (?:that|this|it|the (?:last|previous))\b/, + /\broll back (?:that|this|it|the)\b/, + /\byou misunderstood\b/, + /\byou(?:'ve| have)? got it wrong\b/ + ]; + } +}); + // ../claude-code/lib/redact.mjs function scrubAssignments(text, count) { ASSIGNMENT_RE.lastIndex = 0; @@ -599,9 +637,9 @@ function scrub(text, count) { } function entropy(s) { if (s === null || s === void 0) return 0; - const str2 = typeof s === "string" ? s : String(s); - if (str2.length === 0) return 0; - const buf = Buffer.from(str2, "utf8"); + const str3 = typeof s === "string" ? s : String(s); + if (str3.length === 0) return 0; + const buf = Buffer.from(str3, "utf8"); const n = buf.length; if (n === 0) return 0; const counts = new Uint32Array(256); @@ -1150,6 +1188,24 @@ var init_hook = __esm({ } }); +// ../claude-code/lib/outcome.mjs +function implicitOutcomesEnabled(cfg) { + const mode = str2(cfg && typeof cfg === "object" ? ( + /** @type {any} */ + cfg.outcomeMode + ) : "").toLowerCase(); + return !SILENCED_MODES.has(mode); +} +function str2(v) { + return typeof v === "string" ? v.trim() : ""; +} +var SILENCED_MODES; +var init_outcome = __esm({ + "../claude-code/lib/outcome.mjs"() { + SILENCED_MODES = /* @__PURE__ */ new Set(["off", "explicit"]); + } +}); + // ../claude-code/lib/runid.mjs import { spawnSync } from "node:child_process"; import { createHash as createHash2 } from "node:crypto"; @@ -1458,23 +1514,121 @@ var init_runid = __esm({ } }); -// ../claude-code/lib/spool.mjs +// ../claude-code/lib/scorecard-log.mjs import { closeSync as closeSync2, - existsSync as existsSync7, - linkSync, + fstatSync, openSync as openSync2, readdirSync as readdirSync4, readFileSync as readFileSync7, - renameSync as renameSync3, + readSync, statSync as statSync6, + writeSync as writeSync3 +} from "node:fs"; +import { dirname as dirname6, join as join8 } from "node:path"; +function scorecardPath(cfg, sessionId) { + const id = safeSegment(typeof sessionId === "string" ? sessionId.trim() : "", MAX_ID); + if (!id) return ""; + return join8(resolveDataDir(cfg), SCORE_DIR, `${id}.jsonl`); +} +function appendScoreRow(cfg, sessionId, row) { + try { + if (!row || typeof row !== "object" || Array.isArray(row)) return false; + if (typeof row.kind !== "string" || !row.kind) return false; + const p = scorecardPath(cfg, sessionId); + if (!p || !ensureDir(dirname6(p))) return false; + const line = `${JSON.stringify({ v: SCORE_LOG_VERSION, at: Date.now(), ...row })} +`; + const fd = openSync2(p, "a+"); + try { + const st = fstatSync(fd); + let prefix = ""; + if (st.size > 0) { + const last = Buffer.alloc(1); + readSync(fd, last, 0, 1, st.size - 1); + if (last[0] !== 10) prefix = "\n"; + } + writeSync3(fd, prefix + line); + } finally { + closeSync2(fd); + } + return true; + } catch { + return false; + } +} +function readScoreRows(cfg, sessionId, opts = {}) { + const p = scorecardPath(cfg, sessionId); + if (!p) return []; + return readRowsAt(p, opts); +} +function readRowsAt(p, opts = {}) { + try { + const size = statSync6(p).size; + const tail = Number(opts?.tailBytes); + const want = Number.isFinite(tail) && tail > 0 ? Math.min(tail, MAX_READ_BYTES) : MAX_READ_BYTES; + let text; + let partialHead = false; + if (size > want) { + const fd = openSync2(p, "r"); + try { + const buf = Buffer.alloc(want); + readSync(fd, buf, 0, want, size - want); + text = buf.toString("utf8"); + } finally { + closeSync2(fd); + } + partialHead = true; + } else { + text = readFileSync7(p, "utf8"); + } + const lines = text.split("\n"); + if (partialHead) lines.shift(); + const out = []; + for (const line of lines) { + if (!line.trim()) continue; + try { + const row = JSON.parse(line); + if (row && typeof row === "object" && !Array.isArray(row) && typeof row.kind === "string") { + out.push(row); + } + } catch { + } + } + return out; + } catch { + return []; + } +} +var SCORE_LOG_VERSION, SCORE_DIR, SCORE_LOG_TTL_MS, MAX_READ_BYTES, MAX_ID; +var init_scorecard_log = __esm({ + "../claude-code/lib/scorecard-log.mjs"() { + init_state(); + SCORE_LOG_VERSION = 1; + SCORE_DIR = "scorecard"; + SCORE_LOG_TTL_MS = 7 * 24 * 60 * 60 * 1e3; + MAX_READ_BYTES = 4 * 1024 * 1024; + MAX_ID = 128; + } +}); + +// ../claude-code/lib/spool.mjs +import { + closeSync as closeSync3, + existsSync as existsSync7, + linkSync, + openSync as openSync3, + readdirSync as readdirSync5, + readFileSync as readFileSync8, + renameSync as renameSync3, + statSync as statSync7, unlinkSync as unlinkSync4, writeFileSync as writeFileSync3, - writeSync as writeSync3 + writeSync as writeSync4 } from "node:fs"; -import { join as join8 } from "node:path"; +import { join as join9 } from "node:path"; function spoolDir(cfg, runId) { - return join8(runDir(cfg, runId), "spool"); + return join9(runDir(cfg, runId), "spool"); } function stampOf(dir, name) { const m = /^(\d{10,})-/.exec(name); @@ -1483,7 +1637,7 @@ function stampOf(dir, name) { if (Number.isFinite(n)) return n; } try { - return statSync6(join8(dir, name)).mtimeMs; + return statSync7(join9(dir, name)).mtimeMs; } catch { return 0; } @@ -1493,7 +1647,7 @@ function spoolStats(cfg, runId) { const dir = spoolDir(cfg, runId); let entries; try { - entries = readdirSync4(dir, { withFileTypes: true }); + entries = readdirSync5(dir, { withFileTypes: true }); } catch { return { count: 0, oldestMs: 0 }; } @@ -1520,15 +1674,15 @@ var init_spool = __esm({ // ../claude-code/hooks/src/stage-prompt.mjs var stage_prompt_exports = {}; import { randomUUID } from "node:crypto"; -import { readdirSync as readdirSync5 } from "node:fs"; -import { join as join9 } from "node:path"; +import { readdirSync as readdirSync6 } from "node:fs"; +import { join as join10 } from "node:path"; function stageTurn(cfg, runId, payload) { try { - const promptId = safeSegment(turnKey(payload), MAX_ID); + const promptId = safeSegment(turnKey(payload), MAX_ID2); if (!promptId) return false; - const dir = join9(runDir(cfg, runId), "turns"); + const dir = join10(runDir(cfg, runId), "turns"); if (!ensureDir(dir)) return false; - const file = join9(dir, `${promptId}.json`); + const file = join10(dir, `${promptId}.json`); const prev = readJson(file, null); const base = isObject3(prev) ? prev : {}; const { text, truncated } = clampPrompt(payload?.prompt); @@ -1551,12 +1705,47 @@ function stageTurn(cfg, runId, payload) { } function ordinalFor(dir, mine) { try { - const n = readdirSync5(dir).filter((f) => f.endsWith(".json")).length; + const n = readdirSync6(dir).filter((f) => f.endsWith(".json")).length; return Math.max(1, mine ? n : n + 1); } catch { return 1; } } +function scorePrompt(cfg, payload) { + try { + const sessionId = typeof payload?.session_id === "string" ? payload.session_id : ""; + const promptId = safeSegment(turnKey(payload), MAX_ID2); + if (!sessionId || !promptId) return; + const prompt = typeof payload?.prompt === "string" ? payload.prompt : ""; + const slash = prompt.trim().startsWith("/"); + const rows = readScoreRows(cfg, sessionId, { tailBytes: LOG_TAIL_BYTES }); + let prev = null; + let afterClear = false; + for (let i = rows.length - 1; i >= 0; i--) { + const r = rows[i]; + if (r.kind === "start" && r.source === "clear") afterClear = true; + if (r.kind === "turn" && typeof r.prompt_id === "string" && r.prompt_id && r.prompt_id !== promptId) { + prev = r; + break; + } + } + const correction = !slash && !!prev && !afterClear && isCorrection(prompt, { lastReplyEndedWithQuestion: prev?.ended_with_question === true }); + appendScoreRow(cfg, sessionId, { kind: "prompt", prompt_id: promptId, correction, slash }); + if (!correction || !prev || !implicitOutcomesEnabled(cfg)) return; + const used = Array.isArray(prev.used_refs) ? prev.used_refs.filter((v) => typeof v === "string" && v.trim()) : []; + const prevRun = typeof prev.run_id === "string" ? prev.run_id : ""; + if (!used.length || !prevRun) return; + spawnDetached( + cfg, + "drain", + ["--correct", String(prev.prompt_id), "--run", prevRun], + writePayload(cfg, payload) + ); + log(cfg, "debug", "stage-prompt: correction of the previous turn", { prompt_id: String(prev.prompt_id) }); + } catch (err) { + log(cfg, "warn", `stage-prompt: could not score the prompt (${messageOf(err)})`); + } +} function maybeDrain(cfg, runId, payload) { try { const { count, oldestMs } = spoolStats(cfg, runId); @@ -1572,8 +1761,8 @@ function maybeDrain(cfg, runId, payload) { } } function writePayload(cfg, payload) { - const dir = join9(resolveDataDir(cfg), "tmp"); - const file = join9(dir, `${randomUUID()}.json`); + const dir = join10(resolveDataDir(cfg), "tmp"); + const file = join10(dir, `${randomUUID()}.json`); ensureDir(dir); writeJsonAtomic(file, isObject3(payload) ? payload : {}); return file; @@ -1600,18 +1789,22 @@ function messageOf(err) { } return String(err); } -var BUDGET_MS, MAX_PROMPT_BYTES, MAX_ID; +var BUDGET_MS, MAX_PROMPT_BYTES, MAX_ID2, LOG_TAIL_BYTES; var init_stage_prompt = __esm({ async "../claude-code/hooks/src/stage-prompt.mjs"() { init_config(); + init_correction(); init_hook(); init_log(); + init_outcome(); init_runid(); + init_scorecard_log(); init_spool(); init_state(); BUDGET_MS = 250; MAX_PROMPT_BYTES = 64 * 1024; - MAX_ID = 128; + MAX_ID2 = 128; + LOG_TAIL_BYTES = 64 * 1024; await runHook("stage-prompt", { budgetMs: BUDGET_MS, body: async (payload) => { @@ -1625,6 +1818,7 @@ var init_stage_prompt = __esm({ return { suppressOutput: true }; } stageTurn(cfg, runId, payload); + if (cfg.capture !== false) scorePrompt(cfg, payload); if (cfg.capture) maybeDrain(cfg, runId, payload); return { suppressOutput: true }; } From 2f27570e068b3ab433f99c86c457b8ce482b1b3d Mon Sep 17 00:00:00 2001 From: Eldar <112889004+e1daru@users.noreply.github.com> Date: Tue, 29 Sep 2026 13:33:03 +0100 Subject: [PATCH 4/4] feat(outcome): credit memory by its short id; the once-per-turn review text (main) Published form of #29: the rebuilt bundles without their inline sourcemap line, and the docs, skills and manifests that ship. Sources and tests stay on pre-main. Co-Authored-By: Claude Opus 5.5 --- .claude-plugin/marketplace.json | 2 +- integrations/claude-code/bin/admin.mjs | 11 + integrations/claude-code/mcp/dist/index.js | 282 +++++++++++++++++++-- integrations/codex/bin/admin.mjs | 24 ++ integrations/codex/mcp/dist/index.js | 282 +++++++++++++++++++-- 5 files changed, 568 insertions(+), 33 deletions(-) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 5825a53..004f49f 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -15,7 +15,7 @@ "repository": "https://github.com/mubit-ai/plugins", "license": "Apache-2.0", "keywords": ["memory", "mubit", "long-term-memory", "lessons", "recall"], - "contextCost": { "value": 3125, "cached": 0 } + "contextCost": { "value": 3185, "cached": 0 } } ] } diff --git a/integrations/claude-code/bin/admin.mjs b/integrations/claude-code/bin/admin.mjs index 013d0c1..2de34f5 100755 --- a/integrations/claude-code/bin/admin.mjs +++ b/integrations/claude-code/bin/admin.mjs @@ -1959,6 +1959,17 @@ function str5(v) { return typeof v === "string" ? v.trim() : ""; } +// lib/handles.mjs +var ALPHABET = "abcdefghjkmnpqrstuvwxyz23456789"; +var LEN = 4; +var BODY = `[${ALPHABET}]{${LEN}}`; +var BARE_RE = new RegExp(`^m${BODY}$`); +var TAG_RE = new RegExp(`\\[m${BODY}\\]`, "g"); + +// lib/scorecard-log.mjs +var SCORE_LOG_TTL_MS = 7 * 24 * 60 * 60 * 1e3; +var MAX_READ_BYTES = 4 * 1024 * 1024; + // mcp/src/egress.mjs function selectLessons(rows, o) { const mine = (r) => o.runId !== "" && (r.runId === o.runId || r.sourceRunId === o.runId); diff --git a/integrations/claude-code/mcp/dist/index.js b/integrations/claude-code/mcp/dist/index.js index 1740d86..ce1571b 100644 --- a/integrations/claude-code/mcp/dist/index.js +++ b/integrations/claude-code/mcp/dist/index.js @@ -1057,8 +1057,8 @@ function safeCwd2() { } // mcp/src/egress.mjs -import { readdirSync as readdirSync3, statSync as statSync5 } from "node:fs"; -import { join as join8 } from "node:path"; +import { readdirSync as readdirSync4, statSync as statSync6 } from "node:fs"; +import { join as join9 } from "node:path"; // lib/breaker.mjs import { createHash as createHash3 } from "node:crypto"; @@ -2054,6 +2054,73 @@ function clamp2(v, lo, hi, dflt) { return Math.min(hi, Math.max(lo, Math.trunc(n))); } +// lib/handles.mjs +var ALPHABET = "abcdefghjkmnpqrstuvwxyz23456789"; +var LEN = 4; +var BODY = `[${ALPHABET}]{${LEN}}`; +var BARE_RE = new RegExp(`^m${BODY}$`); +var TAG_RE = new RegExp(`\\[m${BODY}\\]`, "g"); +function handleFor(ref) { + const s = typeof ref === "string" ? ref.trim() : ""; + if (!s) return ""; + let h = 2166136261; + for (let i = 0; i < s.length; i++) { + h ^= s.charCodeAt(i); + h = Math.imul(h, 16777619) >>> 0; + } + let out = "m"; + for (let i = 0; i < LEN; i++) { + out += ALPHABET[h % ALPHABET.length]; + h = Math.floor(h / ALPHABET.length); + } + return out; +} +function isHandle(v) { + return BARE_RE.test(bareOf(v)); +} +function resolveHandles(ids, knownRefs) { + const byHandle = /* @__PURE__ */ new Map(); + for (const ref of Array.isArray(knownRefs) ? knownRefs : []) { + const h = handleFor(ref); + if (h) byHandle.set(h, ref); + } + const out = []; + const unresolved = []; + for (const raw of Array.isArray(ids) ? ids : []) { + if (typeof raw !== "string") continue; + const bare = bareOf(raw); + if (BARE_RE.test(bare)) { + const ref = byHandle.get(bare); + if (ref) out.push(ref); + else { + out.push(bare); + unresolved.push(bare); + } + } else if (raw.trim()) { + out.push(raw.trim()); + } + } + return { ids: out, unresolved }; +} +function knownRefsFromRows(rows) { + const last = /* @__PURE__ */ new Map(); + let n = 0; + const note2 = (ref) => { + if (typeof ref === "string" && ref.trim()) last.set(ref.trim(), n++); + }; + for (const row of Array.isArray(rows) ? rows : []) { + if (!row || typeof row !== "object") continue; + if (row.kind !== "start" && row.kind !== "shown" && row.kind !== "refs") continue; + if (row.lessons && typeof row.lessons === "object") for (const ref of Object.keys(row.lessons)) note2(ref); + if (Array.isArray(row.refs)) for (const ref of row.refs) note2(ref); + } + return [...last.entries()].sort((a, b) => a[1] - b[1]).map(([ref]) => ref); +} +function bareOf(v) { + const s = typeof v === "string" ? v.trim() : ""; + return s.startsWith("[") && s.endsWith("]") ? s.slice(1, -1).trim() : s; +} + // lib/markers.mjs import { join as join7 } from "node:path"; function defaultMarker(runId = "") { @@ -2136,11 +2203,97 @@ function updateMarker(cfg, runId, patch = {}) { } } +// lib/scorecard-log.mjs +import { + closeSync as closeSync2, + fstatSync, + openSync as openSync2, + readdirSync as readdirSync3, + readFileSync as readFileSync5, + readSync, + statSync as statSync5, + writeSync as writeSync2 +} from "node:fs"; +import { dirname as dirname4, join as join8 } from "node:path"; +var SCORE_DIR = "scorecard"; +var SCORE_LOG_TTL_MS = 7 * 24 * 60 * 60 * 1e3; +var MAX_READ_BYTES = 4 * 1024 * 1024; +var MAX_ID = 128; +function scorecardPath(cfg, sessionId) { + const id = safeSegment(typeof sessionId === "string" ? sessionId.trim() : "", MAX_ID); + if (!id) return ""; + return join8(resolveDataDir(cfg), SCORE_DIR, `${id}.jsonl`); +} +function readRowsAt(p, opts = {}) { + try { + const size = statSync5(p).size; + const tail = Number(opts?.tailBytes); + const want = Number.isFinite(tail) && tail > 0 ? Math.min(tail, MAX_READ_BYTES) : MAX_READ_BYTES; + let text; + let partialHead = false; + if (size > want) { + const fd = openSync2(p, "r"); + try { + const buf = Buffer.alloc(want); + readSync(fd, buf, 0, want, size - want); + text = buf.toString("utf8"); + } finally { + closeSync2(fd); + } + partialHead = true; + } else { + text = readFileSync5(p, "utf8"); + } + const lines = text.split("\n"); + if (partialHead) lines.shift(); + const out = []; + for (const line of lines) { + if (!line.trim()) continue; + try { + const row = JSON.parse(line); + if (row && typeof row === "object" && !Array.isArray(row) && typeof row.kind === "string") { + out.push(row); + } + } catch { + } + } + return out; + } catch { + return []; + } +} +function recentScoreLogs(cfg, opts = {}) { + try { + const dir = join8(resolveDataDir(cfg), SCORE_DIR); + const now = Date.now(); + const maxAge = Number(opts?.maxAgeMs) > 0 ? Number(opts.maxAgeMs) : SCORE_LOG_TTL_MS; + const limit = Number(opts?.limit) > 0 ? Math.trunc(Number(opts.limit)) : 20; + const found = []; + for (const name of readdirSync3(dir)) { + if (!name.endsWith(".jsonl")) continue; + const p = join8(dir, name); + try { + const st = statSync5(p); + if (st.isFile() && now - st.mtimeMs <= maxAge) found.push({ p, m: st.mtimeMs }); + } catch { + } + } + found.sort((a, b) => b.m - a.m); + return found.slice(0, limit).map((f) => f.p); + } catch { + return []; + } +} + // mcp/src/egress.mjs var LATTICE = ["run", "session", "global", "org"]; var CEILINGS = ["run", "session", "global"]; var INGEST_PATH = "/v2/control/ingest"; var ARCHIVE_PATH = "/v2/control/archive"; +var OUTCOME_PATH = "/v2/control/outcome"; +var DEREFERENCE_PATH = "/v2/control/dereference"; +var HANDLES_NOTE_KEY = "mubit_handles"; +var HANDLES_HINT = "These ids match no memory line shown in this session and were sent as typed. Use the [m\u2026] id printed on a memory line, or a full reference_id."; var OPEN_TURN_GRACE_MS = 6e4; var TURN_FILES_TO_READ = 16; var LESSONS_PATH = "/v2/control/lessons"; @@ -2276,19 +2429,19 @@ function provenanceStamp(cfg, runId, sessionId, now = Date.now()) { return stamp; } function openTurn(cfg, runId, sessionId, now) { - const dir = join8(runDir(cfg, runId), "turns"); + const dir = join9(runDir(cfg, runId), "turns"); let names; try { - names = readdirSync3(dir).filter((f) => f.endsWith(".json")); + names = readdirSync4(dir).filter((f) => f.endsWith(".json")); } catch { return null; } if (names.length > TURN_FILES_TO_READ) { - names = names.map((f) => ({ f, at: mtimeOf(join8(dir, f)) })).sort((a, b) => b.at - a.at).slice(0, TURN_FILES_TO_READ).map((e) => e.f); + names = names.map((f) => ({ f, at: mtimeOf(join9(dir, f)) })).sort((a, b) => b.at - a.at).slice(0, TURN_FILES_TO_READ).map((e) => e.f); } let best = null; for (const f of names) { - const t = readJson(join8(dir, f), null); + const t = readJson(join9(dir, f), null); if (!isPlainObject3(t)) continue; if (String(t.session_id ?? "") !== sessionId) continue; if (!best || num2(t.started_at) > num2(best.started_at)) best = t; @@ -2302,7 +2455,7 @@ function openTurn(cfg, runId, sessionId, now) { } function mtimeOf(p) { try { - return statSync5(p).mtimeMs; + return statSync6(p).mtimeMs; } catch { return 0; } @@ -2440,6 +2593,50 @@ function recordMcpIngest(cfg, runId, items) { function countItems(body) { return Array.isArray(body?.items) ? body.items.length : 0; } +function resolveOutcomeBody(body, knownRefs) { + return resolveFields(body, knownRefs, true); +} +function resolveDereferenceBody(body, knownRefs) { + return resolveFields(body, knownRefs, false); +} +function knownRefsFor(cfg, sessionId) { + try { + const c = cfg ?? {}; + const own = sessionId ? scorecardPath(c, sessionId) : ""; + const rows = []; + for (const p of recentScoreLogs(c).reverse()) if (p !== own) rows.push(...readRowsAt(p)); + if (own) rows.push(...readRowsAt(own)); + return knownRefsFromRows(rows); + } catch { + return []; + } +} +function carriesHandle(body, withEntries) { + if (!body || typeof body !== "object" || Array.isArray(body)) return false; + if (isHandle(body.reference_id)) return true; + return withEntries && Array.isArray(body.entry_ids) && body.entry_ids.some((id) => isHandle(id)); +} +function resolveFields(body, knownRefs, withEntries) { + const noop = { body, changed: false, unresolved: ( + /** @type {string[]} */ + [] + ) }; + try { + if (!carriesHandle(body, withEntries)) return noop; + const unresolved = []; + const one = (id) => { + if (!isHandle(id)) return id; + const r = resolveHandles([id], knownRefs); + unresolved.push(...r.unresolved); + return r.ids[0] ?? id; + }; + const next = { ...body, reference_id: one(body.reference_id) }; + if (withEntries && Array.isArray(body.entry_ids)) next.entry_ids = body.entry_ids.map(one); + return { body: next, changed: true, unresolved }; + } catch { + return noop; + } +} function installFetchGuard(opts) { const ceiling = resolveCeiling(opts?.ceiling); const runId = typeof opts?.runId === "string" ? opts.runId : ""; @@ -2496,6 +2693,18 @@ function installFetchGuard(opts) { const out = stampProvenance(parsed.value, stampNow(), { at: "body" }); if (out.stamped) sendInit = { ...init, body: JSON.stringify(out.body) }; } + } else if (isPostTo(input, init, OUTCOME_PATH) || isPostTo(input, init, DEREFERENCE_PATH)) { + const parsed = parseBody(init); + const outcome = isPostTo(input, init, OUTCOME_PATH); + if (parsed.ok && carriesHandle(parsed.value, outcome)) { + const refs = knownRefsFor(opts?.cfg, sessionId); + const out = outcome ? resolveOutcomeBody(parsed.value, refs) : resolveDereferenceBody(parsed.value, refs); + if (out.changed) sendInit = { ...init, body: JSON.stringify(out.body) }; + if (out.unresolved.length) { + note2 = { unresolved: out.unresolved, hint: HANDLES_HINT }; + noteKey = HANDLES_NOTE_KEY; + } + } } else if (isLessonsRead(input, init)) { const plan = await planLessons(init); if (plan.answer) return jsonResponse(plan.answer); @@ -2592,8 +2801,16 @@ var INSTRUCTIONS = [ "", "Which tool. mubit_recall for a topic or question in words. mubit_diagnose when a command or test has just failed, which matches the error shape against past failures. mubit_dereference when you already hold a reference_id. Reviewing the whole catalogue, the pattern across many lessons, a named checkpoint, deleting a lesson and an explicit reflect are skills (/mubit-memory:strategies, :checkpoint, :forget, :reflect), not tools.", "", - 'What to write back. mubit_learned records one durable claim \u2014 a constraint, a fix that worked, a standing preference \u2014 stated so it is still true in a later session. It is not a session log: narrating what happened ("the user asked for X", "I refactored Y") is the common way this tool is misused, and every future recall pays for it. mubit_outcome credits the reference_ids that actually helped, which is what makes the memory that helps rank higher next time.' + 'What to write back. mubit_learned records one durable claim \u2014 a constraint, a fix that worked, a standing preference \u2014 stated so it is still true in a later session. It is not a session log: narrating what happened ("the user asked for X", "I refactored Y") is the common way this tool is misused, and every future recall pays for it. Each injected memory line starts with an id in brackets, like [m7k2q]: pass those ids (or reference_ids) to mubit_outcome \u2014 outcome success for entries that helped, failure for ones that were wrong or misled you \u2014 which is what makes the memory that helps rank higher next time.' ].join("\n"); +var ALWAYS_LOAD_META = "anthropic/alwaysLoad"; +var ALWAYS_LOAD_TOOLS = Object.freeze(["mubit_outcome", "mubit_learned"]); +function alwaysLoadFor(cfg) { + const c = cfg && typeof cfg === "object" ? cfg : {}; + if (String(c.outcomeReview ?? "").trim().toLowerCase() === "off") return []; + if (c.host === "codex") return []; + return [...ALWAYS_LOAD_TOOLS]; +} function guardInitialize(message, instructions) { const noop = { message, changed: false }; try { @@ -2614,8 +2831,31 @@ function guardInitialize(message, instructions) { return noop; } } +function guardToolsList(message, names) { + const noop = { message, changed: false }; + try { + const wanted = new Set(Array.isArray(names) ? names : []); + if (!wanted.size) return noop; + if (!message || typeof message !== "object" || Array.isArray(message)) return noop; + if (message.jsonrpc !== "2.0") return noop; + const result = message.result; + if (!result || typeof result !== "object" || !Array.isArray(result.tools)) return noop; + let changed = false; + const tools = result.tools.map((t) => { + if (!t || typeof t !== "object" || !wanted.has(t.name)) return t; + const meta = t._meta && typeof t._meta === "object" && !Array.isArray(t._meta) ? t._meta : {}; + if (meta[ALWAYS_LOAD_META] === true) return t; + changed = true; + return { ...t, _meta: { ...meta, [ALWAYS_LOAD_META]: true } }; + }); + return changed ? { message: { ...message, result: { ...result, tools } }, changed } : noop; + } catch { + return noop; + } +} function installInstructionsGuard(opts) { const instructions = typeof opts?.instructions === "string" ? opts.instructions : ""; + const alwaysLoad = Array.isArray(opts?.alwaysLoad) ? opts.alwaysLoad.filter((n) => typeof n === "string" && n.trim()) : []; if (instructions.trim() === "") return; const stream = opts?.stream ?? process.stdout; const current = stream?.write; @@ -2633,6 +2873,13 @@ function installInstructionsGuard(opts) { } catch { } } + if (alwaysLoad.length && typeof chunk === "string" && chunk.includes('"tools":[')) { + try { + const marked = rewriteLines(chunk, (frame) => guardToolsList(frame, alwaysLoad)); + if (marked !== null) chunk = marked; + } catch { + } + } return base.call(this, chunk, ...rest); }; Object.defineProperty(wrapped, "mubitInstructionsGuardOriginal", { @@ -2641,11 +2888,14 @@ function installInstructionsGuard(opts) { configurable: true, enumerable: false }); - wrapped.mubitInstructionsGuard = { chars: instructions.length }; + wrapped.mubitInstructionsGuard = { chars: instructions.length, alwaysLoad }; stream.write = wrapped; } function fill(chunk, instructions) { if (!chunk.includes('"result"')) return null; + return rewriteLines(chunk, (frame) => guardInitialize(frame, instructions)); +} +function rewriteLines(chunk, guard) { const parts = chunk.split("\n"); let changed = false; for (let i = 0; i < parts.length; i += 1) { @@ -2656,7 +2906,7 @@ function fill(chunk, instructions) { } catch { continue; } - const out = guardInitialize(frame, instructions); + const out = guard(frame); if (!out.changed) continue; parts[i] = JSON.stringify(out.message); changed = true; @@ -2666,7 +2916,7 @@ function fill(chunk, instructions) { // mcp/src/results.mjs import { mkdirSync as mkdirSync3, writeFileSync as writeFileSync2 } from "node:fs"; -import { join as join10 } from "node:path"; +import { join as join11 } from "node:path"; // lib/assemble.mjs var SECTION_KEYS = Object.freeze([ @@ -2752,7 +3002,7 @@ function firstClause(text) { } // lib/seen.mjs -import { join as join9 } from "node:path"; +import { join as join10 } from "node:path"; var SEEN_TTL_MS = 6 * 60 * 60 * 1e3; var MAX_SEEN_REFS = 512; var SEEN_DIR = "seen"; @@ -2764,7 +3014,7 @@ function seenPath(cfg, runId, sessionId) { if (!safeSegment(runId)) return ""; const session = safeSegment(hostSessionId({ session_id: sessionId }), MAX_SESSION_SEGMENT); if (!session) return ""; - return join9(runDir(cfg, runId), SEEN_DIR, `${session}.json`); + return join10(runDir(cfg, runId), SEEN_DIR, `${session}.json`); } function readSeen(cfg, runId, sessionId = "") { try { @@ -3083,10 +3333,10 @@ function spillWriter(cfg, runId) { return (text, shape) => { try { if (!safeSegment(runId)) return ""; - const dir = join10(runDir(cfg, runId), SPILL_DIR); + const dir = join11(runDir(cfg, runId), SPILL_DIR); mkdirSync3(dir, { recursive: true }); const ext = shape === "text" || shape === "error" ? "txt" : "json"; - const p = join10(dir, `${Date.now()}-${safeSegment(shape) || "result"}-${n++}.${ext}`); + const p = join11(dir, `${Date.now()}-${safeSegment(shape) || "result"}-${n++}.${ext}`); writeFileSync2(p, text, { encoding: "utf8", mode: 384 }); return p; } catch { @@ -3183,7 +3433,7 @@ function prepare(env) { const sessionId = hostPayload(env).session_id ?? ""; const ceiling = resolveCeiling(cfg.mcpLessonScope); installFetchGuard({ ceiling, runId, pinRun: true, cfg, sessionId }); - installInstructionsGuard({ instructions: INSTRUCTIONS }); + installInstructionsGuard({ instructions: INSTRUCTIONS, alwaysLoad: alwaysLoadFor(cfg) }); installResultsGuard({ cfg, runId, diff --git a/integrations/codex/bin/admin.mjs b/integrations/codex/bin/admin.mjs index e1c747a..2bc8a1d 100755 --- a/integrations/codex/bin/admin.mjs +++ b/integrations/codex/bin/admin.mjs @@ -2035,6 +2035,18 @@ var init_runpick = __esm({ } }); +// ../claude-code/lib/handles.mjs +var ALPHABET, LEN, BODY, BARE_RE, TAG_RE; +var init_handles = __esm({ + "../claude-code/lib/handles.mjs"() { + ALPHABET = "abcdefghjkmnpqrstuvwxyz23456789"; + LEN = 4; + BODY = `[${ALPHABET}]{${LEN}}`; + BARE_RE = new RegExp(`^m${BODY}$`); + TAG_RE = new RegExp(`\\[m${BODY}\\]`, "g"); + } +}); + // ../claude-code/lib/markers.mjs var init_markers = __esm({ "../claude-code/lib/markers.mjs"() { @@ -2042,6 +2054,16 @@ var init_markers = __esm({ } }); +// ../claude-code/lib/scorecard-log.mjs +var SCORE_LOG_TTL_MS, MAX_READ_BYTES; +var init_scorecard_log = __esm({ + "../claude-code/lib/scorecard-log.mjs"() { + init_state(); + SCORE_LOG_TTL_MS = 7 * 24 * 60 * 60 * 1e3; + MAX_READ_BYTES = 4 * 1024 * 1024; + } +}); + // ../claude-code/mcp/src/egress.mjs function selectLessons(rows, o) { const mine = (r) => o.runId !== "" && (r.runId === o.runId || r.sourceRunId === o.runId); @@ -2067,8 +2089,10 @@ var SHOWING; var init_egress = __esm({ "../claude-code/mcp/src/egress.mjs"() { init_activity(); + init_handles(); init_markers(); init_runid(); + init_scorecard_log(); init_state(); SHOWING = { "": "this run, plus every lesson stored at a scope that reaches past the run that wrote it", diff --git a/integrations/codex/mcp/dist/index.js b/integrations/codex/mcp/dist/index.js index d02c894..cee6c69 100644 --- a/integrations/codex/mcp/dist/index.js +++ b/integrations/codex/mcp/dist/index.js @@ -1057,8 +1057,8 @@ function safeCwd2() { } // ../claude-code/mcp/src/egress.mjs -import { readdirSync as readdirSync3, statSync as statSync5 } from "node:fs"; -import { join as join8 } from "node:path"; +import { readdirSync as readdirSync4, statSync as statSync6 } from "node:fs"; +import { join as join9 } from "node:path"; // ../claude-code/lib/breaker.mjs import { createHash as createHash3 } from "node:crypto"; @@ -2054,6 +2054,73 @@ function clamp2(v, lo, hi, dflt) { return Math.min(hi, Math.max(lo, Math.trunc(n))); } +// ../claude-code/lib/handles.mjs +var ALPHABET = "abcdefghjkmnpqrstuvwxyz23456789"; +var LEN = 4; +var BODY = `[${ALPHABET}]{${LEN}}`; +var BARE_RE = new RegExp(`^m${BODY}$`); +var TAG_RE = new RegExp(`\\[m${BODY}\\]`, "g"); +function handleFor(ref) { + const s = typeof ref === "string" ? ref.trim() : ""; + if (!s) return ""; + let h = 2166136261; + for (let i = 0; i < s.length; i++) { + h ^= s.charCodeAt(i); + h = Math.imul(h, 16777619) >>> 0; + } + let out = "m"; + for (let i = 0; i < LEN; i++) { + out += ALPHABET[h % ALPHABET.length]; + h = Math.floor(h / ALPHABET.length); + } + return out; +} +function isHandle(v) { + return BARE_RE.test(bareOf(v)); +} +function resolveHandles(ids, knownRefs) { + const byHandle = /* @__PURE__ */ new Map(); + for (const ref of Array.isArray(knownRefs) ? knownRefs : []) { + const h = handleFor(ref); + if (h) byHandle.set(h, ref); + } + const out = []; + const unresolved = []; + for (const raw of Array.isArray(ids) ? ids : []) { + if (typeof raw !== "string") continue; + const bare = bareOf(raw); + if (BARE_RE.test(bare)) { + const ref = byHandle.get(bare); + if (ref) out.push(ref); + else { + out.push(bare); + unresolved.push(bare); + } + } else if (raw.trim()) { + out.push(raw.trim()); + } + } + return { ids: out, unresolved }; +} +function knownRefsFromRows(rows) { + const last = /* @__PURE__ */ new Map(); + let n = 0; + const note2 = (ref) => { + if (typeof ref === "string" && ref.trim()) last.set(ref.trim(), n++); + }; + for (const row of Array.isArray(rows) ? rows : []) { + if (!row || typeof row !== "object") continue; + if (row.kind !== "start" && row.kind !== "shown" && row.kind !== "refs") continue; + if (row.lessons && typeof row.lessons === "object") for (const ref of Object.keys(row.lessons)) note2(ref); + if (Array.isArray(row.refs)) for (const ref of row.refs) note2(ref); + } + return [...last.entries()].sort((a, b) => a[1] - b[1]).map(([ref]) => ref); +} +function bareOf(v) { + const s = typeof v === "string" ? v.trim() : ""; + return s.startsWith("[") && s.endsWith("]") ? s.slice(1, -1).trim() : s; +} + // ../claude-code/lib/markers.mjs import { join as join7 } from "node:path"; function defaultMarker(runId = "") { @@ -2136,11 +2203,97 @@ function updateMarker(cfg, runId, patch = {}) { } } +// ../claude-code/lib/scorecard-log.mjs +import { + closeSync as closeSync2, + fstatSync, + openSync as openSync2, + readdirSync as readdirSync3, + readFileSync as readFileSync5, + readSync, + statSync as statSync5, + writeSync as writeSync2 +} from "node:fs"; +import { dirname as dirname4, join as join8 } from "node:path"; +var SCORE_DIR = "scorecard"; +var SCORE_LOG_TTL_MS = 7 * 24 * 60 * 60 * 1e3; +var MAX_READ_BYTES = 4 * 1024 * 1024; +var MAX_ID = 128; +function scorecardPath(cfg, sessionId) { + const id = safeSegment(typeof sessionId === "string" ? sessionId.trim() : "", MAX_ID); + if (!id) return ""; + return join8(resolveDataDir(cfg), SCORE_DIR, `${id}.jsonl`); +} +function readRowsAt(p, opts = {}) { + try { + const size = statSync5(p).size; + const tail = Number(opts?.tailBytes); + const want = Number.isFinite(tail) && tail > 0 ? Math.min(tail, MAX_READ_BYTES) : MAX_READ_BYTES; + let text; + let partialHead = false; + if (size > want) { + const fd = openSync2(p, "r"); + try { + const buf = Buffer.alloc(want); + readSync(fd, buf, 0, want, size - want); + text = buf.toString("utf8"); + } finally { + closeSync2(fd); + } + partialHead = true; + } else { + text = readFileSync5(p, "utf8"); + } + const lines = text.split("\n"); + if (partialHead) lines.shift(); + const out = []; + for (const line of lines) { + if (!line.trim()) continue; + try { + const row = JSON.parse(line); + if (row && typeof row === "object" && !Array.isArray(row) && typeof row.kind === "string") { + out.push(row); + } + } catch { + } + } + return out; + } catch { + return []; + } +} +function recentScoreLogs(cfg, opts = {}) { + try { + const dir = join8(resolveDataDir(cfg), SCORE_DIR); + const now = Date.now(); + const maxAge = Number(opts?.maxAgeMs) > 0 ? Number(opts.maxAgeMs) : SCORE_LOG_TTL_MS; + const limit = Number(opts?.limit) > 0 ? Math.trunc(Number(opts.limit)) : 20; + const found = []; + for (const name of readdirSync3(dir)) { + if (!name.endsWith(".jsonl")) continue; + const p = join8(dir, name); + try { + const st = statSync5(p); + if (st.isFile() && now - st.mtimeMs <= maxAge) found.push({ p, m: st.mtimeMs }); + } catch { + } + } + found.sort((a, b) => b.m - a.m); + return found.slice(0, limit).map((f) => f.p); + } catch { + return []; + } +} + // ../claude-code/mcp/src/egress.mjs var LATTICE = ["run", "session", "global", "org"]; var CEILINGS = ["run", "session", "global"]; var INGEST_PATH = "/v2/control/ingest"; var ARCHIVE_PATH = "/v2/control/archive"; +var OUTCOME_PATH = "/v2/control/outcome"; +var DEREFERENCE_PATH = "/v2/control/dereference"; +var HANDLES_NOTE_KEY = "mubit_handles"; +var HANDLES_HINT = "These ids match no memory line shown in this session and were sent as typed. Use the [m\u2026] id printed on a memory line, or a full reference_id."; var OPEN_TURN_GRACE_MS = 6e4; var TURN_FILES_TO_READ = 16; var LESSONS_PATH = "/v2/control/lessons"; @@ -2276,19 +2429,19 @@ function provenanceStamp(cfg, runId, sessionId, now = Date.now()) { return stamp; } function openTurn(cfg, runId, sessionId, now) { - const dir = join8(runDir(cfg, runId), "turns"); + const dir = join9(runDir(cfg, runId), "turns"); let names; try { - names = readdirSync3(dir).filter((f) => f.endsWith(".json")); + names = readdirSync4(dir).filter((f) => f.endsWith(".json")); } catch { return null; } if (names.length > TURN_FILES_TO_READ) { - names = names.map((f) => ({ f, at: mtimeOf(join8(dir, f)) })).sort((a, b) => b.at - a.at).slice(0, TURN_FILES_TO_READ).map((e) => e.f); + names = names.map((f) => ({ f, at: mtimeOf(join9(dir, f)) })).sort((a, b) => b.at - a.at).slice(0, TURN_FILES_TO_READ).map((e) => e.f); } let best = null; for (const f of names) { - const t = readJson(join8(dir, f), null); + const t = readJson(join9(dir, f), null); if (!isPlainObject3(t)) continue; if (String(t.session_id ?? "") !== sessionId) continue; if (!best || num2(t.started_at) > num2(best.started_at)) best = t; @@ -2302,7 +2455,7 @@ function openTurn(cfg, runId, sessionId, now) { } function mtimeOf(p) { try { - return statSync5(p).mtimeMs; + return statSync6(p).mtimeMs; } catch { return 0; } @@ -2440,6 +2593,50 @@ function recordMcpIngest(cfg, runId, items) { function countItems(body) { return Array.isArray(body?.items) ? body.items.length : 0; } +function resolveOutcomeBody(body, knownRefs) { + return resolveFields(body, knownRefs, true); +} +function resolveDereferenceBody(body, knownRefs) { + return resolveFields(body, knownRefs, false); +} +function knownRefsFor(cfg, sessionId) { + try { + const c = cfg ?? {}; + const own = sessionId ? scorecardPath(c, sessionId) : ""; + const rows = []; + for (const p of recentScoreLogs(c).reverse()) if (p !== own) rows.push(...readRowsAt(p)); + if (own) rows.push(...readRowsAt(own)); + return knownRefsFromRows(rows); + } catch { + return []; + } +} +function carriesHandle(body, withEntries) { + if (!body || typeof body !== "object" || Array.isArray(body)) return false; + if (isHandle(body.reference_id)) return true; + return withEntries && Array.isArray(body.entry_ids) && body.entry_ids.some((id) => isHandle(id)); +} +function resolveFields(body, knownRefs, withEntries) { + const noop = { body, changed: false, unresolved: ( + /** @type {string[]} */ + [] + ) }; + try { + if (!carriesHandle(body, withEntries)) return noop; + const unresolved = []; + const one = (id) => { + if (!isHandle(id)) return id; + const r = resolveHandles([id], knownRefs); + unresolved.push(...r.unresolved); + return r.ids[0] ?? id; + }; + const next = { ...body, reference_id: one(body.reference_id) }; + if (withEntries && Array.isArray(body.entry_ids)) next.entry_ids = body.entry_ids.map(one); + return { body: next, changed: true, unresolved }; + } catch { + return noop; + } +} function installFetchGuard(opts) { const ceiling = resolveCeiling(opts?.ceiling); const runId = typeof opts?.runId === "string" ? opts.runId : ""; @@ -2496,6 +2693,18 @@ function installFetchGuard(opts) { const out = stampProvenance(parsed.value, stampNow(), { at: "body" }); if (out.stamped) sendInit = { ...init, body: JSON.stringify(out.body) }; } + } else if (isPostTo(input, init, OUTCOME_PATH) || isPostTo(input, init, DEREFERENCE_PATH)) { + const parsed = parseBody(init); + const outcome = isPostTo(input, init, OUTCOME_PATH); + if (parsed.ok && carriesHandle(parsed.value, outcome)) { + const refs = knownRefsFor(opts?.cfg, sessionId); + const out = outcome ? resolveOutcomeBody(parsed.value, refs) : resolveDereferenceBody(parsed.value, refs); + if (out.changed) sendInit = { ...init, body: JSON.stringify(out.body) }; + if (out.unresolved.length) { + note2 = { unresolved: out.unresolved, hint: HANDLES_HINT }; + noteKey = HANDLES_NOTE_KEY; + } + } } else if (isLessonsRead(input, init)) { const plan = await planLessons(init); if (plan.answer) return jsonResponse(plan.answer); @@ -2592,8 +2801,16 @@ var INSTRUCTIONS = [ "", "Which tool. mubit_recall for a topic or question in words. mubit_diagnose when a command or test has just failed, which matches the error shape against past failures. mubit_dereference when you already hold a reference_id. Reviewing the whole catalogue, the pattern across many lessons, a named checkpoint, deleting a lesson and an explicit reflect are skills (/mubit-memory:strategies, :checkpoint, :forget, :reflect), not tools.", "", - 'What to write back. mubit_learned records one durable claim \u2014 a constraint, a fix that worked, a standing preference \u2014 stated so it is still true in a later session. It is not a session log: narrating what happened ("the user asked for X", "I refactored Y") is the common way this tool is misused, and every future recall pays for it. mubit_outcome credits the reference_ids that actually helped, which is what makes the memory that helps rank higher next time.' + 'What to write back. mubit_learned records one durable claim \u2014 a constraint, a fix that worked, a standing preference \u2014 stated so it is still true in a later session. It is not a session log: narrating what happened ("the user asked for X", "I refactored Y") is the common way this tool is misused, and every future recall pays for it. Each injected memory line starts with an id in brackets, like [m7k2q]: pass those ids (or reference_ids) to mubit_outcome \u2014 outcome success for entries that helped, failure for ones that were wrong or misled you \u2014 which is what makes the memory that helps rank higher next time.' ].join("\n"); +var ALWAYS_LOAD_META = "anthropic/alwaysLoad"; +var ALWAYS_LOAD_TOOLS = Object.freeze(["mubit_outcome", "mubit_learned"]); +function alwaysLoadFor(cfg) { + const c = cfg && typeof cfg === "object" ? cfg : {}; + if (String(c.outcomeReview ?? "").trim().toLowerCase() === "off") return []; + if (c.host === "codex") return []; + return [...ALWAYS_LOAD_TOOLS]; +} function guardInitialize(message, instructions) { const noop = { message, changed: false }; try { @@ -2614,8 +2831,31 @@ function guardInitialize(message, instructions) { return noop; } } +function guardToolsList(message, names) { + const noop = { message, changed: false }; + try { + const wanted = new Set(Array.isArray(names) ? names : []); + if (!wanted.size) return noop; + if (!message || typeof message !== "object" || Array.isArray(message)) return noop; + if (message.jsonrpc !== "2.0") return noop; + const result = message.result; + if (!result || typeof result !== "object" || !Array.isArray(result.tools)) return noop; + let changed = false; + const tools = result.tools.map((t) => { + if (!t || typeof t !== "object" || !wanted.has(t.name)) return t; + const meta = t._meta && typeof t._meta === "object" && !Array.isArray(t._meta) ? t._meta : {}; + if (meta[ALWAYS_LOAD_META] === true) return t; + changed = true; + return { ...t, _meta: { ...meta, [ALWAYS_LOAD_META]: true } }; + }); + return changed ? { message: { ...message, result: { ...result, tools } }, changed } : noop; + } catch { + return noop; + } +} function installInstructionsGuard(opts) { const instructions = typeof opts?.instructions === "string" ? opts.instructions : ""; + const alwaysLoad = Array.isArray(opts?.alwaysLoad) ? opts.alwaysLoad.filter((n) => typeof n === "string" && n.trim()) : []; if (instructions.trim() === "") return; const stream = opts?.stream ?? process.stdout; const current = stream?.write; @@ -2633,6 +2873,13 @@ function installInstructionsGuard(opts) { } catch { } } + if (alwaysLoad.length && typeof chunk === "string" && chunk.includes('"tools":[')) { + try { + const marked = rewriteLines(chunk, (frame) => guardToolsList(frame, alwaysLoad)); + if (marked !== null) chunk = marked; + } catch { + } + } return base.call(this, chunk, ...rest); }; Object.defineProperty(wrapped, "mubitInstructionsGuardOriginal", { @@ -2641,11 +2888,14 @@ function installInstructionsGuard(opts) { configurable: true, enumerable: false }); - wrapped.mubitInstructionsGuard = { chars: instructions.length }; + wrapped.mubitInstructionsGuard = { chars: instructions.length, alwaysLoad }; stream.write = wrapped; } function fill(chunk, instructions) { if (!chunk.includes('"result"')) return null; + return rewriteLines(chunk, (frame) => guardInitialize(frame, instructions)); +} +function rewriteLines(chunk, guard) { const parts = chunk.split("\n"); let changed = false; for (let i = 0; i < parts.length; i += 1) { @@ -2656,7 +2906,7 @@ function fill(chunk, instructions) { } catch { continue; } - const out = guardInitialize(frame, instructions); + const out = guard(frame); if (!out.changed) continue; parts[i] = JSON.stringify(out.message); changed = true; @@ -2666,7 +2916,7 @@ function fill(chunk, instructions) { // ../claude-code/mcp/src/results.mjs import { mkdirSync as mkdirSync3, writeFileSync as writeFileSync2 } from "node:fs"; -import { join as join10 } from "node:path"; +import { join as join11 } from "node:path"; // ../claude-code/lib/assemble.mjs var SECTION_KEYS = Object.freeze([ @@ -2752,7 +3002,7 @@ function firstClause(text) { } // ../claude-code/lib/seen.mjs -import { join as join9 } from "node:path"; +import { join as join10 } from "node:path"; var SEEN_TTL_MS = 6 * 60 * 60 * 1e3; var MAX_SEEN_REFS = 512; var SEEN_DIR = "seen"; @@ -2764,7 +3014,7 @@ function seenPath(cfg, runId, sessionId) { if (!safeSegment(runId)) return ""; const session = safeSegment(hostSessionId({ session_id: sessionId }), MAX_SESSION_SEGMENT); if (!session) return ""; - return join9(runDir(cfg, runId), SEEN_DIR, `${session}.json`); + return join10(runDir(cfg, runId), SEEN_DIR, `${session}.json`); } function readSeen(cfg, runId, sessionId = "") { try { @@ -3083,10 +3333,10 @@ function spillWriter(cfg, runId) { return (text, shape) => { try { if (!safeSegment(runId)) return ""; - const dir = join10(runDir(cfg, runId), SPILL_DIR); + const dir = join11(runDir(cfg, runId), SPILL_DIR); mkdirSync3(dir, { recursive: true }); const ext = shape === "text" || shape === "error" ? "txt" : "json"; - const p = join10(dir, `${Date.now()}-${safeSegment(shape) || "result"}-${n++}.${ext}`); + const p = join11(dir, `${Date.now()}-${safeSegment(shape) || "result"}-${n++}.${ext}`); writeFileSync2(p, text, { encoding: "utf8", mode: 384 }); return p; } catch { @@ -3183,7 +3433,7 @@ function prepare(env) { const sessionId = hostPayload(env).session_id ?? ""; const ceiling = resolveCeiling(cfg.mcpLessonScope); installFetchGuard({ ceiling, runId, pinRun: true, cfg, sessionId }); - installInstructionsGuard({ instructions: INSTRUCTIONS }); + installInstructionsGuard({ instructions: INSTRUCTIONS, alwaysLoad: alwaysLoadFor(cfg) }); installResultsGuard({ cfg, runId,