From accb0312aa31c2f7a95c52d62404b17e9961ea4e Mon Sep 17 00:00:00 2001 From: Sebastian Gonzalez Date: Mon, 21 Sep 2026 14:01:38 -0700 Subject: [PATCH 1/3] fix(capture): deny add_memory in sessions the plugin already captures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit add_memory is for MCP clients with no capture hooks. In a Claude Code session running this plugin every turn is already uploaded verbatim by the Stop hook, so an add_memory call only writes a second conversation whose user turn the agent paraphrased, which Studio then lists beside the real session. A PreToolUse gate denies MemHub's add_memory (any server exposing it: the plugin's own and a claude.ai connector) when per-turn capture is live — not opted out, a transcript present, and a credential capture can use, judged by the capture-health banner's own check — and points the agent at save_artifact. Everywhere else it allows the call, and it fails open. Co-Authored-By: Claude Opus 5 (1M context) --- README.md | 12 ++ plugins/memhub/hooks/claude-hooks.json | 10 ++ plugins/memhub/scripts/add_memory_gate.py | 125 ++++++++++++++++ tests/add_memory_gate_test.py | 172 ++++++++++++++++++++++ tests/claude_hook_guard_test.py | 3 +- 5 files changed, 321 insertions(+), 1 deletion(-) create mode 100644 plugins/memhub/scripts/add_memory_gate.py create mode 100644 tests/add_memory_gate_test.py diff --git a/README.md b/README.md index 868b2bce..e78290cf 100644 --- a/README.md +++ b/README.md @@ -231,6 +231,18 @@ reads and manual imports. The state retains one exact sample per measured record rather than a second transcript archive. Unobserved historical usage stays unknown; a later read never invents missing measurements. +### No second copy through `add_memory` + +`add_memory` saves a turn for MCP clients that capture nothing themselves. +While this plugin is capturing a Claude Code session (per-turn capture not +switched off, and a credential capture can use), a `PreToolUse` hook +(`add_memory_gate.py`) denies it from any server that exposes it, including a +claude.ai MemHub connector. The session is already stored verbatim; an +`add_memory` call there only writes a second conversation, with a user turn +the agent reconstructed. The denial points the agent at `save_artifact`, +which is where findings meant for a brain belong. With capture off the call +is allowed, and the hook fails open. + ### Directive recall Independent of capture: `PreToolUse` (Edit/Write/NotebookEdit/Bash) and diff --git a/plugins/memhub/hooks/claude-hooks.json b/plugins/memhub/hooks/claude-hooks.json index 8608439a..86c4c474 100644 --- a/plugins/memhub/hooks/claude-hooks.json +++ b/plugins/memhub/hooks/claude-hooks.json @@ -33,6 +33,16 @@ "command": "IN=$(cat); if [ -n \"${CLAUDE_PLUGIN_ROOT:-}\" ] && printf %s \"$IN\" | python3 \"${CLAUDE_PLUGIN_ROOT}/scripts/claude_hook_guard.py\" ignore PreToolUse; then printf %s \"$IN\" | python3 \"${CLAUDE_PLUGIN_ROOT}/scripts/rulebook_hook.py\" pre; fi" } ] + }, + { + "matcher": "^mcp__.+__add_memory$", + "hooks": [ + { + "type": "command", + "timeout": 5, + "command": "IN=$(cat); if [ -n \"${CLAUDE_PLUGIN_ROOT:-}\" ] && printf %s \"$IN\" | python3 \"${CLAUDE_PLUGIN_ROOT}/scripts/claude_hook_guard.py\" ignore PreToolUse; then printf %s \"$IN\" | python3 \"${CLAUDE_PLUGIN_ROOT}/scripts/add_memory_gate.py\"; fi" + } + ] } ], "PostToolUse": [ diff --git a/plugins/memhub/scripts/add_memory_gate.py b/plugins/memhub/scripts/add_memory_gate.py new file mode 100644 index 00000000..0e75ee39 --- /dev/null +++ b/plugins/memhub/scripts/add_memory_gate.py @@ -0,0 +1,125 @@ +#!/usr/bin/env python3 +"""Deny ``add_memory`` in a session this plugin is already capturing (stdlib only). + +**The failure.** ``add_memory`` exists for MCP clients with no capture hooks: the +caller passes one turn — what the user said, what the assistant replied — and +the server stores it as a conversation. In a Claude Code session running this +plugin that is never needed, because the ``Stop`` hook already uploads every +turn verbatim. Agents call it anyway, to make a finding "findable later" in a +brain, and they fill ``user_message`` with a paraphrase they wrote themselves. +The server stores that as a second conversation, whose user turn the user never +typed, and Studio lists it beside the real session. Blind live sessions on an +unmodified build, asked only to put a finished answer into a new brain, did +this more often than not (the trials are in the PR that added this file). +``save_artifact`` is the tool for what they were after, and the agents that +did not call ``add_memory`` used it. + +**Why here and not on the server.** The server cannot know whether a client +captures its own sessions; only this plugin knows its ``Stop`` hook is live. So +the rule is scoped to exactly that condition and nothing wider: + +* the tool is MemHub's ``add_memory`` — matched on the name's suffix AND on its + ``user_message`` argument, so it covers this plugin's own server and a + claude.ai MemHub connector alike (the blind agents used the connector) without + catching an unrelated server's ``add_memory``; +* per-turn capture is not switched off (``MEMHUB_TURN_FLUSH=0``); +* the payload names a transcript, which is what the flush reads; +* the plugin holds a credential capture can authenticate with, judged by the + same network-free check the capture-health banner uses — so this and that + banner can never disagree about whether capture is running. + +When any of those is false, capture is not happening and ``add_memory`` is the +only way to save the turn, so the call is allowed. + +**Fails OPEN.** Any unexpected error allows the call: a broken guard must not +take a tool away from a session. + +Run the self-test: python3 tests/add_memory_gate_test.py (from the repo root) +""" +from __future__ import annotations + +import json +import os +import re +import sys +from pathlib import Path +from typing import Mapping + +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +# `mcp____add_memory`, whatever the server segment is called. +_ADD_MEMORY = re.compile(r"^mcp__.+__add_memory$") +_CAPTURE_OFF = {"0", "off", "false"} +# `_token_problem` verdicts under which the flush still authenticates today: +# none at all, a key close to expiry, and an OAuth token that works but cannot +# renew. The other verdicts ("never", "key_expired", "unrenewable") mean the +# flush has no working credential, so nothing is being captured. +_CREDENTIAL_WORKS = {None, "key_expiring", "no_refresh"} + +DENY_REASON = ( + "MemHub is already capturing this Claude Code session: the plugin uploads " + "every turn verbatim. add_memory here would store a second conversation " + "whose user message the user never typed, and teammates would see it as a " + "separate session. To keep findings where a brain's readers will find " + "them, save them with save_artifact (pass agent_brain_id). Do not retry " + "add_memory in this session." +) +USER_LINE = ("MemHub: blocked add_memory. This session is already captured, " + "so findings belong in save_artifact.") + + +def is_memhub_add_memory(payload: object) -> bool: + """True for MemHub's add_memory tool, from any server that exposes it.""" + if not isinstance(payload, dict): + return False + name = payload.get("tool_name") + if not isinstance(name, str) or not _ADD_MEMORY.match(name): + return False + args = payload.get("tool_input") + return isinstance(args, dict) and "user_message" in args + + +def capture_is_active(payload: dict, + environ: Mapping[str, str] | None = None) -> bool: + """True when this plugin's per-turn capture is running for this session.""" + env = os.environ if environ is None else environ + if env.get("MEMHUB_TURN_FLUSH", "").strip().lower() in _CAPTURE_OFF: + return False + transcript = payload.get("transcript_path") + if not isinstance(transcript, str) or not transcript.strip(): + return False + import capture_health # noqa: PLC0415 — stdlib-only, beside this file + host = capture_health._env_host() + if not host: + return False + return capture_health._token_problem(host) in _CREDENTIAL_WORKS + + +def decide(payload: object, + environ: Mapping[str, str] | None = None) -> dict | None: + """The hook's stdout document, or None to allow the call silently.""" + if not is_memhub_add_memory(payload) or not capture_is_active(payload, environ): + return None + return { + "hookSpecificOutput": { + "hookEventName": "PreToolUse", + "permissionDecision": "deny", + "permissionDecisionReason": DENY_REASON, + }, + "systemMessage": USER_LINE, + } + + +def main() -> int: + try: + raw = sys.stdin.read() + out = decide(json.loads(raw) if raw.strip() else {}) + except Exception: # noqa: BLE001 — fail open, see module docstring + return 0 + if out: + print(json.dumps(out)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/add_memory_gate_test.py b/tests/add_memory_gate_test.py new file mode 100644 index 00000000..c040c751 --- /dev/null +++ b/tests/add_memory_gate_test.py @@ -0,0 +1,172 @@ +#!/usr/bin/env python3 +"""add_memory is denied exactly when this plugin is already capturing the session. + +Run: python3 tests/add_memory_gate_test.py (stdlib only) +""" +from __future__ import annotations + +import json +import os +import re +import subprocess +import sys +import tempfile +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +PLUGIN = ROOT / "plugins" / "memhub" +SCRIPT = PLUGIN / "scripts" / "add_memory_gate.py" +sys.path.insert(0, str(PLUGIN / "scripts")) + +# Redirect HOME before importing, so the credential stores these tests write +# (and the ones capture_health reads at import time) are scratch, never real. +_TMP_HOME = tempfile.mkdtemp(prefix="add-memory-gate-test-") +os.environ["HOME"] = _TMP_HOME +os.environ["USERPROFILE"] = _TMP_HOME +HOST = "api.staging.memhub.xtrace.ai" +os.environ["MEMHUB_MCP_BASE_URL"] = f"https://{HOST}" +os.environ.pop("MEMHUB_TOKEN", None) +os.environ.pop("MEMHUB_TURN_FLUSH", None) + +import add_memory_gate as gate # noqa: E402 +import capture_health as ch # noqa: E402 +import pak # noqa: E402 + +# The names the tool actually arrived under in live sessions: a claude.ai +# MemHub connector, and this plugin's own server in both builds. +OBSERVED_NAMES = ( + "mcp__claude_ai_Xtrace__add_memory", + "mcp__plugin_memhub-staging_memhub__add_memory", + "mcp__plugin_memhub_memhub__add_memory", +) + + +def _payload(name: str = OBSERVED_NAMES[0], **overrides) -> dict: + payload = { + "session_id": "s-1", + "hook_event_name": "PreToolUse", + "transcript_path": "/Users/x/.claude/projects/p/s-1.jsonl", + "tool_name": name, + "tool_input": {"user_message": "q", "assistant_message": "a"}, + } + payload.update(overrides) + return payload + + +def _with_key() -> None: + pak.save(ch._mcp_url_for(HOST), {"secret": "mhk_x", "label": "test"}) + + +def _clear_credentials() -> None: + pak.forget(ch._mcp_url_for(HOST)) + token = ch.CACHE_DIR / f"tokens-{HOST}.json" + if token.exists(): + token.unlink() + + +def _denies(out: dict | None) -> bool: + return bool(out) and out["hookSpecificOutput"]["permissionDecision"] == "deny" + + +def test_denies_when_capture_is_live(): + _clear_credentials() + _with_key() + out = gate.decide(_payload()) + assert _denies(out), out + assert out["hookSpecificOutput"]["hookEventName"] == "PreToolUse" + assert "save_artifact" in out["hookSpecificOutput"]["permissionDecisionReason"] + assert "save_artifact" in out["systemMessage"] + print("PASS test_denies_when_capture_is_live") + + +def test_every_observed_server_name_is_gated_and_nothing_else(): + _clear_credentials() + _with_key() + for name in OBSERVED_NAMES: + assert _denies(gate.decide(_payload(name))), name + # A tool that merely shares the prefix, a built-in, and another server's + # add_memory with a different signature are not MemHub's add_memory. + for name in ("mcp__notes__add_memory_note", "Bash", "mcp__memhub__save_artifact"): + assert gate.decide(_payload(name)) is None, name + assert gate.decide(_payload(tool_input={"text": "x"})) is None + assert gate.decide(_payload(tool_input="user_message")) is None + print("PASS test_every_observed_server_name_is_gated_and_nothing_else") + + +def test_allows_when_capture_is_switched_off(): + _clear_credentials() + _with_key() + for value in ("0", "off", "FALSE", " false "): + assert gate.decide(_payload(), {"MEMHUB_TURN_FLUSH": value}) is None, value + # Any other value leaves capture on. + assert _denies(gate.decide(_payload(), {"MEMHUB_TURN_FLUSH": "1"})) + print("PASS test_allows_when_capture_is_switched_off") + + +def test_allows_without_a_working_credential(): + _clear_credentials() + assert gate.decide(_payload()) is None # never logged in + # A lapsed key with no OAuth fallback captures nothing either. + pak.save(ch._mcp_url_for(HOST), {"secret": "mhk_x", "label": "test", + "expires_at": "2000-01-01T00:00:00Z"}) + assert gate.decide(_payload()) is None + # An OAuth token that can renew is a working credential. + _clear_credentials() + ch.CACHE_DIR.mkdir(parents=True, exist_ok=True) + (ch.CACHE_DIR / f"tokens-{HOST}.json").write_text(json.dumps({ + "access_token": "opaque", "refresh_token": "r"}), encoding="utf-8") + assert _denies(gate.decide(_payload())) + print("PASS test_allows_without_a_working_credential") + + +def test_allows_without_a_transcript(): + _clear_credentials() + _with_key() + for transcript in (None, "", " "): + payload = _payload(transcript_path=transcript) + assert gate.decide(payload) is None, transcript + print("PASS test_allows_without_a_transcript") + + +def test_registered_for_every_observed_name_behind_the_guard(): + doc = json.loads((PLUGIN / "hooks" / "claude-hooks.json") + .read_text(encoding="utf-8")) + entries = [(group["matcher"], hook["command"]) + for group in doc["hooks"]["PreToolUse"] + for hook in group["hooks"] + if "add_memory_gate.py" in hook["command"]] + assert len(entries) == 1, entries + matcher, command = entries[0] + for name in OBSERVED_NAMES: + assert re.search(matcher, name), name + for name in ("mcp__notes__add_memory_note", "Bash", "add_memory"): + assert not re.search(matcher, name), name + # Cursor imports these hooks too; the guard must run first. + assert command.index("claude_hook_guard.py") < command.index("add_memory_gate.py") + assert "ignore PreToolUse" in command + print("PASS test_registered_for_every_observed_name_behind_the_guard") + + +def _run(stdin: str, **env) -> subprocess.CompletedProcess: + return subprocess.run( + [sys.executable, str(SCRIPT)], input=stdin, capture_output=True, + text=True, timeout=30, env=dict(os.environ, **env)) + + +def test_script_denies_and_fails_open(): + _clear_credentials() + _with_key() + done = _run(json.dumps(_payload())) + assert done.returncode == 0, done.stderr + assert _denies(json.loads(done.stdout)), done.stdout + for garbage in ("{{{", "", "[]", json.dumps({"tool_name": 3})): + done = _run(garbage) + assert done.returncode == 0 and done.stdout == "", (garbage, done) + print("PASS test_script_denies_and_fails_open") + + +if __name__ == "__main__": + for name, fn in sorted(globals().items()): + if name.startswith("test_") and callable(fn): + fn() + print("ALL PASS") diff --git a/tests/claude_hook_guard_test.py b/tests/claude_hook_guard_test.py index 58ec58d9..83e92349 100644 --- a/tests/claude_hook_guard_test.py +++ b/tests/claude_hook_guard_test.py @@ -79,7 +79,8 @@ def test_every_claude_handler_is_guarded_and_only_boundaries_capture(): for group in groups: for handler in group["hooks"]: commands.append((event, handler["command"])) - assert len(commands) == 21 # + UserPromptSubmit (brain_brief.py prompt, + assert len(commands) == 22 # + PreToolUse (add_memory_gate.py), + # + UserPromptSubmit (brain_brief.py prompt, # and rulebook_hook.py prompt — the lane that # arms a prompt-armed obligation), # + Stop (harness_stop.py, a no-op unless From a2cdee410ef9362a721f522476c018efa239bb79 Mon Sep 17 00:00:00 2001 From: Sebastian Gonzalez Date: Mon, 21 Sep 2026 14:01:38 -0700 Subject: [PATCH 2/3] chore(release): memhub v0.77.0 Co-Authored-By: Claude Opus 5 (1M context) --- plugins/memhub-staging/.claude-plugin/plugin.json | 2 +- plugins/memhub-staging/.codex-plugin/plugin.json | 2 +- plugins/memhub-staging/.mcp.json | 2 +- plugins/memhub/.claude-plugin/plugin.json | 2 +- plugins/memhub/.codex-plugin/plugin.json | 2 +- plugins/memhub/.cursor-plugin/plugin.json | 2 +- plugins/memhub/.mcp.json | 2 +- plugins/memhub/mcp.json | 2 +- plugins/memhub/plugin.json | 2 +- 9 files changed, 9 insertions(+), 9 deletions(-) diff --git a/plugins/memhub-staging/.claude-plugin/plugin.json b/plugins/memhub-staging/.claude-plugin/plugin.json index 2d542da4..c11c440d 100644 --- a/plugins/memhub-staging/.claude-plugin/plugin.json +++ b/plugins/memhub-staging/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "memhub-staging", "description": "STAGING build of the MemHub plugin \u2014 identical behavior to `memhub` but pointed at the staging backend (api.staging.memhub.xtrace.ai) for MemHub developers iterating on the service. Skills, hooks, and scripts are shared with `memhub` (symlinked), so the two never drift; only the backend URL and OAuth client differ. Install this instead of `memhub` \u2014 never both (both register an MCP server named `memhub`). Spec workflows: spec-work guides implementation, spec-check freshly reviews changes, and spec-maintain offers local/cloud bootstrap and reviewable Git or Brain proposals through the shared spec entry point.", - "version": "0.76.1", + "version": "0.77.0", "license": "Apache-2.0", "hooks": "./hooks/claude-hooks.json", "author": { diff --git a/plugins/memhub-staging/.codex-plugin/plugin.json b/plugins/memhub-staging/.codex-plugin/plugin.json index 8ee01c1c..5b4f47a7 100644 --- a/plugins/memhub-staging/.codex-plugin/plugin.json +++ b/plugins/memhub-staging/.codex-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "memhub-staging", "description": "MemHub staging for Codex: team memory and Rulebook hooks. Use the setup skill to verify the active hook route and credential. Repository shell commands in projectless tasks must include an explicit cd prefix when the host omits workdir. Spec workflows: spec-work guides implementation, spec-check freshly reviews changes, and spec-maintain offers local/cloud bootstrap and reviewable Git or Brain proposals through the shared spec entry point.", - "version": "0.76.1", + "version": "0.77.0", "license": "Apache-2.0", "author": { "name": "XTrace", diff --git a/plugins/memhub-staging/.mcp.json b/plugins/memhub-staging/.mcp.json index 06173bbe..e83f8270 100644 --- a/plugins/memhub-staging/.mcp.json +++ b/plugins/memhub-staging/.mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "memhub": { "type": "http", - "url": "https://api.staging.memhub.xtrace.ai/mcp-server/mcp?memhub_plugin_version=0.76.1", + "url": "https://api.staging.memhub.xtrace.ai/mcp-server/mcp?memhub_plugin_version=0.77.0", "auth": { "CLIENT_ID": "mYTjrWldX9ZFGtDES3hQDEHSis9gIjiq" }, diff --git a/plugins/memhub/.claude-plugin/plugin.json b/plugins/memhub/.claude-plugin/plugin.json index de310652..3e13b59d 100644 --- a/plugins/memhub/.claude-plugin/plugin.json +++ b/plugins/memhub/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "memhub", "description": "Auto-capture Claude Code sessions into MemHub team memory. A Stop hook ships each turn as it happens \u2014 sending only the transcript bytes written since the last successful flush, so a session is captured even if it never reaches a clean exit \u2014 alongside flushes on commit/PR events (PostToolUse hooks) and a SessionEnd backstop. The memhub MCP runs tool-aware (agentic) extraction of facts, episodes, and artifacts with watermark-based delta dedup, batching extraction so per-turn capture doesn't fragment episodes. A PreToolUse hook surfaces situated team directives (lessons/procedures) before each Edit/Write/Bash by firing the memhub recall_directives symbol-tripwire on the in-flight call and injecting any hits as context (a client-side precision gate drops directives whose triggers don't concretely match the call, e.g. an over-broad repo-name trigger). Includes skills for: plugin install/health (setup), plugin authentication with its own hook credential (login), repo brain creation and seeding (onboard), terminal artifact uploads (save-artifact), on-demand session import (import-session), team-memory recall (search-memory), teammate session handoff (handoff-session), git-authored spec development with ownership frontmatter (spec), Rulebook rule authoring (create-rule), rules proposed from CLAUDE.md and past sessions in one run (start-rulebook), and PR babysitting (pr-babysit) \u2014 a hook arms a self-paced loop after `gh pr create` that fixes review-bot findings and saves the fixing process to the repo's agent brain. Rules live in RULEBOOKS \u2014 containers with their own membership, so a rule reaches exactly the people its book binds, and one person can be bound by several (their org's book plus their team's): the authoring skills resolve which book a rule lands in and offer to create one when there is none, the hooks cache every book that binds you, and when two books fire on one call the wider book's rule is the one the per-call cap keeps. A PostToolUse hook reminds authors when an edited file belongs to a git-authored spec. Spec ownership frontmatter replaces the retired artifact map; the backend maintains read-only mirrors and opens bootstrap or audit PRs. Sessions are named with the title their own host shows \u2014 Claude Code's generated title, or on Codex the thread_name Codex generated (from the rollout, else ~/.codex/session_index.jsonl), passed through verbatim so MemHub's sessions list matches the editor's; a session its host never named falls back to a title derived from its first prompt \u2014 and route into the repo's own agent brain via a per-user room cache (~/.config/memhub-plugin/rooms.json, never written into the repo) \u2014 resolved once by /memhub:onboard and read by every writer, including the two automatic capture paths, so team memory lands in the repo's room instead of personal memory. Session \u2194 PR linking: after a call that addresses GitHub (`gh pr \u2026`, a `curl`/`gh api` REST request, or a GitHub MCP tool) whose output names exactly one pull request, a PostToolUse hook asks the backend one question and injects one instruction \u2014 a session that ran `gh pr create` \u2014 or the GitHub MCP create tool \u2014 links itself unconditionally (opening it is work the session did, and a PR has many sessions), while every other GitHub call, a hand-rolled `curl` POST included, leaves the authorship judgment to the agent, and an org with no GitHub integration is told once that connecting it is what links sessions to shipped code. The hook is stateless apart from one short-lived negative answer (an enabled org with GitHub disconnected, cached 30 minutes per credential, and never inferred from a reply that merely says the feature is off) and degrades to silence on every failure. Two skills complete it: link-pr (link this or a named session by hand \u2014 also the answer for a PR opened by a script or CI helper, which is deliberately not detected) and find-contributing-sessions (scan this machine's local session history for the sessions that wrote a PR's code, rank the candidates by the evidence that matched, and link the ones you approve). Every rulebook fire is now disclosed to the user in a fixed shape on two channels: the hook's own terminal line leads with `\ud83d\udccf Rule fired: ` (`\u26d4\ufe0f` when a gate actually blocked the call) above the detail line it already showed, and the agent is told to echo that same byte-identical line at the top of its reply so the fire reaches the transcript everything downstream reads. Session start now says what rules ARE \u2014 standing instructions from teammates carrying CLAUDE.md's weight \u2014 before listing them. The agent records ONE work type for a pull request it links (feat, fix, chore, docs, perf, refactor, other), and only when the PR has none yet \u2014 the first label is permanent, and asking again would cost the link, not just the type. Spec workflows: spec-work guides implementation, spec-check freshly reviews changes, and spec-maintain offers local/cloud bootstrap and reviewable Git or Brain proposals through the shared spec entry point.", - "version": "0.76.1", + "version": "0.77.0", "license": "Apache-2.0", "hooks": "./hooks/claude-hooks.json", "author": { diff --git a/plugins/memhub/.codex-plugin/plugin.json b/plugins/memhub/.codex-plugin/plugin.json index 2568500f..e510fadc 100644 --- a/plugins/memhub/.codex-plugin/plugin.json +++ b/plugins/memhub/.codex-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "memhub", "description": "XTrace MemHub team memory in Codex: automatic session capture, situated directive recall, artifact sync, plus skills for setup, login, onboarding, artifact upload, session import, search, handoff, spec-driven development, Rulebook rule authoring (create-rule) and a team's first Rulebook — tested starter rules fitted to the repo, and rules proposed from CLAUDE.md + past sessions (start-rulebook), and PR babysitting. A rule lives in a RULEBOOK \u2014 a container with its own membership, so it reaches exactly the people that book binds; the authoring skills resolve which book it lands in and offer to create one when there is none. Run memhub:setup to verify hook routing and authentication. Current desktop builds support bundled hooks; older hosts need the user bridge. Bundled handlers defer to matching installed user bridge handlers, which require trust. Codex's own threads are not captured as your sessions: spawned subagents, guardian action-reviews and memory consolidation are recognised from the rollout's thread_source and skipped by name, while every other kind \u2014 including a product surface we have not seen, since Codex's ThreadSource type is open-ended \u2014 is captured, because refusing an unknown kind would silently and unrecoverably drop real work. Every rule fire is reported under the same namespaced session id capture uploads, so a fire is linked to the Codex session it fired in. Session \u2194 PR linking: after a call that addresses GitHub (`gh pr \u2026`, a `curl`/`gh api` REST request, or a GitHub MCP tool) whose output names exactly one pull request, a PostToolUse hook asks the backend one question and injects one instruction \u2014 a session that ran `gh pr create` \u2014 or the GitHub MCP create tool \u2014 links itself unconditionally (opening it is work the session did, and a PR has many sessions), while every other GitHub call, a hand-rolled `curl` POST included, leaves the authorship judgment to the agent, and an org with no GitHub integration is told once that connecting it is what links sessions to shipped code. The hook is stateless apart from one short-lived negative answer (an enabled org with GitHub disconnected, cached 30 minutes per credential, and never inferred from a reply that merely says the feature is off) and degrades to silence on every failure. Two skills complete it: link-pr (link this or a named session by hand \u2014 also the answer for a PR opened by a script or CI helper, which is deliberately not detected) and find-contributing-sessions (scan this machine's local session history for the sessions that wrote a PR's code, rank the candidates by the evidence that matched, and link the ones you approve). The agent records ONE work type for a pull request it links (feat, fix, chore, docs, perf, refactor, other), and only when the PR has none yet \u2014 the first label is permanent, and asking again would cost the link, not just the type. Spec workflows: spec-work guides implementation, spec-check freshly reviews changes, and spec-maintain offers local/cloud bootstrap and reviewable Git or Brain proposals through the shared spec entry point.", - "version": "0.76.1", + "version": "0.77.0", "license": "Apache-2.0", "author": { "name": "XTrace", diff --git a/plugins/memhub/.cursor-plugin/plugin.json b/plugins/memhub/.cursor-plugin/plugin.json index 1113dd20..20b5b39e 100644 --- a/plugins/memhub/.cursor-plugin/plugin.json +++ b/plugins/memhub/.cursor-plugin/plugin.json @@ -2,7 +2,7 @@ "name": "memhub", "displayName": "MemHub", "description": "XTrace MemHub team memory in Cursor: search_memory / save_artifact / import_conversation tools plus skills for setup, login, onboarding, artifact upload, session import, recall, handoff, spec-driven development, Rulebook rule authoring (create-rule) and a team's first Rulebook — tested starter rules fitted to the repo, and rules proposed from CLAUDE.md + past sessions (start-rulebook), and PR babysitting. A rule lives in a RULEBOOK \u2014 a container with its own membership, so it reaches exactly the people that book binds; the authoring skills resolve which book it lands in and offer to create one when there is none. Sessions are captured automatically. Session \u2194 PR linking ships as skills on Cursor: link-pr and find-contributing-sessions. Cursor's shell hooks are observational \u2014 `afterShellExecution` defines no reply the agent can see \u2014 so the automatic post-`gh pr create` link that Claude Code and Codex get is not available here. The agent records ONE work type for a pull request it links (feat, fix, chore, docs, perf, refactor, other), and only when the PR has none yet \u2014 the first label is permanent, and asking again would cost the link, not just the type. Spec workflows: spec-work guides implementation, spec-check freshly reviews changes, and spec-maintain offers local/cloud bootstrap and reviewable Git or Brain proposals through the shared spec entry point.", - "version": "0.76.1", + "version": "0.77.0", "license": "Apache-2.0", "author": { "name": "XTrace" diff --git a/plugins/memhub/.mcp.json b/plugins/memhub/.mcp.json index 6af3c90c..d8dd5a9e 100644 --- a/plugins/memhub/.mcp.json +++ b/plugins/memhub/.mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "memhub": { "type": "http", - "url": "https://api.memhub.xtrace.ai/mcp-server/mcp?memhub_plugin_version=0.76.1", + "url": "https://api.memhub.xtrace.ai/mcp-server/mcp?memhub_plugin_version=0.77.0", "auth": { "CLIENT_ID": "xHoQkYd7uNUfaNX6Tyv143e1Ev7u3XS0" }, diff --git a/plugins/memhub/mcp.json b/plugins/memhub/mcp.json index 7783fb79..809b0d06 100644 --- a/plugins/memhub/mcp.json +++ b/plugins/memhub/mcp.json @@ -3,7 +3,7 @@ "mcpServers": { "memhub": { "type": "streamable-http", - "url": "https://api.memhub.xtrace.ai/mcp-server/mcp?memhub_plugin_version=0.76.1" + "url": "https://api.memhub.xtrace.ai/mcp-server/mcp?memhub_plugin_version=0.77.0" } } } diff --git a/plugins/memhub/plugin.json b/plugins/memhub/plugin.json index 7096bb94..c8e6e0f9 100644 --- a/plugins/memhub/plugin.json +++ b/plugins/memhub/plugin.json @@ -2,7 +2,7 @@ "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", "name": "memhub", "description": "XTrace MemHub team memory: search_memory / save_artifact / import_conversation MCP tools plus skills for setup, login, onboarding, artifact upload, session import, team-memory recall, session handoff, spec-driven development, Rulebook rule authoring (create-rule) and a team's first Rulebook — tested starter rules fitted to the repo, and rules proposed from CLAUDE.md + past sessions (start-rulebook), and PR babysitting. A rule lives in a RULEBOOK \u2014 a container with its own membership, so it reaches exactly the people that book binds; the authoring skills resolve which book it lands in and offer to create one when there is none. Session auto-capture ships through the host-native plugin layer (hooks are outside the Agent Plugins spec). Session \u2194 PR linking: after a call that addresses GitHub (`gh pr \u2026`, a `curl`/`gh api` REST request, or a GitHub MCP tool) whose output names exactly one pull request, a PostToolUse hook asks the backend one question and injects one instruction \u2014 a session that ran `gh pr create` \u2014 or the GitHub MCP create tool \u2014 links itself unconditionally (opening it is work the session did, and a PR has many sessions), while every other GitHub call, a hand-rolled `curl` POST included, leaves the authorship judgment to the agent, and an org with no GitHub integration is told once that connecting it is what links sessions to shipped code. The hook is stateless apart from one short-lived negative answer (an enabled org with GitHub disconnected, cached 30 minutes per credential, and never inferred from a reply that merely says the feature is off) and degrades to silence on every failure. Two skills complete it: link-pr (link this or a named session by hand \u2014 also the answer for a PR opened by a script or CI helper, which is deliberately not detected) and find-contributing-sessions (scan this machine's local session history for the sessions that wrote a PR's code, rank the candidates by the evidence that matched, and link the ones you approve). The agent records ONE work type for a pull request it links (feat, fix, chore, docs, perf, refactor, other), and only when the PR has none yet \u2014 the first label is permanent, and asking again would cost the link, not just the type. Spec workflows: spec-work guides implementation, spec-check freshly reviews changes, and spec-maintain offers local/cloud bootstrap and reviewable Git or Brain proposals through the shared spec entry point.", - "version": "0.76.1", + "version": "0.77.0", "license": "Apache-2.0", "author": { "name": "XTrace", From 5701888e61987849ffc3d86ff5f27252ed4ab9bd Mon Sep 17 00:00:00 2001 From: Sebastian Gonzalez Date: Mon, 21 Sep 2026 15:00:45 -0700 Subject: [PATCH 3/3] fix(capture): allow add_memory in a harness authoring child #268 made MEMHUB_HARNESS_CHILD switch transcript capture off: flush_turn and flush_session both return early for the harness's forked session. The gate must read the same flag through the same helper (is_harness_child), or it denies add_memory in a session nothing is capturing. Co-Authored-By: Claude Opus 5 (1M context) --- plugins/memhub/scripts/add_memory_gate.py | 7 ++++++- tests/add_memory_gate_test.py | 14 ++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/plugins/memhub/scripts/add_memory_gate.py b/plugins/memhub/scripts/add_memory_gate.py index 0e75ee39..708fbe54 100644 --- a/plugins/memhub/scripts/add_memory_gate.py +++ b/plugins/memhub/scripts/add_memory_gate.py @@ -22,7 +22,9 @@ ``user_message`` argument, so it covers this plugin's own server and a claude.ai MemHub connector alike (the blind agents used the connector) without catching an unrelated server's ``add_memory``; -* per-turn capture is not switched off (``MEMHUB_TURN_FLUSH=0``); +* per-turn capture is not switched off (``MEMHUB_TURN_FLUSH=0``), and this is + not a harness authoring child (``MEMHUB_HARNESS_CHILD``), which the flush + scripts never capture; * the payload names a transcript, which is what the flush reads; * the plugin holds a credential capture can authenticate with, judged by the same network-free check the capture-health banner uses — so this and that @@ -85,6 +87,9 @@ def capture_is_active(payload: dict, env = os.environ if environ is None else environ if env.get("MEMHUB_TURN_FLUSH", "").strip().lower() in _CAPTURE_OFF: return False + import transcript_filter # noqa: PLC0415 — stdlib-only, beside this file + if transcript_filter.is_harness_child(env): + return False # both flush scripts skip the harness's forked copy transcript = payload.get("transcript_path") if not isinstance(transcript, str) or not transcript.strip(): return False diff --git a/tests/add_memory_gate_test.py b/tests/add_memory_gate_test.py index c040c751..bbb1da0c 100644 --- a/tests/add_memory_gate_test.py +++ b/tests/add_memory_gate_test.py @@ -103,6 +103,20 @@ def test_allows_when_capture_is_switched_off(): print("PASS test_allows_when_capture_is_switched_off") +def test_allows_in_a_harness_child(): + # The harness's forked authoring session is never captured (flush_turn and + # flush_session both return early on it), so add_memory is not a copy there. + _clear_credentials() + _with_key() + for value in ("1", "on", "TRUE", "yes"): + env = {"MEMHUB_HARNESS_CHILD": value} + assert gate.decide(_payload(), env) is None, value + for value in ("", "0", "off"): + env = {"MEMHUB_HARNESS_CHILD": value} + assert _denies(gate.decide(_payload(), env)), value + print("PASS test_allows_in_a_harness_child") + + def test_allows_without_a_working_credential(): _clear_credentials() assert gate.decide(_payload()) is None # never logged in