Skip to content
87 changes: 73 additions & 14 deletions .agent/harness/hooks/claude_code_post_tool.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@
# UP 3 = .agent/
HERE = os.path.dirname(os.path.abspath(__file__))
AGENT_ROOT = os.path.abspath(os.path.join(HERE, "..", ".."))
PROJECT_ROOT = os.path.dirname(AGENT_ROOT)

sys.path.insert(0, os.path.join(AGENT_ROOT, "harness"))
sys.path.insert(0, os.path.join(AGENT_ROOT, "tools"))
Expand All @@ -42,6 +43,38 @@
from hooks.on_failure import on_failure # noqa: E402


def _normalize_path(value):
"""Return a portable path label for anything persisted to episodic memory.

Episodic entries are meant to be shared, diffed, and eventually exported
via the data flywheel — a raw absolute path bakes the operator's
username into every entry that touches a file. Normalize to a
project-relative or home-relative form instead.
"""
if not isinstance(value, str) or not value:
return "?"
expanded = os.path.abspath(os.path.expanduser(value))
try:
rel = os.path.relpath(expanded, PROJECT_ROOT)
except ValueError:
rel = None
if rel is not None and not rel.startswith(".."):
return rel
home = os.path.expanduser("~")
if home and expanded.startswith(home + os.sep):
return "~" + expanded[len(home):]
return "<external>"


def _input_path(tool_input):
if not isinstance(tool_input, dict):
return None
for key in ("file_path", "path", "new_path", "notebook_path"):
if isinstance(tool_input.get(key), str) and tool_input[key]:
return tool_input[key]
return None


# ---------------------------------------------------------------------------
# Importance scoring
# ---------------------------------------------------------------------------
Expand Down Expand Up @@ -386,15 +419,15 @@ def _action_label(tool_name: str, tool_input: dict) -> str:
or tool_input.get("path")
or tool_input.get("new_path")
or "?")
return f"edit: {path}"
return f"edit: {_normalize_path(path)}"

if tool_name == "Write":
path = tool_input.get("file_path") or tool_input.get("path") or "?"
return f"write: {path}"
return f"write: {_normalize_path(path)}"

if tool_name == "Read":
path = tool_input.get("file_path") or tool_input.get("path") or "?"
return f"read: {path}"
return f"read: {_normalize_path(path)}"

if tool_name == "TodoWrite":
todos = tool_input.get("todos", [])
Expand Down Expand Up @@ -435,7 +468,6 @@ def _reflection(tool_name: str, tool_input: dict,
4. Keep under ~200 chars so detail field carries the rest.
"""
parts = []
inp_str = json.dumps(tool_input)

# --- Bash ---
if tool_name == "Bash":
Expand All @@ -462,13 +494,12 @@ def _reflection(tool_name: str, tool_input: dict,

# --- Edit ---
elif tool_name in ("Edit", "MultiEdit"):
path = tool_input.get("file_path") or tool_input.get("path") or "?"
old = (tool_input.get("old_string") or "")[:50]
new = (tool_input.get("new_string") or "")[:50]
path = _normalize_path(tool_input.get("file_path") or tool_input.get("path") or "?")
old = tool_input.get("old_string") or ""
new = tool_input.get("new_string") or ""
if old and new:
parts.append(
f"Edited {path}: replaced {repr(old[:30])} "
f"with {repr(new[:30])}"
f"Edited {path}: {len(old)} chars -> {len(new)} chars"
)
else:
parts.append(f"Edited {path}")
Expand All @@ -477,7 +508,7 @@ def _reflection(tool_name: str, tool_input: dict,

# --- Write ---
elif tool_name == "Write":
path = tool_input.get("file_path") or tool_input.get("path") or "?"
path = _normalize_path(tool_input.get("file_path") or tool_input.get("path") or "?")
content = tool_input.get("content") or ""
lines = content.count("\n") + 1 if content else 0
parts.append(f"Wrote {path} ({lines} lines)")
Expand Down Expand Up @@ -506,8 +537,12 @@ def _reflection(tool_name: str, tool_input: dict,
else:
status = "successfully" if success else "with failure"
parts.append(f"Tool {tool_name} completed {status}")
if inp_str and len(inp_str) < 80:
parts.append(inp_str)
# Input values can carry content or paths; persist key names only.
path = _input_path(tool_input)
if path:
parts.append(f"on {_normalize_path(path)}")
if isinstance(tool_input, dict) and tool_input:
parts.append(f"keys: {', '.join(sorted(map(str, tool_input)))}")

return ". ".join(parts) if parts else f"Tool {tool_name} ran"

Expand All @@ -521,9 +556,12 @@ def _detail(tool_name: str, tool_input: dict,
"""
Stored in `detail`. More verbose than reflection. Truncated to 500 chars
by log_execution anyway.

Persists normalized metadata only — never a raw dump of tool_input.
A raw dump embeds file contents, edit diffs, and absolute paths
verbatim into a log meant to be shared, diffed, and exported.
"""
output = _extract_output(tool_response)
inp_str = json.dumps(tool_input, separators=(",", ":"))[:300]

if tool_name == "Bash":
cmd = tool_input.get("command", "")[:120]
Expand All @@ -533,7 +571,28 @@ def _detail(tool_name: str, tool_input: dict,
out_snip = output[:200] if output else ""
return f"cmd={cmd!r}" + (f" | out={out_snip}" if out_snip else "")

return inp_str + (f" | {output[:150]}" if output else "")
path = _input_path(tool_input)
meta = {"tool": tool_name}
if path:
meta["path"] = _normalize_path(path)
content = tool_input.get("content")
if isinstance(content, str):
meta["content_chars"] = len(content)
old = tool_input.get("old_string")
new = tool_input.get("new_string")
if isinstance(old, str) or isinstance(new, str):
meta["old_string_chars"] = len(old or "")
meta["new_string_chars"] = len(new or "")

# Tool output and errors (a Read of a secrets file, Grep matches, an
# error echoing a path) are content too: persist their size only.
if output:
meta["output_chars"] = len(output)
if not success:
err = _extract_error(tool_response)
Comment thread
macroscopeapp[bot] marked this conversation as resolved.
if err:
meta["error_chars"] = len(err)
return json.dumps(meta, separators=(",", ":"))


# ---------------------------------------------------------------------------
Expand Down
1 change: 1 addition & 0 deletions .agent/harness/llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,7 @@ def _call_minimax(system, user, *, temperature, max_tokens, model):
r = c.chat.completions.create(
model=model,
temperature=temperature,
max_tokens=max_tokens,
messages=[{"role": "system", "content": system},
{"role": "user", "content": user}],
)
Expand Down
40 changes: 37 additions & 3 deletions .agent/tools/data_layer_export.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,24 @@
VALID_WINDOWS = {"7d", "30d", "90d", "all"}
VALID_BUCKETS = {"hour", "day", "week", "month"}

# Finite value sets the loop supervisor (harness_manager/loops/runner.py and
# process.py) actually writes to runtime/loops/events.jsonl. normalize_loop_event must
# redact anything outside these sets rather than copy arbitrary loop-event
# content into the exported dashboard/analytics surface.
VALID_LOOP_EVENTS = {
"created", "awaiting_approval", "worktree_created", "paused",
"interrupted", "maker_finished", "verifier_finished", "checker_finished",
"exhausted", "completed", "cancelled",
# written by the agentic-stack-desktop supervisor
"phase_started",
}
VALID_LOOP_STATUSES = {
"created", "awaiting_approval", "paused", "exhausted", "interrupted",
"completed", "cancelled", "audit_failed", "failed", "rejected",
"failed_to_start", "timed_out", "running",
}
VALID_LOOP_DECISIONS = {"APPROVE", "ESCALATE", "MALFORMED"}


def _e(*codes: int) -> str:
return f"\x1b[{';'.join(map(str, codes))}m"
Expand Down Expand Up @@ -393,14 +411,30 @@ def normalize_agent_event(entry: dict[str, Any], idx: int, args: argparse.Namesp
return base


def _allowed_or_unknown(value: Any, allowed: set[str]) -> str:
text = str(value) if value is not None else ""
return text if text in allowed else "unknown"


def normalize_loop_event(entry: dict[str, Any]) -> dict[str, Any]:
"""Map the privacy-whitelisted loop event shape into data-layer fields."""
"""Map the privacy-whitelisted loop event shape into data-layer fields.

entry's `event`/`status`/`decision` values come from a supervisor-controlled
finite set (VALID_LOOP_EVENTS/STATUSES/DECISIONS); anything else is redacted
to "unknown" rather than copied through, so a malformed or unexpected
events.jsonl row can't smuggle arbitrary text into the exported surface.
"""
status_or_decision = entry.get("status") or entry.get("decision")
return {
"timestamp": entry.get("timestamp") or now_iso(),
"skill": "agentic-loop",
"action": str(entry.get("event") or "loop_event"),
"action": _allowed_or_unknown(entry.get("event"), VALID_LOOP_EVENTS),
"workflow": str(entry.get("loop") or "loop"),
"result": str(entry.get("status") or entry.get("decision") or "observed"),
"result": (
_allowed_or_unknown(status_or_decision, VALID_LOOP_STATUSES | VALID_LOOP_DECISIONS)
if status_or_decision is not None
else "observed"
),
"harness": "agentic-loop",
"source": {"run_id": entry.get("run_id")},
"privacy_level": "local_only",
Expand Down
19 changes: 16 additions & 3 deletions .agent/tools/test_learn_episodic_mirror.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,12 +24,18 @@ def _load_learn(base_dir):
"""Load .agent/tools/learn.py with BASE/CANDIDATES pointed at base_dir.

Sibling modules (text.word_set, cluster.pattern_id) are stubbed so the
test needs no part of the harness beyond learn.py itself.
test needs no part of the harness beyond learn.py itself. The stubs are
process-wide (sys.modules), so any previous entry for "text"/"cluster"
is saved and restored (or removed, if there was none) once exec_module
finishes -- otherwise a later test or import in the same process would
silently pick up these throwaway stand-ins instead of the real modules.
"""
previous_modules = {}
for name, attrs in [
("text", {"word_set": lambda *a, **k: set()}),
("cluster", {"pattern_id": lambda claim, cond: "testcid" + str(abs(hash((claim, tuple(cond)))))[:6]}),
]:
previous_modules[name] = sys.modules.pop(name, None)
m = types.ModuleType(name)
for k, v in attrs.items():
setattr(m, k, v)
Expand All @@ -38,7 +44,14 @@ def _load_learn(base_dir):
module_path = Path(__file__).with_name("learn.py")
spec = importlib.util.spec_from_file_location("learn_under_test", module_path)
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
try:
spec.loader.exec_module(mod)
finally:
for name, previous in previous_modules.items():
if previous is None:
sys.modules.pop(name, None)
else:
sys.modules[name] = previous
mod.BASE = base_dir
mod.CANDIDATES = os.path.join(base_dir, "memory", "candidates")
os.makedirs(mod.CANDIDATES, exist_ok=True)
Expand Down Expand Up @@ -68,7 +81,7 @@ def test_stage_writes_one_episodic_mirror(self):
def test_evidence_id_resolves_to_the_mirror(self):
mod = _load_learn(self.tmp)
cid, path = mod.stage("Serialize timestamps in UTC", ["timestamps", "utc"])
candidate = json.loads(Path(path).read_text())
candidate = json.loads(Path(path).read_text(encoding="utf-8"))
evidence_ts = candidate["evidence_ids"][0]
matching = [e for e in _episodic(self.tmp) if e["timestamp"] == evidence_ts]
self.assertEqual(len(matching), 1)
Expand Down
23 changes: 23 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,29 @@ All notable changes to this project.
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
and the project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).

## [Unreleased]

### Fixed
- **Claude Code hook leaked home paths and file content** (#66, #67). Episodic
entries now store project- or `~`-relative paths (`<external>` otherwise),
input key names, and character counts in place of raw edit/write content,
raw `tool_input` dumps, and raw tool output or error text.
- **Loop event export copied arbitrary text** (#64). `data_layer_export.py`
redacts loop `event`/`status`/`decision` values outside the set the loop
supervisor and agentic-stack-desktop write (including `exhausted`,
`failed_to_start`, `timed_out`, `phase_started`, `running`) to `unknown`.
- **MiniMax ignored `max_tokens`** (#65). The OpenAI-wire call now forwards it
as `max_tokens` (`max_completion_tokens` does not exist in the minimum
supported `openai==1.40.0`).
- **`test_learn_episodic_mirror.py` leaked module stubs** (#65). Stubbed
`sys.modules` entries are restored after import; candidate JSON is read as
UTF-8.

### Docs
- README: seed-skill list and count (14), repo layout (`loops/`, Copilot/Pi
hooks, `tests/`), Windsurf rule path, and install notes moved under
"Once installed".

## [0.19.1] — 2026-08-07

Patch release. Four correctness fixes in memory retrieval, project upgrade, and
Expand Down
Loading