diff --git a/.agent/harness/hooks/claude_code_post_tool.py b/.agent/harness/hooks/claude_code_post_tool.py index ba0633d..97ee91f 100644 --- a/.agent/harness/hooks/claude_code_post_tool.py +++ b/.agent/harness/hooks/claude_code_post_tool.py @@ -34,6 +34,7 @@ # UP 3 = .agent/ HERE = os.path.dirname(os.path.abspath(__file__)) AGENT_ROOT = os.path.abspath(os.path.join(HERE, "..", "..")) +PROJECT_ROOT = os.path.dirname(AGENT_ROOT) sys.path.insert(0, os.path.join(AGENT_ROOT, "harness")) sys.path.insert(0, os.path.join(AGENT_ROOT, "tools")) @@ -42,6 +43,38 @@ from hooks.on_failure import on_failure # noqa: E402 +def _normalize_path(value): + """Return a portable path label for anything persisted to episodic memory. + + Episodic entries are meant to be shared, diffed, and eventually exported + via the data flywheel — a raw absolute path bakes the operator's + username into every entry that touches a file. Normalize to a + project-relative or home-relative form instead. + """ + if not isinstance(value, str) or not value: + return "?" + expanded = os.path.abspath(os.path.expanduser(value)) + try: + rel = os.path.relpath(expanded, PROJECT_ROOT) + except ValueError: + rel = None + if rel is not None and not rel.startswith(".."): + return rel + home = os.path.expanduser("~") + if home and expanded.startswith(home + os.sep): + return "~" + expanded[len(home):] + return "" + + +def _input_path(tool_input): + if not isinstance(tool_input, dict): + return None + for key in ("file_path", "path", "new_path", "notebook_path"): + if isinstance(tool_input.get(key), str) and tool_input[key]: + return tool_input[key] + return None + + # --------------------------------------------------------------------------- # Importance scoring # --------------------------------------------------------------------------- @@ -386,15 +419,15 @@ def _action_label(tool_name: str, tool_input: dict) -> str: or tool_input.get("path") or tool_input.get("new_path") or "?") - return f"edit: {path}" + return f"edit: {_normalize_path(path)}" if tool_name == "Write": path = tool_input.get("file_path") or tool_input.get("path") or "?" - return f"write: {path}" + return f"write: {_normalize_path(path)}" if tool_name == "Read": path = tool_input.get("file_path") or tool_input.get("path") or "?" - return f"read: {path}" + return f"read: {_normalize_path(path)}" if tool_name == "TodoWrite": todos = tool_input.get("todos", []) @@ -435,7 +468,6 @@ def _reflection(tool_name: str, tool_input: dict, 4. Keep under ~200 chars so detail field carries the rest. """ parts = [] - inp_str = json.dumps(tool_input) # --- Bash --- if tool_name == "Bash": @@ -462,13 +494,12 @@ def _reflection(tool_name: str, tool_input: dict, # --- Edit --- elif tool_name in ("Edit", "MultiEdit"): - path = tool_input.get("file_path") or tool_input.get("path") or "?" - old = (tool_input.get("old_string") or "")[:50] - new = (tool_input.get("new_string") or "")[:50] + path = _normalize_path(tool_input.get("file_path") or tool_input.get("path") or "?") + old = tool_input.get("old_string") or "" + new = tool_input.get("new_string") or "" if old and new: parts.append( - f"Edited {path}: replaced {repr(old[:30])} " - f"with {repr(new[:30])}" + f"Edited {path}: {len(old)} chars -> {len(new)} chars" ) else: parts.append(f"Edited {path}") @@ -477,7 +508,7 @@ def _reflection(tool_name: str, tool_input: dict, # --- Write --- elif tool_name == "Write": - path = tool_input.get("file_path") or tool_input.get("path") or "?" + path = _normalize_path(tool_input.get("file_path") or tool_input.get("path") or "?") content = tool_input.get("content") or "" lines = content.count("\n") + 1 if content else 0 parts.append(f"Wrote {path} ({lines} lines)") @@ -506,8 +537,12 @@ def _reflection(tool_name: str, tool_input: dict, else: status = "successfully" if success else "with failure" parts.append(f"Tool {tool_name} completed {status}") - if inp_str and len(inp_str) < 80: - parts.append(inp_str) + # Input values can carry content or paths; persist key names only. + path = _input_path(tool_input) + if path: + parts.append(f"on {_normalize_path(path)}") + if isinstance(tool_input, dict) and tool_input: + parts.append(f"keys: {', '.join(sorted(map(str, tool_input)))}") return ". ".join(parts) if parts else f"Tool {tool_name} ran" @@ -521,9 +556,12 @@ def _detail(tool_name: str, tool_input: dict, """ Stored in `detail`. More verbose than reflection. Truncated to 500 chars by log_execution anyway. + + Persists normalized metadata only — never a raw dump of tool_input. + A raw dump embeds file contents, edit diffs, and absolute paths + verbatim into a log meant to be shared, diffed, and exported. """ output = _extract_output(tool_response) - inp_str = json.dumps(tool_input, separators=(",", ":"))[:300] if tool_name == "Bash": cmd = tool_input.get("command", "")[:120] @@ -533,7 +571,28 @@ def _detail(tool_name: str, tool_input: dict, out_snip = output[:200] if output else "" return f"cmd={cmd!r}" + (f" | out={out_snip}" if out_snip else "") - return inp_str + (f" | {output[:150]}" if output else "") + path = _input_path(tool_input) + meta = {"tool": tool_name} + if path: + meta["path"] = _normalize_path(path) + content = tool_input.get("content") + if isinstance(content, str): + meta["content_chars"] = len(content) + old = tool_input.get("old_string") + new = tool_input.get("new_string") + if isinstance(old, str) or isinstance(new, str): + meta["old_string_chars"] = len(old or "") + meta["new_string_chars"] = len(new or "") + + # Tool output and errors (a Read of a secrets file, Grep matches, an + # error echoing a path) are content too: persist their size only. + if output: + meta["output_chars"] = len(output) + if not success: + err = _extract_error(tool_response) + if err: + meta["error_chars"] = len(err) + return json.dumps(meta, separators=(",", ":")) # --------------------------------------------------------------------------- diff --git a/.agent/harness/llm.py b/.agent/harness/llm.py index d2aecde..c592dd1 100644 --- a/.agent/harness/llm.py +++ b/.agent/harness/llm.py @@ -73,6 +73,7 @@ def _call_minimax(system, user, *, temperature, max_tokens, model): r = c.chat.completions.create( model=model, temperature=temperature, + max_tokens=max_tokens, messages=[{"role": "system", "content": system}, {"role": "user", "content": user}], ) diff --git a/.agent/tools/data_layer_export.py b/.agent/tools/data_layer_export.py index 1002ada..9741b23 100755 --- a/.agent/tools/data_layer_export.py +++ b/.agent/tools/data_layer_export.py @@ -24,6 +24,24 @@ VALID_WINDOWS = {"7d", "30d", "90d", "all"} VALID_BUCKETS = {"hour", "day", "week", "month"} +# Finite value sets the loop supervisor (harness_manager/loops/runner.py and +# process.py) actually writes to runtime/loops/events.jsonl. normalize_loop_event must +# redact anything outside these sets rather than copy arbitrary loop-event +# content into the exported dashboard/analytics surface. +VALID_LOOP_EVENTS = { + "created", "awaiting_approval", "worktree_created", "paused", + "interrupted", "maker_finished", "verifier_finished", "checker_finished", + "exhausted", "completed", "cancelled", + # written by the agentic-stack-desktop supervisor + "phase_started", +} +VALID_LOOP_STATUSES = { + "created", "awaiting_approval", "paused", "exhausted", "interrupted", + "completed", "cancelled", "audit_failed", "failed", "rejected", + "failed_to_start", "timed_out", "running", +} +VALID_LOOP_DECISIONS = {"APPROVE", "ESCALATE", "MALFORMED"} + def _e(*codes: int) -> str: return f"\x1b[{';'.join(map(str, codes))}m" @@ -393,14 +411,30 @@ def normalize_agent_event(entry: dict[str, Any], idx: int, args: argparse.Namesp return base +def _allowed_or_unknown(value: Any, allowed: set[str]) -> str: + text = str(value) if value is not None else "" + return text if text in allowed else "unknown" + + def normalize_loop_event(entry: dict[str, Any]) -> dict[str, Any]: - """Map the privacy-whitelisted loop event shape into data-layer fields.""" + """Map the privacy-whitelisted loop event shape into data-layer fields. + + entry's `event`/`status`/`decision` values come from a supervisor-controlled + finite set (VALID_LOOP_EVENTS/STATUSES/DECISIONS); anything else is redacted + to "unknown" rather than copied through, so a malformed or unexpected + events.jsonl row can't smuggle arbitrary text into the exported surface. + """ + status_or_decision = entry.get("status") or entry.get("decision") return { "timestamp": entry.get("timestamp") or now_iso(), "skill": "agentic-loop", - "action": str(entry.get("event") or "loop_event"), + "action": _allowed_or_unknown(entry.get("event"), VALID_LOOP_EVENTS), "workflow": str(entry.get("loop") or "loop"), - "result": str(entry.get("status") or entry.get("decision") or "observed"), + "result": ( + _allowed_or_unknown(status_or_decision, VALID_LOOP_STATUSES | VALID_LOOP_DECISIONS) + if status_or_decision is not None + else "observed" + ), "harness": "agentic-loop", "source": {"run_id": entry.get("run_id")}, "privacy_level": "local_only", diff --git a/.agent/tools/test_learn_episodic_mirror.py b/.agent/tools/test_learn_episodic_mirror.py index 808e3cc..fccfde5 100644 --- a/.agent/tools/test_learn_episodic_mirror.py +++ b/.agent/tools/test_learn_episodic_mirror.py @@ -24,12 +24,18 @@ def _load_learn(base_dir): """Load .agent/tools/learn.py with BASE/CANDIDATES pointed at base_dir. Sibling modules (text.word_set, cluster.pattern_id) are stubbed so the - test needs no part of the harness beyond learn.py itself. + test needs no part of the harness beyond learn.py itself. The stubs are + process-wide (sys.modules), so any previous entry for "text"/"cluster" + is saved and restored (or removed, if there was none) once exec_module + finishes -- otherwise a later test or import in the same process would + silently pick up these throwaway stand-ins instead of the real modules. """ + previous_modules = {} for name, attrs in [ ("text", {"word_set": lambda *a, **k: set()}), ("cluster", {"pattern_id": lambda claim, cond: "testcid" + str(abs(hash((claim, tuple(cond)))))[:6]}), ]: + previous_modules[name] = sys.modules.pop(name, None) m = types.ModuleType(name) for k, v in attrs.items(): setattr(m, k, v) @@ -38,7 +44,14 @@ def _load_learn(base_dir): module_path = Path(__file__).with_name("learn.py") spec = importlib.util.spec_from_file_location("learn_under_test", module_path) mod = importlib.util.module_from_spec(spec) - spec.loader.exec_module(mod) + try: + spec.loader.exec_module(mod) + finally: + for name, previous in previous_modules.items(): + if previous is None: + sys.modules.pop(name, None) + else: + sys.modules[name] = previous mod.BASE = base_dir mod.CANDIDATES = os.path.join(base_dir, "memory", "candidates") os.makedirs(mod.CANDIDATES, exist_ok=True) @@ -68,7 +81,7 @@ def test_stage_writes_one_episodic_mirror(self): def test_evidence_id_resolves_to_the_mirror(self): mod = _load_learn(self.tmp) cid, path = mod.stage("Serialize timestamps in UTC", ["timestamps", "utc"]) - candidate = json.loads(Path(path).read_text()) + candidate = json.loads(Path(path).read_text(encoding="utf-8")) evidence_ts = candidate["evidence_ids"][0] matching = [e for e in _episodic(self.tmp) if e["timestamp"] == evidence_ts] self.assertEqual(len(matching), 1) diff --git a/CHANGELOG.md b/CHANGELOG.md index 084e8ba..5875a2c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,29 @@ All notable changes to this project. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] + +### Fixed +- **Claude Code hook leaked home paths and file content** (#66, #67). Episodic + entries now store project- or `~`-relative paths (`` otherwise), + input key names, and character counts in place of raw edit/write content, + raw `tool_input` dumps, and raw tool output or error text. +- **Loop event export copied arbitrary text** (#64). `data_layer_export.py` + redacts loop `event`/`status`/`decision` values outside the set the loop + supervisor and agentic-stack-desktop write (including `exhausted`, + `failed_to_start`, `timed_out`, `phase_started`, `running`) to `unknown`. +- **MiniMax ignored `max_tokens`** (#65). The OpenAI-wire call now forwards it + as `max_tokens` (`max_completion_tokens` does not exist in the minimum + supported `openai==1.40.0`). +- **`test_learn_episodic_mirror.py` leaked module stubs** (#65). Stubbed + `sys.modules` entries are restored after import; candidate JSON is read as + UTF-8. + +### Docs +- README: seed-skill list and count (14), repo layout (`loops/`, Copilot/Pi + hooks, `tests/`), Windsurf rule path, and install notes moved under + "Once installed". + ## [0.19.1] — 2026-08-07 Patch release. Four correctness fixes in memory retrieval, project upgrade, and diff --git a/README.md b/README.md index 7e9a387..3bd1396 100644 --- a/README.md +++ b/README.md @@ -126,6 +126,29 @@ verb-style subcommands (works with both `install.sh` and `install.ps1`): PowerShell uses the same verbs, for example `.\install.ps1 dashboard`. +Bare `./install.sh` (no arguments) opens a **multi-select wizard** on +a fresh project — check every harness you actually use, hit enter, +each one gets installed. The wizard auto-detects harnesses already on +disk and pre-checks them. On a project that already has an +`install.json`, bare interactive `./install.sh` opens the dashboard. +In non-TTY shells (CI), it stays script-safe and prints the available +subcommands instead of opening a TUI. + +Upgrading from pre-v0.9? Run `./install.sh doctor` first — it +synthesizes `install.json` from on-disk adapter signals so the new +backend can track them. Installing on top without migration would +orphan the prior installs. + +Upgrading an already-installed project after `brew upgrade`? Run +`agentic-stack upgrade --dry-run` in the project first, then +`agentic-stack upgrade --yes` to refresh only skeleton-owned `.agent` +infrastructure (`harness/**/*.py`, top-level `memory/*.py`, `tools/*.py`, +the generated skill index, and new skill directories). It does not rewrite +`CLAUDE.md`, `.claude/settings.json`, personal/semantic/episodic/working +memory, candidates, or existing skill directories. `agentic-stack +sync-manifest` is available as a repair command if `_manifest.jsonl` drifts +from installed `SKILL.md` files. + ### Optional: external Brain integration [`codejunkie99/brain`](https://github.com/codejunkie99/brain) is the @@ -152,29 +175,6 @@ Installed `.agent/` projects also get `python3 .agent/tools/brain_bridge.py` and a `brain` seed skill so host agents can query or write Brain memory when a task needs cross-harness long-term recall. -Bare `./install.sh` (no arguments) opens a **multi-select wizard** on -a fresh project — check every harness you actually use, hit enter, -each one gets installed. The wizard auto-detects harnesses already on -disk and pre-checks them. On a project that already has an -`install.json`, bare interactive `./install.sh` opens the dashboard. -In non-TTY shells (CI), it stays script-safe and prints the available -subcommands instead of opening a TUI. - -Upgrading from pre-v0.9? Run `./install.sh doctor` first — it -synthesizes `install.json` from on-disk adapter signals so the new -backend can track them. Installing on top without migration would -orphan the prior installs. - -Upgrading an already-installed project after `brew upgrade`? Run -`agentic-stack upgrade --dry-run` in the project first, then -`agentic-stack upgrade --yes` to refresh only skeleton-owned `.agent` -infrastructure (`harness/**/*.py`, top-level `memory/*.py`, `tools/*.py`, -the generated skill index, and new skill directories). It does not rewrite -`CLAUDE.md`, `.claude/settings.json`, personal/semantic/episodic/working -memory, candidates, or existing skill directories. `agentic-stack -sync-manifest` is available as a repair command if `_manifest.jsonl` drifts -from installed `SKILL.md` files. - ## Onboarding wizard If you ran bare `./install.sh` (no adapter name), the wizard starts @@ -279,7 +279,7 @@ See [`docs/architecture.md`](docs/architecture.md) for the full lifecycle. Every guide shows the folder structure. This repo gives you the folder structure **plus the files that actually go inside**: a working portable -brain with nine seed skills, four memory layers, enforced permissions, a +brain with fourteen seed skills, four memory layers, enforced permissions, a nightly staging cycle, host-agent review tools, and adapters for multiple harnesses. @@ -305,11 +305,6 @@ harnesses. context cards, eval cases, training-ready JSONL, and readiness metrics without training a model or sending telemetry. -## Releases & changelog - -Per-version release notes live in [CHANGELOG.md](CHANGELOG.md). The -latest release, what broke, what's new, upgrade path, all there. - ## Memory search `[BETA]` Opt-in FTS5 keyword search over all memory documents: @@ -333,6 +328,8 @@ The index is stored at `.agent/memory/.index/` and gitignored. ├── harness/ # conductor + hooks (standalone path) │ └── hooks/ │ ├── claude_code_post_tool.py # rich PostToolUse logging (v0.8+) +│ ├── copilot_cli_post_tool.py # Copilot CLI postToolUse logging +│ ├── pi_post_tool.py # Pi tool_result logging │ ├── pre_tool_call.py # permissions enforcement │ ├── post_execution.py # log_execution() entry point │ └── on_failure.py # failure write + repeated-failure rewrite flag @@ -344,6 +341,7 @@ The index is stored at `.agent/memory/.index/` and gitignored. │ ├── review_state.py # candidate lifecycle + decision log │ ├── render_lessons.py # lessons.jsonl → LESSONS.md │ └── memory_search.py # [BETA] FTS5 search (opt-in) +├── loops/ # bounded loop contracts (budget, constraints, loops) ├── skills/ # _index.md + _manifest.jsonl + SKILL.md files ├── protocols/ # permissions + tool schemas + delegation │ └── hook_patterns.json # user-owned high/medium-stakes regex (v0.8+) @@ -395,9 +393,12 @@ harness_manager/ # v0.9.0 manifest-driven Python backend ├── transfer_bundle.py # export/import bundle codec + merge logic ├── skill_manifest.py # rebuilds skills/_manifest.jsonl from SKILL.md ├── upgrade.py # safe .agent infrastructure refresh +├── status.py # one-screen installed-adapter view +├── loops/ # loop supervisor: runner, process, worktrees, storage └── cli.py # argparse dispatcher for install.sh / install.ps1 docs/ # architecture, getting-started, per-harness +tests/ # pytest suite, incl. test_claude_code_hook.py (62 checks) schemas/data-layer/ # local dashboard/event schemas examples/data-layer/ # sanitized data-layer shapes schemas/flywheel/ # data-flywheel artifact schemas @@ -412,8 +413,7 @@ onboard_ui.py # ANSI palette, banner, clack-style layout onboard_widgets.py # arrow-key prompts (text, select, confirm) onboard_render.py # answers → PREFERENCES.md content onboard_write.py # atomic file write with backup -test_claude_code_hook.py # hook validation suite (54 checks) -verify_codex_fixes.py # v0.8.0 regression checks (33 checks) +verify_codex_fixes.py # v0.8.0 regression checks ``` ## Supported harnesses @@ -424,7 +424,7 @@ verify_codex_fixes.py # v0.8.0 regression checks (33 checks) | **GitHub Copilot CLI** | `AGENTS.md` + `.github/instructions/*.instructions.md` | yes (postToolUse, sessionEnd) | | **Cursor** | `.cursor/rules/*.mdc` | no (manual reflect calls) | | **Google Gemini CLI** | `gemini.md` + `.gemini/skills/` | no (manual reflect calls) | -| **Windsurf** | `.windsurfrules` | no (manual reflect calls) | +| **Windsurf** | `.windsurf/rules/*.md` + legacy `.windsurfrules` | no (manual reflect calls) | | **OpenCode** | `AGENTS.md` + `opencode.json` | partial (permission rules) | | **OpenClaw** | `AGENTS.md` (auto-injected) + per-project `openclaw agents add --workspace` | varies by fork | | **Hermes Agent** | `AGENTS.md` (agentskills.io compatible) | partial (own memory) | @@ -449,6 +449,11 @@ verify_codex_fixes.py # v0.8.0 regression checks (33 checks) training-ready JSONL, and flywheel metrics - **tldraw** — opt-in beta skill for live canvas diagrams with a local snapshot store under `.agent/skills/tldraw/` +- **brain** — queries and writes the optional external Brain memory through + `brain_bridge.py` +- **loop-triage / loop-verifier / loop-constraints / loop-guard** — read + `.agent/loops` contracts: read-only triage, deterministic verification, + path and approval gates, and pause/budget/stagnation decisions ## How it compounds diff --git a/tests/test_claude_code_hook.py b/tests/test_claude_code_hook.py index 0f1b98c..64fb5b7 100644 --- a/tests/test_claude_code_hook.py +++ b/tests/test_claude_code_hook.py @@ -340,6 +340,103 @@ def test_failure_write(mod): for label, passed in checks: (ok if passed else fail)(f" entry.{label}") +def test_no_raw_content_or_paths_persisted(mod): + section("10b. Privacy — no raw content, no raw absolute paths persisted") + + home_file = os.path.join(os.path.expanduser("~"), "supabase", "secrets.env") + payload = { + "tool_name": "Edit", + "tool_input": { + "file_path": home_file, + "old_string": "STRIPE_SECRET_KEY=sk_live_topsecretvalue12345", + "new_string": "STRIPE_SECRET_KEY=sk_live_rotatedvalue67890", + }, + "tool_response": {"output": "", "exit_code": 0, "error": ""}, + } + rc, entry, stderr = run_hook(payload) + if entry is None: + fail("no entry written for privacy-check Edit case") + return + ok("privacy-check Edit entry written") + + blob = json.dumps(entry) + checks = [ + ("no raw home directory in entry", + os.path.expanduser("~") not in blob), + ("action uses normalized path, not raw home path", + home_file not in entry.get("action", "")), + ("reflection does not contain old secret value", + "sk_live_topsecretvalue12345" not in blob), + ("reflection does not contain new secret value", + "sk_live_rotatedvalue67890" not in blob), + ("detail carries char counts, not the raw strings", + "old_string_chars" in entry.get("detail", "") + or "chars ->" in entry.get("reflection", "")), + ] + for label, passed in checks: + (ok if passed else fail)(f" entry.{label}") + + write_payload = { + "tool_name": "Write", + "tool_input": { + "file_path": os.path.join(os.path.expanduser("~"), "notes", "private.md"), + "content": "API_KEY=super-secret-value-should-not-leak\n", + }, + "tool_response": {"output": "", "exit_code": 0, "error": ""}, + } + rc, entry, stderr = run_hook(write_payload) + if entry is None: + fail("no entry written for privacy-check Write case") + return + blob = json.dumps(entry) + checks = [ + ("Write entry has no raw home path", + os.path.expanduser("~") not in blob), + ("Write entry does not contain file content", + "super-secret-value-should-not-leak" not in blob), + ] + for label, passed in checks: + (ok if passed else fail)(f" entry.{label}") + + # Fallback branch (no explicit handler) and tool output. + cases = [ + ("NotebookEdit fallback", { + "tool_name": "NotebookEdit", + "tool_input": { + # Short on purpose: the old fallback dumped inputs < 80 chars. + "notebook_path": os.path.join(os.path.expanduser("~"), "n"), + "new_source": "K=nb-sec", + }, + "tool_response": {"output": "", "exit_code": 0, "error": ""}, + }, "nb-sec"), + ("Read output", { + "tool_name": "Read", + "tool_input": {"file_path": os.path.join(os.path.expanduser("~"), ".env")}, + "tool_response": {"output": "DB_PASSWORD=read-output-secret", "exit_code": 0}, + }, "read-output-secret"), + ("failed Read error", { + "tool_name": "Read", + "tool_input": {"file_path": os.path.join(os.path.expanduser("~"), ".env")}, + "tool_response": {"output": "", "is_error": True, + "error": "cannot parse: DB_PASSWORD=read-error-secret"}, + }, "read-error-secret"), + ] + for label, case_payload, secret in cases: + rc, entry, stderr = run_hook(case_payload) + if entry is None: + fail(f"no entry written for privacy-check {label} case") + continue + blob = json.dumps(entry) + (ok if os.path.expanduser("~") not in blob else fail)( + f" entry.{label} has no raw home path") + (ok if secret not in blob else fail)( + f" entry.{label} does not contain raw content") + + outside = os.path.join(os.sep, "nonexistent-root", "bob", "secrets.env") + (ok if mod._normalize_path(outside) == "" else fail)( + " _normalize_path labels paths outside project and home ") + + def test_dream_cycle(): section("11. Dream cycle produces staged candidates from rich entries") # Use a universally high-stakes command so importance=9 / pain_score=5 @@ -509,6 +606,7 @@ def main(): test_reflection_non_empty(mod) test_full_write(mod) test_failure_write(mod) + test_no_raw_content_or_paths_persisted(mod) test_dream_cycle() test_memory_reflect_pain_flag() test_post_execution_pain_param() diff --git a/tests/test_data_layer_export.py b/tests/test_data_layer_export.py index 27033ec..031c06a 100644 --- a/tests/test_data_layer_export.py +++ b/tests/test_data_layer_export.py @@ -148,6 +148,60 @@ def test_exports_privacy_safe_loop_events_and_quality_counts(self): self.assertNotIn("do not export", exported) self.assertIn("agentic-loop", exported) + def test_redacts_loop_event_status_and_decision_outside_the_allowed_sets(self): + with tempfile.TemporaryDirectory() as tmp: + work = Path(tmp) + events = work / ".agent" / "runtime" / "loops" + events.mkdir(parents=True) + (events / "events.jsonl").write_text( + "\n".join( + json.dumps(row) + for row in [ + { + "run_id": "run-a", + "loop": "ci-sweeper", + "event": "", + }, + { + "run_id": "run-b", + "loop": "ci-sweeper", + "decision": "leaked prompt content here", + }, + {"run_id": "run-c", "loop": "ci-sweeper", "event": "completed"}, + # Real values the runner emits: breaker stop uses the + # run status as the event name; process.py statuses. + {"run_id": "run-d", "loop": "ci-sweeper", "event": "exhausted"}, + { + "run_id": "run-e", + "loop": "ci-sweeper", + "event": "maker_finished", + "status": "timed_out", + }, + { + "run_id": "run-f", + "loop": "ci-sweeper", + "event": "phase_started", + "status": "running", + }, + ] + ) + + "\n", + encoding="utf-8", + ) + result = self.run_export(work, "--window", "all", "--date", "2026-04-25") + self.assertEqual(result.returncode, 0, result.stderr) + out = work / ".agent" / "data-layer" / "exports" / "2026-04-25" + exported = (out / "agent-events.jsonl").read_text() + self.assertNotIn("exfiltrate", exported) + self.assertNotIn("leaked prompt content", exported) + self.assertNotIn("