Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions docs/privacy-boundary.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,13 @@ Only redacted runtime audit previews are uploaded by default:
PII summaries use only an allowlisted category name, per-category count, and a
bounded total value count. They never contain the matched value or evidence.

When a locally visible model payload contains a credential or personal-data
value, the existing audit `reasons` fields carry a human-readable exposure
summary and the actual enforcement outcome. Evidence uses fixed masks such as
`sk-****` and `***@***`; it never contains characters copied from the matched
value. Observer-only hooks explicitly say that the model call was not blocked.
This does not add fields or change the version-1 Cloud wire envelope.

`payloadBytes`, when present, means the exact UTF-8 byte length of the
serialized request body. Character counts and previews are not substitutes;
adapters leave the field absent and report `exact_payload_bytes` as missing
Expand Down
79 changes: 75 additions & 4 deletions src/runtime/protect.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ import { consumeApprovedApproval, writePendingApproval, type ApprovalRecord } fr
import { flushEventSpool, spoolEvent, writeAuditLog } from './audit.js';
import { evaluateRuntimeAction } from './decision.js';
import { isAgentGuardCliCommand } from './self-command.js';
import { redactText } from './redaction.js';
import { redactText, summarizeSensitiveData } from './redaction.js';
import type {
CoverageLevel,
CredentialKind,
Expand Down Expand Up @@ -119,6 +119,9 @@ export async function protectAction(options: ProtectOptions): Promise<ProtectRes
|| isClaudeObserverEvent(claudeHookEvent);
if (!auditSafe && shouldSuppressRuntimeReport(decision)) return null;

const enforcementStatus = action.enforcementStatus ?? enforcementStatusFor(action, decision);
decision = withFriendlySensitiveDataReason(action, decision, enforcementStatus);

const event: RuntimeAuditEvent = {
...action,
actionId: decision.actionId,
Expand All @@ -128,17 +131,17 @@ export async function protectAction(options: ProtectOptions): Promise<ProtectRes
riskLevel: decision.riskLevel,
reasons: decision.reasons,
policyVersion: decision.policyVersion,
...(decision.piiSummary ? { privacySummary: decision.piiSummary } : {}),
coverageLevel: decision.ruleEvaluations?.length
? mergeCoverage(action.coverageLevel, decision.coverageLevel)
: action.coverageLevel ?? decision.coverageLevel,
enforcementStatus: action.enforcementStatus ?? enforcementStatusFor(action, decision),
enforcementStatus,
missingFacts: uniqueMissingFacts([...(action.missingFacts ?? []), ...(decision.missingFacts ?? [])]),
metadata: {
...(action.metadata || {}),
evaluation: policySource === 'cloud-decision' ? 'cloud' : 'local-oss',
policySource,
...(decision.ruleEvaluations?.length ? { privacyRules: decision.ruleEvaluations } : {}),
...(decision.piiSummary ? { privacySummary: decision.piiSummary } : {}),
...(approvedGrant
? {
approvedByLocalGrant: true,
Expand Down Expand Up @@ -193,7 +196,13 @@ function enforceNativeHookDecision(
|| claudeEvent === 'UserPromptSubmit' || claudeEvent === 'PostToolUse'
|| claudeEvent === 'PostToolBatch' || claudeEvent === 'PostToolUseFailure'
|| claudeEvent === 'MessageDisplay' || claudeEvent === 'Stop' || claudeEvent === 'ConfigChange';
const containsSensitiveContent = scansVisibleContent
const blockingVisibleModelInput = action.actionType === 'llm_request'
&& action.canBlockCurrentAction !== false
&& (action.lifecycleStage === 'user_prompt'
|| action.lifecycleStage === 'run_start'
|| action.lifecycleStage === 'model_request'
|| action.lifecycleStage === 'post_tool_batch');
const containsSensitiveContent = (scansVisibleContent || blockingVisibleModelInput)
&& (redactText(action.input) !== action.input || action.metadata?.configSensitive === true);
const untrustedExpansion = claudeEvent === 'UserPromptExpansion'
&& isUntrustedClaudePromptExpansion(action.metadata);
Expand Down Expand Up @@ -242,6 +251,68 @@ function enforceNativeHookDecision(
};
}

function withFriendlySensitiveDataReason(
action: RuntimeAction,
decision: RuntimeDecision,
enforcementStatus: EnforcementStatus,
): RuntimeDecision {
const relevantLifecycle = action.actionType === 'llm_request'
|| action.actionType === 'llm_response'
|| decision.reasons.some((reason) => reason.code === 'PII_EGRESS');
if (!relevantLifecycle) return decision;

const summaries = summarizeSensitiveData(action.input).slice(0, 3);
if (summaries.length === 0) return decision;

const labels = summaries.map((item) => item.label);
const subject = labels.length === 1 ? labels[0] : 'sensitive data';
const outcome = sensitiveDataOutcome(enforcementStatus, decision.decision);
const reasonIndex = decision.reasons.findIndex((reason) => reason.code === 'PII_EGRESS');
const friendlyReason = {
code: 'PII_EGRESS',
severity: reasonIndex >= 0 ? decision.reasons[reasonIndex].severity : 'critical' as const,
title: `${capitalize(subject)} exposure ${outcome.title}`,
description: `Detected ${formatList(labels)} in content visible to the model; ${outcome.description}.`,
evidence: `types=${summaries.map((item) => item.kind).join(',')};masked=${summaries.map((item) => item.maskedValue).join(',')}`,
};
const reasons = [...decision.reasons];
if (reasonIndex >= 0) reasons[reasonIndex] = friendlyReason;
else reasons.unshift(friendlyReason);
return { ...decision, reasons };
}

function sensitiveDataOutcome(
enforcementStatus: EnforcementStatus,
decision: RuntimeDecision['decision'],
): { title: string; description: string } {
if (enforcementStatus === 'enforced' && decision === 'block') {
return { title: 'blocked', description: 'the current action was blocked' };
}
if (enforcementStatus === 'enforced' && decision === 'require_approval') {
return { title: 'held for approval', description: 'the current action was held for user approval' };
}
if (enforcementStatus === 'display_only') {
return { title: 'detected', description: 'only the displayed text could be masked; the model response was not blocked' };
}
if (enforcementStatus === 'would_block' || enforcementStatus === 'observed') {
return { title: 'detected', description: 'this lifecycle event was observe-only and did not block the model call' };
}
if (enforcementStatus === 'unsupported') {
return { title: 'detected', description: 'the host did not expose a supported blocking point' };
}
return { title: 'detected', description: 'the event was recorded without claiming that the model call was blocked' };
}

function formatList(values: string[]): string {
if (values.length <= 1) return values[0] ?? 'sensitive data';
if (values.length === 2) return `${values[0]} and ${values[1]}`;
return `${values.slice(0, -1).join(', ')}, and ${values.at(-1)}`;
}

function capitalize(value: string): string {
return value.length === 0 ? value : `${value[0].toUpperCase()}${value.slice(1)}`;
}

function isUntrustedClaudePromptExpansion(metadata: Record<string, unknown> | undefined): boolean {
const source = String(metadata?.commandSource || '').toLowerCase();
const type = String(metadata?.expansionType || '').toLowerCase();
Expand Down
89 changes: 89 additions & 0 deletions src/runtime/redaction.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,30 @@ import { redactPiiText } from '../scanner/rules/privacy.js';

const REDACTED = '[REDACTED]';

export type SensitiveDataKind =
| 'agentguard_api_token'
| 'openai_api_token'
| 'bearer_token'
| 'private_key'
| 'credential'
| 'national_id'
| 'bank_account'
| 'biometric'
| 'minor_data'
| 'health_record'
| 'location_trace'
| 'contact_dump'
| 'phone_number'
| 'email_address'
| 'hardcoded_dataset';

export interface SensitiveDataSummaryItem {
kind: SensitiveDataKind;
label: string;
maskedValue: string;
count: number;
}

const SECRET_VALUE_PATTERN =
/(?:token|api[_-]?key|secret|password|passwd|authorization|access[_-]?key|client[_-]?secret)=([^&\s'"`]+)/gi;
/**
Expand Down Expand Up @@ -65,6 +89,58 @@ export function redactText(value: unknown): string {
return redactUrlSecrets(redacted);
}

/**
* Describe sensitive values without retaining any characters from the match.
*
* The result is safe for audit reasons and Cloud timelines: masks are fixed
* placeholders such as `sk-****` and `***@***`, never partial raw values or
* reversible hashes. Detection happens locally against the original content.
*/
export function summarizeSensitiveData(value: unknown): SensitiveDataSummaryItem[] {
const text = String(value ?? '');
const summaries = new Map<SensitiveDataKind, SensitiveDataSummaryItem>();
const add = (kind: SensitiveDataKind, label: string, maskedValue: string, count = 1): void => {
const existing = summaries.get(kind);
if (existing) {
existing.count += count;
return;
}
summaries.set(kind, { kind, label, maskedValue, count });
};

if (/\bag_live_[A-Za-z0-9_-]{12,}\b/.test(text)) {
add('agentguard_api_token', 'AgentGuard API token', 'ag_live_****');
}
if (/\bsk-(?:or-v1-)?[A-Za-z0-9_-]{12,}\b/.test(text)) {
add('openai_api_token', 'OpenAI-compatible API token', 'sk-****');
}
if (/\bBearer\s+[A-Za-z0-9._~+/=-]{12,}\b/i.test(text)) {
add('bearer_token', 'Bearer token', 'Bearer ****');
}
if (/-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/.test(text)) {
add('private_key', 'private key', '-----BEGIN **** PRIVATE KEY-----');
}
if ((SECRET_VALUE_PATTERN.test(text) || SECRET_COLON_PATTERN.test(text))
&& !summaries.has('agentguard_api_token')
&& !summaries.has('openai_api_token')
&& !summaries.has('bearer_token')) {
add('credential', 'credential', 'credential=****');
}
SECRET_VALUE_PATTERN.lastIndex = 0;
SECRET_COLON_PATTERN.lastIndex = 0;

const piiText = redactPiiText(text);
for (const [marker, kind, label, maskedValue] of PII_SUMMARY_MARKERS) {
const count = piiText.split(marker).length - 1;
if (count > 0) add(kind, label, maskedValue, count);
}

if (summaries.size === 0 && redactText(text) !== text) {
add('credential', 'sensitive credential', 'credential=****');
}
return [...summaries.values()];
}

export function redactPreview(value: unknown, maxLength = 2000): string {
return redactText(value).slice(0, maxLength);
}
Expand Down Expand Up @@ -173,3 +249,16 @@ function redactUrlSecrets(value: string): string {
}
});
}

const PII_SUMMARY_MARKERS: ReadonlyArray<readonly [string, SensitiveDataKind, string, string]> = [
['[REDACTED:PII_EMAIL_ADDRESS]', 'email_address', 'email address', '***@***'],
['[REDACTED:PII_PHONE_NUMBER]', 'phone_number', 'phone number', '***-***-****'],
['[REDACTED:PII_NATIONAL_ID]', 'national_id', 'national ID', 'ID-****'],
['[REDACTED:PII_BANK_ACCOUNT]', 'bank_account', 'bank account', 'account-****'],
['[REDACTED:PII_BIOMETRIC]', 'biometric', 'biometric data', 'biometric-****'],
['[REDACTED:PII_MINOR_DATA]', 'minor_data', 'minor data', 'minor-data-****'],
['[REDACTED:PII_HEALTH_RECORD]', 'health_record', 'health record', 'health-record-****'],
['[REDACTED:PII_LOCATION_TRACE]', 'location_trace', 'location trace', 'location-****'],
['[REDACTED:PII_CONTACT_DUMP]', 'contact_dump', 'contact list', 'contacts-****'],
['[REDACTED:PII_HARDCODED_DATASET]', 'hardcoded_dataset', 'personal-data dataset', 'dataset-****'],
];
95 changes: 94 additions & 1 deletion src/tests/runtime-privacy.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,13 +6,106 @@ import { tmpdir } from 'node:os';
import { evaluateLocalAction } from '../runtime/evaluator.js';
import { evaluateLlmPrivacy } from '../runtime/privacy.js';
import { buildAuditEvent, writeAuditLog } from '../runtime/audit.js';
import { redactText } from '../runtime/redaction.js';
import { redactText, summarizeSensitiveData } from '../runtime/redaction.js';
import { getDefaultEffectiveRuntimePolicy } from '../runtime/policy.js';
import { exitCodeForDecision, formatProtectResult, protectAction } from '../runtime/protect.js';
import type { AgentGuardConfig } from '../config.js';
import type { LlmEndpointTier, RuntimeAction, RuntimeAuditEvent } from '../runtime/types.js';

describe('Runtime LLM privacy evaluation', () => {
it('builds fixed-mask summaries without retaining token or email characters', () => {
const token = 'sk-live-private-token-1234567890';
const email = 'private.person@corp.invalid';
const summary = summarizeSensitiveData(`api_key=${token} personal_email=${email}`);

assert.deepEqual(summary.map((item) => [item.kind, item.maskedValue]), [
['openai_api_token', 'sk-****'],
['email_address', '***@***'],
]);
assert.doesNotMatch(JSON.stringify(summary), /private-token|private\.person|corp\.invalid/);
});

it('reports a friendly blocked exposure reason without changing the audit wire shape', async () => {
const dir = mkdtempSync(join(tmpdir(), 'agentguard-friendly-privacy-event-'));
const config: AgentGuardConfig = {
version: 1,
level: 'balanced',
policyCachePath: join(dir, 'policy.json'),
auditPath: join(dir, 'audit.jsonl'),
eventSpoolPath: join(dir, 'spool.jsonl'),
};
const token = 'sk-live-private-token-1234567890';
const email = 'private.person@corp.invalid';
const result = await protectAction({
config,
agentHost: 'openclaw',
actionType: 'llm_request',
toolName: 'openclaw.before_agent_run',
rawInput: {
input: `api_key=${token} personal_email=${email}`,
sessionId: 'sess_friendly_privacy',
lifecycleStage: 'run_start',
canBlockCurrentAction: true,
coverageLevel: 'partial',
},
});

assert.ok(result);
assert.equal(result.decision.decision, 'block');
assert.equal(result.event.enforcementStatus, 'enforced');
const reason = result.decision.reasons.find((item) => item.code === 'PII_EGRESS');
assert.match(reason?.title ?? '', /exposure blocked/i);
assert.match(reason?.description ?? '', /API token.*email address.*current action was blocked/i);
assert.equal(reason?.evidence, 'types=openai_api_token,email_address;masked=sk-****,***@***');

const event = buildAuditEvent(result.event);
assert.equal(event.input, '[LOCAL_ONLY_LLM_CONTENT]');
assert.deepEqual(event.privacySummary, {
categories: [{ category: 'email_address', count: 1 }],
valueCount: 1,
});
assert.equal(event.metadata?.privacySummary, undefined);
assert.deepEqual(
Object.keys(JSON.parse(JSON.stringify(event))).sort(),
Object.keys(JSON.parse(JSON.stringify(result.event))).sort(),
);
const serialized = JSON.stringify(event);
assert.match(serialized, /sk-\*\*\*\*/);
assert.match(serialized, /\*\*\*@\*\*\*/);
assert.doesNotMatch(serialized, /private-token|private\.person|corp\.invalid/);
});

it('does not claim an observer-only privacy finding was blocked', async () => {
const dir = mkdtempSync(join(tmpdir(), 'agentguard-friendly-privacy-observer-'));
const config: AgentGuardConfig = {
version: 1,
level: 'balanced',
policyCachePath: join(dir, 'policy.json'),
auditPath: join(dir, 'audit.jsonl'),
eventSpoolPath: join(dir, 'spool.jsonl'),
};
const result = await protectAction({
config,
agentHost: 'openclaw',
actionType: 'llm_request',
toolName: 'openclaw.llm_input',
rawInput: {
input: 'personal_email=private.person@corp.invalid',
sessionId: 'sess_friendly_observer',
lifecycleStage: 'model_request',
canBlockCurrentAction: false,
coverageLevel: 'observe_only',
},
auditSafe: true,
});

assert.ok(result);
const reason = result.decision.reasons.find((item) => item.code === 'PII_EGRESS');
assert.match(reason?.title ?? '', /exposure detected/i);
assert.match(reason?.description ?? '', /observe-only and did not block/i);
assert.doesNotMatch(reason?.description ?? '', /was blocked/);
});

it('maps visible PII through T0-T4 without auto-allowing medium-severity findings', async () => {
const policy = getDefaultEffectiveRuntimePolicy();
const expected = new Map<LlmEndpointTier, string>([
Expand Down
Loading