From 77473919128f78897e8f4bce7b30f4d58aeaaa12 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 16 Sep 2026 14:33:09 +0000 Subject: [PATCH] feat: add efficiency regression metrics (calls, latency, tokens) (#114) Track operational cost alongside correctness with relative deltas. Efficiency is informational only and never drives pass/fail alone. Co-authored-by: Abhinaysai Kamineni --- ROADMAP.md | 2 +- docs/efficiency.md | 43 ++++ docs/probes.md | 6 + examples/efficiency/sample_metrics.json | 26 +++ src/tool_semantics/efficiency.py | 251 ++++++++++++++++++++++++ src/tool_semantics/report.py | 5 + tests/test_efficiency.py | 82 ++++++++ 7 files changed, 414 insertions(+), 1 deletion(-) create mode 100644 docs/efficiency.md create mode 100644 examples/efficiency/sample_metrics.json create mode 100644 src/tool_semantics/efficiency.py create mode 100644 tests/test_efficiency.py diff --git a/ROADMAP.md b/ROADMAP.md index 608ab8c..5ad60ab 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -88,7 +88,7 @@ - [ ] Output-schema compatibility — [#89](https://github.com/askmy-stack/tool-semantics/issues/89) **P1** - [ ] Prompt / resource / extension diffs — [#90](https://github.com/askmy-stack/tool-semantics/issues/90), [#91](https://github.com/askmy-stack/tool-semantics/issues/91) **P1/P2** - [ ] Integrity monitoring — [#113](https://github.com/askmy-stack/tool-semantics/issues/113) **P2** -- [ ] Efficiency regression — [#114](https://github.com/askmy-stack/tool-semantics/issues/114) **P2** +- [x] Efficiency regression — [#114](https://github.com/askmy-stack/tool-semantics/issues/114) **P2** ## Milestone 12+ — Prove (benchmarks, research, DX) - [ ] Difficulty / messiness + horizon — [#109](https://github.com/askmy-stack/tool-semantics/issues/109), [#110](https://github.com/askmy-stack/tool-semantics/issues/110) **P2** diff --git a/docs/efficiency.md b/docs/efficiency.md new file mode 100644 index 0000000..129ac54 --- /dev/null +++ b/docs/efficiency.md @@ -0,0 +1,43 @@ +# Efficiency metrics (#114) + +Track operational cost alongside correctness. **Efficiency never replaces +correctness** in pass/fail — it only surfaces regressions for review. + +## Metrics + +| Field | Meaning | +| --- | --- | +| `tool_call_count` | Total tool calls | +| `failed_call_count` | Failed / errored calls | +| `retry_count` | Retries / re-attempts | +| `latency_ms` | Aggregate model/tool latency | +| `wall_clock_ms` | End-to-end duration | +| `prompt_tokens` / `completion_tokens` | Optional token usage | +| `model_cost` / `currency` | Optional provider cost | + +## Usage + +```python +from tool_semantics.efficiency import ( + EfficiencyMetrics, + append_efficiency_section, + compare_efficiency, + load_efficiency_pair, + render_efficiency_markdown, +) + +baseline, candidate = load_efficiency_pair("examples/efficiency/sample_metrics.json") +report = compare_efficiency(baseline, candidate) +assert report.has_regression +assert report.affects_pass_fail is False +print(render_efficiency_markdown(report)) +# Attach to any eval / probe markdown when metrics exist: +print(append_efficiency_section("# Eval report\n", report)) +``` + +Relative deltas are easy to read (e.g. tool calls `8→31`, latency `+94%`). +Default regression thresholds: +50% for calls/retries/failures, latency, +tokens, and cost (override via `*_regression_ratio` kwargs). + +Cost / token fields are omitted from comparison when absent on both sides. +Fixture: [`examples/efficiency/sample_metrics.json`](../examples/efficiency/sample_metrics.json). diff --git a/docs/probes.md b/docs/probes.md index a999a86..093ac9c 100644 --- a/docs/probes.md +++ b/docs/probes.md @@ -64,6 +64,12 @@ probe = Probe( | `TOOL_SEMANTICS_MODEL` / `OPENAI_MODEL` | Model id (default `gpt-4o-mini`) | | `TOOL_SEMANTICS_BASE_URL` / `OPENAI_BASE_URL` | OpenAI-compatible base URL | +## Efficiency (informational) + +When baseline/candidate run metrics are available, attach an +[`## EFFICIENCY`](efficiency.md) section via `append_efficiency_section`. +Efficiency regressions never flip probe/compare pass/fail on their own. + ## Safety - Do not embed secrets in probe intents or expected arguments. diff --git a/examples/efficiency/sample_metrics.json b/examples/efficiency/sample_metrics.json new file mode 100644 index 0000000..028a6a7 --- /dev/null +++ b/examples/efficiency/sample_metrics.json @@ -0,0 +1,26 @@ +{ + "baseline": { + "label": "github-v1-suite", + "tool_call_count": 8, + "failed_call_count": 0, + "retry_count": 0, + "latency_ms": 100, + "wall_clock_ms": 1200, + "prompt_tokens": 4000, + "completion_tokens": 800, + "model_cost": 0.02, + "currency": "USD" + }, + "candidate": { + "label": "github-v2-suite", + "tool_call_count": 31, + "failed_call_count": 2, + "retry_count": 4, + "latency_ms": 194, + "wall_clock_ms": 2800, + "prompt_tokens": 9200, + "completion_tokens": 1400, + "model_cost": 0.055, + "currency": "USD" + } +} diff --git a/src/tool_semantics/efficiency.py b/src/tool_semantics/efficiency.py new file mode 100644 index 0000000..28a1e49 --- /dev/null +++ b/src/tool_semantics/efficiency.py @@ -0,0 +1,251 @@ +"""Efficiency regression metrics (#114). + +Correctness remains the only pass/fail gate. Efficiency metrics surface +operational cost (calls, retries, latency, tokens) and relative regressions. +""" + +from __future__ import annotations + +from pathlib import Path +from typing import Any + +from pydantic import BaseModel, Field + + +class EfficiencyMetrics(BaseModel): + """Operational cost for a probe suite / workflow run.""" + + tool_call_count: int = 0 + failed_call_count: int = 0 + retry_count: int = 0 + latency_ms: float | None = None + wall_clock_ms: float | None = None + prompt_tokens: int | None = None + completion_tokens: int | None = None + # Optional provider-reported currency cost when available. + model_cost: float | None = None + currency: str | None = None + label: str = "" + + +class EfficiencyDelta(BaseModel): + metric: str + baseline: float | None = None + candidate: float | None = None + absolute: float | None = None + relative: float | None = None # fraction, e.g. 0.94 = +94% + regressing: bool = False + + +class EfficiencyReport(BaseModel): + baseline: EfficiencyMetrics = Field(default_factory=EfficiencyMetrics) + candidate: EfficiencyMetrics = Field(default_factory=EfficiencyMetrics) + deltas: list[EfficiencyDelta] = Field(default_factory=list) + # Informational only — never drives compare/probe exit codes by itself. + has_regression: bool = False + + @property + def affects_pass_fail(self) -> bool: + return False + + +def _rel(baseline: float, candidate: float) -> float | None: + if baseline == 0: + return None + return (candidate - baseline) / abs(baseline) + + +def _f(value: int | None) -> float | None: + return None if value is None else float(value) + + +def compare_efficiency( + baseline: EfficiencyMetrics, + candidate: EfficiencyMetrics, + *, + call_regression_ratio: float = 0.5, + latency_regression_ratio: float = 0.5, + token_regression_ratio: float = 0.5, + cost_regression_ratio: float = 0.5, +) -> EfficiencyReport: + """Compare efficiency metrics; flag relative regressions (informational).""" + deltas: list[EfficiencyDelta] = [] + + def add( + name: str, + base: float | None, + cand: float | None, + *, + threshold: float | None = None, + higher_is_worse: bool = True, + ) -> None: + if base is None and cand is None: + return + absolute = None + relative = None + regressing = False + if base is not None and cand is not None: + absolute = cand - base + relative = _rel(base, cand) + if threshold is not None and relative is not None: + if higher_is_worse and relative >= threshold: + regressing = True + if not higher_is_worse and relative <= -threshold: + regressing = True + deltas.append( + EfficiencyDelta( + metric=name, + baseline=base, + candidate=cand, + absolute=absolute, + relative=relative, + regressing=regressing, + ) + ) + + add( + "tool_call_count", + float(baseline.tool_call_count), + float(candidate.tool_call_count), + threshold=call_regression_ratio, + ) + add( + "failed_call_count", + float(baseline.failed_call_count), + float(candidate.failed_call_count), + threshold=call_regression_ratio, + ) + add( + "retry_count", + float(baseline.retry_count), + float(candidate.retry_count), + threshold=call_regression_ratio, + ) + add( + "latency_ms", + baseline.latency_ms, + candidate.latency_ms, + threshold=latency_regression_ratio, + ) + add( + "wall_clock_ms", + baseline.wall_clock_ms, + candidate.wall_clock_ms, + threshold=latency_regression_ratio, + ) + add( + "prompt_tokens", + _f(baseline.prompt_tokens), + _f(candidate.prompt_tokens), + threshold=token_regression_ratio, + ) + add( + "completion_tokens", + _f(baseline.completion_tokens), + _f(candidate.completion_tokens), + threshold=token_regression_ratio, + ) + add( + "model_cost", + baseline.model_cost, + candidate.model_cost, + threshold=cost_regression_ratio, + ) + + return EfficiencyReport( + baseline=baseline, + candidate=candidate, + deltas=deltas, + has_regression=any(item.regressing for item in deltas), + ) + + +def aggregate_efficiency( + records: list[EfficiencyMetrics], + *, + label: str = "", +) -> EfficiencyMetrics: + """Sum / average metrics across per-probe or per-trial records.""" + if not records: + return EfficiencyMetrics(label=label) + latency_values = [item.latency_ms for item in records if item.latency_ms is not None] + wall_values = [item.wall_clock_ms for item in records if item.wall_clock_ms is not None] + prompt_values = [item.prompt_tokens for item in records if item.prompt_tokens is not None] + completion_values = [ + item.completion_tokens for item in records if item.completion_tokens is not None + ] + cost_values = [item.model_cost for item in records if item.model_cost is not None] + currency = next((item.currency for item in records if item.currency), None) + return EfficiencyMetrics( + tool_call_count=sum(item.tool_call_count for item in records), + failed_call_count=sum(item.failed_call_count for item in records), + retry_count=sum(item.retry_count for item in records), + latency_ms=sum(latency_values) if latency_values else None, + wall_clock_ms=sum(wall_values) if wall_values else None, + prompt_tokens=sum(prompt_values) if prompt_values else None, + completion_tokens=sum(completion_values) if completion_values else None, + model_cost=sum(cost_values) if cost_values else None, + currency=currency, + label=label, + ) + + +def load_efficiency_pair(path: Path | str) -> tuple[EfficiencyMetrics, EfficiencyMetrics]: + """Load baseline/candidate metrics from a JSON fixture. + + Expected shape:: + + {"baseline": {...}, "candidate": {...}} + """ + import json + + payload: dict[str, Any] = json.loads(Path(path).read_text(encoding="utf-8")) + return ( + EfficiencyMetrics.model_validate(payload["baseline"]), + EfficiencyMetrics.model_validate(payload["candidate"]), + ) + + +def _fmt_rel(relative: float | None) -> str: + if relative is None: + return "n/a" + sign = "+" if relative >= 0 else "" + return f"{sign}{relative:.0%}" + + +def render_efficiency_markdown(report: EfficiencyReport) -> str: + lines = [ + "## EFFICIENCY", + "", + "_Informational only — does not affect pass/fail._", + "", + "| Metric | Baseline | Candidate | Δ | Relative | Regression? |", + "| --- | ---: | ---: | ---: | ---: | --- |", + ] + for item in report.deltas: + base = "n/a" if item.baseline is None else f"{item.baseline:g}" + cand = "n/a" if item.candidate is None else f"{item.candidate:g}" + absolute = "n/a" if item.absolute is None else f"{item.absolute:g}" + lines.append( + f"| `{item.metric}` | {base} | {cand} | {absolute} | " + f"{_fmt_rel(item.relative)} | " + f"{'yes' if item.regressing else 'no'} |" + ) + lines.append("") + if report.has_regression: + lines.append("Efficiency regressions detected (review cost/latency).") + else: + lines.append("No efficiency regressions above configured thresholds.") + lines.append("") + return "\n".join(lines) + + +def append_efficiency_section( + markdown: str, + report: EfficiencyReport | None, +) -> str: + """Append an EFFICIENCY section when metrics are available.""" + if report is None: + return markdown + body = markdown.rstrip() + "\n\n" + render_efficiency_markdown(report) + return body if body.endswith("\n") else body + "\n" diff --git a/src/tool_semantics/report.py b/src/tool_semantics/report.py index dfbf853..7654397 100644 --- a/src/tool_semantics/report.py +++ b/src/tool_semantics/report.py @@ -6,6 +6,11 @@ from typing import Any from tool_semantics.diff import CompatibilityReport, Severity +from tool_semantics.efficiency import ( # noqa: F401 — re-export for report consumers (#114) + EfficiencyReport, + append_efficiency_section, + render_efficiency_markdown, +) from tool_semantics.probes import ModelProbeReport, ProbeMetrics, ProbeReport, StabilityReport diff --git a/tests/test_efficiency.py b/tests/test_efficiency.py new file mode 100644 index 0000000..2bb5c45 --- /dev/null +++ b/tests/test_efficiency.py @@ -0,0 +1,82 @@ +"""Tests for efficiency regression metrics (#114).""" + +from __future__ import annotations + +from pathlib import Path + +from tool_semantics.efficiency import ( + EfficiencyMetrics, + aggregate_efficiency, + append_efficiency_section, + compare_efficiency, + load_efficiency_pair, + render_efficiency_markdown, +) + + +def test_efficiency_does_not_affect_pass_fail() -> None: + baseline = EfficiencyMetrics(tool_call_count=8, latency_ms=100) + candidate = EfficiencyMetrics(tool_call_count=31, latency_ms=194) + report = compare_efficiency(baseline, candidate) + assert report.has_regression + assert report.affects_pass_fail is False + calls = next(item for item in report.deltas if item.metric == "tool_call_count") + assert calls.baseline == 8 + assert calls.candidate == 31 + assert calls.regressing + latency = next(item for item in report.deltas if item.metric == "latency_ms") + assert latency.relative is not None + assert abs(latency.relative - 0.94) < 0.01 + md = render_efficiency_markdown(report) + assert "## EFFICIENCY" in md + assert "Informational only" in md + assert "tool_call_count" in md + assert "8" in md and "31" in md + assert "+94%" in md + + +def test_optional_cost_and_tokens() -> None: + baseline = EfficiencyMetrics( + tool_call_count=2, + prompt_tokens=100, + completion_tokens=50, + model_cost=0.01, + currency="USD", + ) + candidate = EfficiencyMetrics( + tool_call_count=2, + prompt_tokens=120, + completion_tokens=60, + model_cost=0.012, + currency="USD", + ) + report = compare_efficiency(baseline, candidate) + assert any(item.metric == "model_cost" for item in report.deltas) + assert any(item.metric == "prompt_tokens" for item in report.deltas) + # Same call count → not a call regression + assert not next(item for item in report.deltas if item.metric == "tool_call_count").regressing + + +def test_fixture_pair_and_report_section() -> None: + baseline, candidate = load_efficiency_pair(Path("examples/efficiency/sample_metrics.json")) + report = compare_efficiency(baseline, candidate) + assert report.has_regression + md = append_efficiency_section("# Probe report\n\n**Result:** `PASS`\n", report) + assert "## EFFICIENCY" in md + assert "Informational only" in md + assert append_efficiency_section("# ok\n", None) == "# ok\n" + + +def test_aggregate_efficiency() -> None: + total = aggregate_efficiency( + [ + EfficiencyMetrics(tool_call_count=3, failed_call_count=1, latency_ms=10), + EfficiencyMetrics(tool_call_count=5, retry_count=2, latency_ms=20), + ], + label="suite", + ) + assert total.tool_call_count == 8 + assert total.failed_call_count == 1 + assert total.retry_count == 2 + assert total.latency_ms == 30 + assert total.label == "suite"