From c6a02c00c2cb59e4c736876013aafde45a323d37 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 16 Sep 2026 14:56:33 +0000 Subject: [PATCH] feat: add semantic distance matrix and tool clustering (#82) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Deterministic pairwise similarity, SEARCH/WRITE/DESTRUCTIVE action families, cluster CLI, and N≥500 subsample / --allow-large performance guard. Co-authored-by: Abhinaysai Kamineni --- ROADMAP.md | 2 +- docs/semantic-distance.md | 48 ++++ examples/semantic/github_catalog.json | 83 +++++++ src/tool_semantics/cli.py | 89 +++++++ src/tool_semantics/semantic.py | 340 ++++++++++++++++++++++++++ tests/test_semantic.py | 70 ++++++ 6 files changed, 631 insertions(+), 1 deletion(-) create mode 100644 docs/semantic-distance.md create mode 100644 examples/semantic/github_catalog.json create mode 100644 src/tool_semantics/semantic.py create mode 100644 tests/test_semantic.py diff --git a/ROADMAP.md b/ROADMAP.md index 608ab8c..1597ee9 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -74,7 +74,7 @@ - [ ] Multi-signal rename confidence — [#80](https://github.com/askmy-stack/tool-semantics/issues/80) **P1** - [ ] Optional embedding layer — [#117](https://github.com/askmy-stack/tool-semantics/issues/117) **P2** - [ ] Optional LLM semantic judge — [#81](https://github.com/askmy-stack/tool-semantics/issues/81) **P2** -- [ ] Semantic distance / clustering — [#82](https://github.com/askmy-stack/tool-semantics/issues/82) **P2** +- [x] Semantic distance / clustering — [#82](https://github.com/askmy-stack/tool-semantics/issues/82) **P2** ## Milestone 10 — Traces & workflows - [ ] Trace schema + capture — [#83](https://github.com/askmy-stack/tool-semantics/issues/83) **P1** diff --git a/docs/semantic-distance.md b/docs/semantic-distance.md new file mode 100644 index 0000000..886bfdf --- /dev/null +++ b/docs/semantic-distance.md @@ -0,0 +1,48 @@ +# Semantic distance matrix and tool clustering (#82) + +Help developers understand large MCP catalogs: pairwise similarity plus +SEARCH / WRITE / DESTRUCTIVE-style action families. + +## Defaults + +| Mode | Behavior | +| --- | --- | +| Deterministic (default) | Weighted token Jaccard (`diff._tool_similarity`) — same heuristic as rename detection | +| Embeddings (optional) | Blend token score with cosine similarity when an `EmbeddingProvider` is passed | +| N ≥ 500 | Pairwise matrix is **subsampled** unless `--allow-large` / `allow_large=True` | + +Full pairwise cost is O(N²). For large catalogs prefer subsample or raise +`--max-tools`. + +## Action families (heuristic) + +| Family | Signals | +| --- | --- | +| `SEARCH` | `read_only` risk or name/description tokens like search/find/list/get | +| `WRITE` | `external_write` or create/update/write/send tokens | +| `DESTRUCTIVE` | `destructive` risk or delete/remove/drop tokens | +| `OTHER` | Everything else | + +## CLI + +```bash +tool-semantics capture examples/semantic/github_catalog.json -o snap.json +tool-semantics cluster snap.json --top 20 +tool-semantics cluster snap.json --allow-large # N≥500 full matrix +``` + +## Library + +```python +from tool_semantics.scanner import capture_manifest +from tool_semantics.semantic import compute_semantic_matrix, render_semantic_matrix_markdown + +snap = capture_manifest("examples/semantic/github_catalog.json") +report = compute_semantic_matrix(snap) +print(render_semantic_matrix_markdown(report)) +``` + +## Related + +- Tool collision / confusability (#79) +- Optional embeddings (#117) diff --git a/examples/semantic/github_catalog.json b/examples/semantic/github_catalog.json new file mode 100644 index 0000000..175bfb5 --- /dev/null +++ b/examples/semantic/github_catalog.json @@ -0,0 +1,83 @@ +{ + "protocol": "mcp-manifest-demo", + "serverName": "github-catalog-demo", + "serverVersion": "1.0.0", + "tools": [ + { + "name": "search_issues", + "description": "Search open GitHub issues matching a query.", + "risk": "read_only", + "inputSchema": { + "type": "object", + "properties": {"query": {"type": "string"}}, + "required": ["query"] + } + }, + { + "name": "find_pull_requests", + "description": "Find pull requests matching a search query.", + "risk": "read_only", + "inputSchema": { + "type": "object", + "properties": {"query": {"type": "string"}}, + "required": ["query"] + } + }, + { + "name": "list_repositories", + "description": "List repositories for the authenticated user.", + "risk": "read_only", + "inputSchema": {"type": "object", "properties": {}} + }, + { + "name": "create_issue", + "description": "Create a new GitHub issue.", + "risk": "external_write", + "inputSchema": { + "type": "object", + "properties": {"title": {"type": "string"}, "body": {"type": "string"}}, + "required": ["title", "body"] + } + }, + { + "name": "update_issue", + "description": "Update an existing GitHub issue.", + "risk": "external_write", + "inputSchema": { + "type": "object", + "properties": {"number": {"type": "integer"}, "title": {"type": "string"}}, + "required": ["number"] + } + }, + { + "name": "create_pull_request", + "description": "Create a pull request.", + "risk": "external_write", + "inputSchema": { + "type": "object", + "properties": {"title": {"type": "string"}, "head": {"type": "string"}}, + "required": ["title", "head"] + } + }, + { + "name": "delete_repository", + "description": "Permanently delete a repository.", + "risk": "destructive", + "inputSchema": { + "type": "object", + "properties": {"name": {"type": "string"}}, + "required": ["name"] + } + }, + { + "name": "remove_collaborator", + "description": "Remove a collaborator from a repository.", + "risk": "destructive", + "inputSchema": { + "type": "object", + "properties": {"user": {"type": "string"}}, + "required": ["user"] + } + } + ] +} diff --git a/src/tool_semantics/cli.py b/src/tool_semantics/cli.py index 51a515c..7f5af22 100644 --- a/src/tool_semantics/cli.py +++ b/src/tool_semantics/cli.py @@ -702,3 +702,92 @@ def compare( markdown_output.write_text(render_markdown(report), encoding="utf-8") if fails_policy: raise typer.Exit(code=1) + + +@app.command("cluster") +def cluster_cmd( + snapshot: Annotated[ + Path, + typer.Argument(help="Tool-Semantics snapshot JSON."), + ], + top: Annotated[ + int, + typer.Option("--top", help="How many top similar pairs to show."), + ] = 20, + threshold: Annotated[ + float, + typer.Option( + "--threshold", + help="Similarity threshold for pairs / clusters (0–1).", + ), + ] = 0.55, + allow_large: Annotated[ + bool, + typer.Option( + "--allow-large", + help="Compute full pairwise matrix when N≥500 (expensive).", + ), + ] = False, + max_tools: Annotated[ + int, + typer.Option( + "--max-tools", + help="Subsample size when N≥500 and --allow-large is not set.", + ), + ] = 250, + json_output: Annotated[ + Path | None, + typer.Option("--json-output", help="Write JSON semantic matrix report."), + ] = None, + markdown_output: Annotated[ + Path | None, + typer.Option("--markdown-output", help="Write Markdown semantic report."), + ] = None, + verbose: Annotated[ + bool, + typer.Option("--verbose", "-v", help="Log cluster steps to stderr."), + ] = False, +) -> None: + """Surface semantic distance matrix highlights and action-family clusters (#82).""" + from tool_semantics.semantic import ( + compute_semantic_matrix, + matrix_as_dict, + render_semantic_matrix_markdown, + ) + + _require_snapshot_file(snapshot, "Snapshot") + if top < 1: + console.print("[red]--top must be >= 1[/red]") + raise typer.Exit(code=2) + if not 0.0 <= threshold <= 1.0: + console.print("[red]--threshold must be in [0, 1][/red]") + raise typer.Exit(code=2) + if max_tools < 2: + console.print("[red]--max-tools must be >= 2[/red]") + raise typer.Exit(code=2) + + try: + snap = read_snapshot(snapshot) + except (ManifestError, FileNotFoundError, ValueError, OSError) as exc: + console.print(f"[red]cluster load failed:[/red] {exc}") + raise typer.Exit(code=2) from exc + + _log_verbose(verbose, f"tools={len(snap.tools)} top={top} allow_large={allow_large}") + report = compute_semantic_matrix( + snap, + top_k=top, + similar_threshold=threshold, + allow_large=allow_large, + max_tools=max_tools, + ) + console.print(render_semantic_matrix_markdown(report)) + + if json_output is not None: + json_output.parent.mkdir(parents=True, exist_ok=True) + json_output.write_text( + json.dumps(matrix_as_dict(report), indent=2) + "\n", + encoding="utf-8", + ) + if markdown_output is not None: + markdown_output.parent.mkdir(parents=True, exist_ok=True) + markdown_output.write_text(render_semantic_matrix_markdown(report), encoding="utf-8") diff --git a/src/tool_semantics/semantic.py b/src/tool_semantics/semantic.py new file mode 100644 index 0000000..5ded530 --- /dev/null +++ b/src/tool_semantics/semantic.py @@ -0,0 +1,340 @@ +"""Semantic distance matrix and tool clustering (#82). + +Deterministic default (token Jaccard via ``diff._tool_similarity``). Optional +embedding blend when a provider is supplied. Large catalogs (N≥500) require an +explicit flag or are subsampled for the pairwise matrix. +""" + +from __future__ import annotations + +import math +from collections.abc import Callable, Sequence +from enum import StrEnum +from typing import Any, Protocol + +from pydantic import BaseModel, Field + +from tool_semantics.diff import _tool_similarity +from tool_semantics.models import InterfaceSnapshot, RiskLevel, ToolContract + +DEFAULT_SIMILAR_PAIR_THRESHOLD = 0.55 +LARGE_CATALOG_N = 500 +_EMBEDDING_BLEND = 0.5 + + +class ActionFamily(StrEnum): + """Documented heuristic action families for catalog browsing.""" + + SEARCH = "SEARCH" + WRITE = "WRITE" + DESTRUCTIVE = "DESTRUCTIVE" + OTHER = "OTHER" + + +class EmbeddingProvider(Protocol): + def embed_texts(self, texts: Sequence[str]) -> list[list[float]]: ... + + +class SimilarPair(BaseModel): + left: str + right: str + score: float + layer: str = "token" # token | mixed + + +class ToolCluster(BaseModel): + family: ActionFamily | None = None + label: str + tools: list[str] = Field(default_factory=list) + + +class SemanticMatrixReport(BaseModel): + tool_names: list[str] = Field(default_factory=list) + # Dense upper-triangle encoded as list of (i, j, score) with i < j. + pairs: list[SimilarPair] = Field(default_factory=list) + top_similar: list[SimilarPair] = Field(default_factory=list) + action_families: list[ToolCluster] = Field(default_factory=list) + similarity_clusters: list[ToolCluster] = Field(default_factory=list) + subsampled: bool = False + subsample_note: str = "" + tool_count: int = 0 + compared_count: int = 0 + + +_SEARCH_TOKENS = frozenset( + { + "search", + "find", + "list", + "get", + "read", + "fetch", + "query", + "lookup", + "show", + "describe", + } +) +_WRITE_TOKENS = frozenset( + { + "create", + "add", + "update", + "edit", + "write", + "set", + "put", + "post", + "patch", + "send", + "upload", + "insert", + } +) +_DESTRUCTIVE_TOKENS = frozenset( + { + "delete", + "remove", + "drop", + "destroy", + "purge", + "revoke", + "cancel", + "archive", + "force", + } +) + + +def _name_tokens(name: str) -> set[str]: + return {tok for tok in name.lower().replace("-", "_").split("_") if tok} + + +def classify_action_family(tool: ToolContract) -> ActionFamily: + """Heuristic SEARCH / WRITE / DESTRUCTIVE / OTHER from name tokens + risk.""" + tokens = _name_tokens(tool.name) | { + tok + for tok in "".join(ch.lower() if ch.isalnum() else " " for ch in tool.description).split() + if tok + } + if tool.risk == RiskLevel.DESTRUCTIVE or tokens & _DESTRUCTIVE_TOKENS: + return ActionFamily.DESTRUCTIVE + if tool.risk == RiskLevel.EXTERNAL_WRITE or tokens & _WRITE_TOKENS: + return ActionFamily.WRITE + if tool.risk == RiskLevel.READ_ONLY or tokens & _SEARCH_TOKENS: + return ActionFamily.SEARCH + return ActionFamily.OTHER + + +def _cosine(left: Sequence[float], right: Sequence[float]) -> float: + if len(left) != len(right) or not left: + return 0.0 + dot = sum(a * b for a, b in zip(left, right, strict=True)) + norm_l = math.sqrt(sum(a * a for a in left)) + norm_r = math.sqrt(sum(b * b for b in right)) + if norm_l == 0.0 or norm_r == 0.0: + return 0.0 + return max(0.0, min(1.0, dot / (norm_l * norm_r))) + + +def _tool_text(tool: ToolContract) -> str: + return f"{tool.name}\n{tool.description}" + + +def pairwise_similarity( + left: ToolContract, + right: ToolContract, + *, + embeddings: EmbeddingProvider | None = None, +) -> tuple[float, str]: + layer1 = _tool_similarity(left, right) + if embeddings is None: + return round(layer1, 4), "token" + vectors = embeddings.embed_texts([_tool_text(left), _tool_text(right)]) + if len(vectors) != 2: + raise ValueError("embed_texts must return one vector per input") + layer2 = _cosine(vectors[0], vectors[1]) + blended = ((1.0 - _EMBEDDING_BLEND) * layer1) + (_EMBEDDING_BLEND * layer2) + return round(blended, 4), "mixed" + + +def _subsample_tools( + tools: list[ToolContract], + *, + max_tools: int, +) -> list[ToolContract]: + """Deterministic stride subsample preserving name order.""" + if len(tools) <= max_tools: + return tools + step = len(tools) / max_tools + indexes = sorted({min(len(tools) - 1, int(index * step)) for index in range(max_tools)}) + return [tools[index] for index in indexes] + + +def compute_semantic_matrix( + snapshot: InterfaceSnapshot, + *, + top_k: int = 20, + similar_threshold: float = DEFAULT_SIMILAR_PAIR_THRESHOLD, + embeddings: EmbeddingProvider | None = None, + allow_large: bool = False, + max_tools: int = 250, +) -> SemanticMatrixReport: + """Compute pairwise similarities and cluster tools for a snapshot. + + For N≥500, either pass ``allow_large=True`` (full O(N²) matrix — expensive) + or the catalog is deterministically subsampled to ``max_tools`` (default 250). + """ + ordered = sorted(snapshot.tools, key=lambda tool: tool.name) + tool_count = len(ordered) + subsampled = False + note = "" + working = ordered + if tool_count >= LARGE_CATALOG_N and not allow_large: + working = _subsample_tools(ordered, max_tools=max_tools) + subsampled = True + note = ( + f"Catalog has {tool_count} tools (≥{LARGE_CATALOG_N}); " + f"pairwise matrix subsampled to {len(working)}. " + "Pass allow_large=True / --allow-large for the full matrix." + ) + elif tool_count >= LARGE_CATALOG_N and allow_large: + note = ( + f"Computing full pairwise matrix for {tool_count} tools " + f"(O(N²) ≈ {tool_count * (tool_count - 1) // 2} pairs)." + ) + + names = [tool.name for tool in working] + pairs: list[SimilarPair] = [] + for index, left in enumerate(working): + for right in working[index + 1 :]: + score, layer = pairwise_similarity(left, right, embeddings=embeddings) + pairs.append(SimilarPair(left=left.name, right=right.name, score=score, layer=layer)) + + ranked = sorted(pairs, key=lambda item: (-item.score, item.left, item.right)) + top_similar = [item for item in ranked if item.score >= similar_threshold][:top_k] + if not top_similar: + top_similar = ranked[:top_k] + + # Action-family clusters (heuristic). + family_buckets: dict[ActionFamily, list[str]] = {family: [] for family in ActionFamily} + for tool in ordered: + family_buckets[classify_action_family(tool)].append(tool.name) + action_families = [ + ToolCluster(family=family, label=family.value, tools=sorted(members)) + for family, members in family_buckets.items() + if members + ] + + # Similarity clusters: connected components of pairs ≥ threshold. + parent: dict[str, str] = {} + + def find(name: str) -> str: + parent.setdefault(name, name) + while parent[name] != name: + parent[name] = parent[parent[name]] + name = parent[name] + return name + + def union(left: str, right: str) -> None: + root_l, root_r = find(left), find(right) + if root_l != root_r: + parent[root_r] = root_l + + for pair in pairs: + if pair.score >= similar_threshold: + union(pair.left, pair.right) + buckets: dict[str, set[str]] = {} + for name in names: + buckets.setdefault(find(name), set()).add(name) + similarity_clusters = [ + ToolCluster( + family=None, + label=f"cluster-{index + 1}", + tools=sorted(members), + ) + for index, members in enumerate( + sorted(buckets.values(), key=lambda group: (-len(group), sorted(group)[0])) + ) + if len(members) > 1 + ] + + return SemanticMatrixReport( + tool_names=names, + pairs=pairs, + top_similar=top_similar, + action_families=action_families, + similarity_clusters=similarity_clusters, + subsampled=subsampled, + subsample_note=note, + tool_count=tool_count, + compared_count=len(working), + ) + + +def render_semantic_matrix_markdown(report: SemanticMatrixReport) -> str: + lines = [ + "## Semantic distance / clustering", + "", + f"Tools in snapshot: **{report.tool_count}**; compared: **{report.compared_count}**.", + "", + ] + if report.subsample_note: + lines.append(f"_{report.subsample_note}_") + lines.append("") + + lines.extend( + [ + "### Top similar pairs", + "", + "| Left | Right | Score | Layer |", + "| --- | --- | ---: | --- |", + ] + ) + if not report.top_similar: + lines.append("| — | — | — | — |") + for pair in report.top_similar: + lines.append(f"| `{pair.left}` | `{pair.right}` | {pair.score:.2f} | `{pair.layer}` |") + lines.append("") + + lines.extend( + [ + "### Action families", + "", + "| Family | Tools |", + "| --- | --- |", + ] + ) + for cluster in report.action_families: + tools = ", ".join(f"`{name}`" for name in cluster.tools) or "—" + lines.append(f"| `{cluster.label}` | {tools} |") + lines.append("") + + if report.similarity_clusters: + lines.extend( + [ + "### Similarity clusters", + "", + "| Cluster | Tools |", + "| --- | --- |", + ] + ) + for cluster in report.similarity_clusters: + tools = ", ".join(f"`{name}`" for name in cluster.tools) + lines.append(f"| `{cluster.label}` | {tools} |") + lines.append("") + + lines.append( + "_Deterministic token Jaccard by default; embeddings optional. " + "Action families are documented heuristics (SEARCH / WRITE / DESTRUCTIVE / OTHER)._" + ) + lines.append("") + return "\n".join(lines) + + +# Optional hook for callers that want a custom scorer. +SimilarityFn = Callable[[ToolContract, ToolContract], tuple[float, str]] + + +def matrix_as_dict(report: SemanticMatrixReport) -> dict[str, Any]: + return report.model_dump(mode="json") diff --git a/tests/test_semantic.py b/tests/test_semantic.py new file mode 100644 index 0000000..9afda44 --- /dev/null +++ b/tests/test_semantic.py @@ -0,0 +1,70 @@ +"""Tests for semantic distance matrix / clustering (#82).""" + +from __future__ import annotations + +from pathlib import Path + +from typer.testing import CliRunner + +from tool_semantics.benchmarks import snapshot_from_manifest, synthesize_manifest +from tool_semantics.cli import app +from tool_semantics.scanner import capture_manifest, write_snapshot +from tool_semantics.semantic import ( + LARGE_CATALOG_N, + ActionFamily, + classify_action_family, + compute_semantic_matrix, + render_semantic_matrix_markdown, +) + +runner = CliRunner() +FIXTURE = Path("examples/semantic/github_catalog.json") + + +def test_action_families_and_matrix_on_github_catalog() -> None: + snap = capture_manifest(FIXTURE) + assert classify_action_family(next(t for t in snap.tools if t.name == "search_issues")) == ( + ActionFamily.SEARCH + ) + assert classify_action_family(next(t for t in snap.tools if t.name == "create_issue")) == ( + ActionFamily.WRITE + ) + assert ( + classify_action_family(next(t for t in snap.tools if t.name == "delete_repository")) + == ActionFamily.DESTRUCTIVE + ) + + report = compute_semantic_matrix(snap, top_k=10, similar_threshold=0.3) + assert report.tool_count == 8 + assert report.pairs + assert report.top_similar + families = {cluster.label: cluster.tools for cluster in report.action_families} + assert "search_issues" in families["SEARCH"] + assert "create_issue" in families["WRITE"] + assert "delete_repository" in families["DESTRUCTIVE"] + md = render_semantic_matrix_markdown(report) + assert "Semantic distance" in md + assert "Action families" in md + + +def test_large_catalog_subsamples_unless_allow_large() -> None: + snap = snapshot_from_manifest(synthesize_manifest(LARGE_CATALOG_N, prefix="tool")) + limited = compute_semantic_matrix(snap, allow_large=False, max_tools=50) + assert limited.subsampled + assert limited.compared_count == 50 + assert "subsampled" in limited.subsample_note.lower() or "Catalog has" in limited.subsample_note + + # Full matrix is expensive; only check it does not subsample when allowed with smaller N. + small = snapshot_from_manifest(synthesize_manifest(20, prefix="t")) + full = compute_semantic_matrix(small, allow_large=True) + assert not full.subsampled + assert full.compared_count == 20 + + +def test_cluster_cli(tmp_path: Path) -> None: + snap_path = tmp_path / "snap.json" + write_snapshot(capture_manifest(FIXTURE), snap_path) + result = runner.invoke(app, ["cluster", str(snap_path), "--top", "5"]) + assert result.exit_code == 0, result.stdout + result.stderr + assert "Semantic distance" in result.stdout + assert "SEARCH" in result.stdout