Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,10 @@
Generated from [docs/PLAN.md](docs/PLAN.md) — the single
source of truth. Newest first.

## v2.92.0

- one participant's trajectory: `similar --agent NAME` aligns a single agent's action stream instead of the interleaved multi-agent run — two research agents taking different paths inside otherwise identical runs become distinguishable, and the interleaved noise (the other agent's identical steps, messages, responses) stops propping up false neighbors. tokens()/rank_similar/similar_payload take the filter; CLI, MCP schema and docs/mcp-tools.json carry it. payload gains an "agent" key when scoped.

## v2.91.0

- the audit reaches standing surfaces: `audit` joins the MCP surface (39 tools — doctor section, gate section, digest trend + forecast, `fix` repairs in place), and `init` now wires a nightly `agent-audit.yml` (cron 03:00, `--fix` on) beside the PR gate — the composed door is something a schedule runs, so both standing surfaces carry it. banner counts and docs/mcp-tools.json follow.
Expand Down
9 changes: 9 additions & 0 deletions docs/PLAN.md
Original file line number Diff line number Diff line change
Expand Up @@ -3333,3 +3333,12 @@ Anthropic, *Effective Context Engineering for AI Agents*, 2025.
fix-in-place; helpers extracted (xenon).
- init wires agent-audit.yml (nightly cron, --fix on) beside
the PR gate; init tests updated for four files.

341. **v2.92.0 - one participant's trajectory** (planned):
- align.tokens/rank_similar/similar_payload gain agent=; the
filter keeps only that agent's tool steps (unattributed
steps never match a named filter).
- `similar --agent` CLI + MCP schema + mcp-tools.json.
- A variable-reuse incident wrote cli content over align.py
mid-round; caught by import, reverted from git, re-applied
atomically with per-edit anchors. No main-branch impact.
4 changes: 4 additions & 0 deletions docs/mcp-tools.json
Original file line number Diff line number Diff line change
Expand Up @@ -483,6 +483,10 @@
"type": "string",
"description": "compare against traces in another store"
},
"agent": {
"type": "string",
"description": "align one participant's trajectory"
},
"top": {
"type": "number",
"description": "how many candidates (default 5)"
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "hatchling.build"

[project]
name = "approximately"
version = "2.91.0"
version = "2.92.0"
description = "Approximate memory, exact accountability: a context runtime (budgeted windows, pins, recall probes) plus a flight recorder (MAST failure attribution, replay, regression tests) for AI agents."
readme = "README.md"
license = { file = "LICENSE" }
Expand Down
43 changes: 29 additions & 14 deletions src/approximately/align.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@
from __future__ import annotations

import json
from typing import List, Tuple
from typing import List, Optional, Tuple

from .trace import TOOL_CALL, Step, Trace

Expand Down Expand Up @@ -54,9 +54,18 @@ def normalize_step(step: Step) -> Tuple[str, str]:
return f"{head}:{values}", f"{head} {{{keys}}}"


def tokens(trace: Trace) -> List[Tuple[str, str]]:
"""Identity/structure token pairs for every tool-call step of a trace."""
return [normalize_step(s) for s in trace.steps if s.kind == TOOL_CALL]
def tokens(trace: Trace,
agent: Optional[str] = None) -> List[Tuple[str, str]]:
"""Identity/structure token pairs for every tool-call step.

With ``agent``, only that agent's steps align — the trajectory
of one participant in a multi-agent run, not the interleaved
stream of everyone. Steps without an agent never match a named
filter (the unattributed stream is its own alignment problem)."""
steps = (s for s in trace.steps if s.kind == TOOL_CALL)
if agent is not None:
steps = (s for s in steps if s.agent == agent)
return [normalize_step(s) for s in steps]


def align_score(a: List[Tuple[str, str]], b: List[Tuple[str, str]]) -> float:
Expand Down Expand Up @@ -103,7 +112,9 @@ def similarity(a: Trace, b: Trace) -> float:


def rank_similar(target: Trace, traces: List[Trace],
top: int = 5) -> List[Tuple[Trace, float]]:
top: int = 5,
agent: Optional[str] = None
) -> List[Tuple[Trace, float]]:
"""The ``top`` most alignment-similar traces to *target*, best first.

Two savings keep this linear-ish on big stores while returning
Expand All @@ -116,14 +127,14 @@ def rank_similar(target: Trace, traces: List[Trace],
DP is skipped. Input order is preserved, so ties break exactly
as they did before.
"""
target_tokens = tokens(target)
target_tokens = tokens(target, agent=agent)
n_target = len(target_tokens)
scored: List[Tuple[Trace, float]] = []
threshold = 0.0 # score of the current Nth place (0 while unfilled)
for candidate in traces:
if candidate.id == target.id:
continue
cand_tokens = tokens(candidate)
cand_tokens = tokens(candidate, agent=agent)
n_cand = len(cand_tokens)
if n_target == 0 or n_cand == 0:
score = 0.0
Expand Down Expand Up @@ -196,14 +207,18 @@ def dedupe_traces(traces: list, threshold: float = 0.95,


def similar_payload(target: "Trace", traces: list, top: int = 5,
min_score: float = 0.0) -> dict:
min_score: float = 0.0,
agent: Optional[str] = None) -> dict:
"""The `similar` answer as data (shared by CLI --json and MCP).

``min_score`` cuts weak neighbours: only matches scoring at least
that much survive (the top-N cap still applies)."""
ranked = rank_similar(target, traces, top=max(0, top))
return {"trace": target.id,
"matches": [{"id": c.id, "task": c.task,
"similarity": round(score, 4)}
for c, score in ranked
if score >= min_score]}
ranked = rank_similar(target, traces, top=max(0, top),
agent=agent)
matches = [{"id": c.id, "task": c.task,
"similarity": round(score, 4)}
for c, score in ranked if score >= min_score]
payload = {"trace": target.id, "matches": matches}
if agent is not None:
payload["agent"] = agent
return payload
6 changes: 5 additions & 1 deletion src/approximately/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -400,7 +400,8 @@ def cmd_similar(args: argparse.Namespace) -> int:
candidates = TraceStore(other).list_traces()
payload = similar_payload(trace, candidates, top=args.top,
min_score=getattr(args, "min_score",
0.0) or 0.0)
0.0) or 0.0,
agent=getattr(args, "agent", None))
if getattr(args, "json", False):
print(json.dumps(payload, indent=2))
return 0
Expand Down Expand Up @@ -3184,6 +3185,9 @@ def build_parser() -> argparse.ArgumentParser:
"(cross-project nearest neighbours)")
p.add_argument("--json", action="store_true",
help="emit the same payload as the MCP similar tool")
p.add_argument("--agent", metavar="NAME",
help="align one participant's trajectory instead "
"of the interleaved multi-agent stream")
p.set_defaults(func=cmd_similar)

p = sub.add_parser("predict", parents=[common],
Expand Down
7 changes: 5 additions & 2 deletions src/approximately/mcp_server.py
Original file line number Diff line number Diff line change
Expand Up @@ -454,7 +454,9 @@
"description": "compare against "
"traces in another "
"store"},
"top": {"type": "number",
"agent": {"type": "string",
"description": "align one participant's trajectory"},
"top": {"type": "number",
"description": "how many candidates "
"(default 5)"},
"min_score": {"type": "number",
Expand Down Expand Up @@ -1432,13 +1434,14 @@ def _tool_similar(ctx: ServerContext, args: Dict[str, Any]) -> dict:
min_score = (float(args["min_score"])
if args.get("min_score") is not None else 0.0)
candidates = store.list_traces()
agent = args.get("agent")
if args.get("other_store"):
from pathlib import Path as _Path

from .store import TraceStore as _TS

candidates = _TS(_Path(str(args["other_store"]))).list_traces()
return similar_payload(trace, candidates, top=top, min_score=min_score)
return similar_payload(trace, candidates, top=top, min_score=min_score, agent=agent)


def _tool_drift(ctx: ServerContext, args: Dict[str, Any]) -> dict:
Expand Down
88 changes: 88 additions & 0 deletions tests/test_v279_agent_align.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
"""Night VI, round 23: one participant's trajectory.

``similar --agent NAME`` aligns a single agent's action stream
instead of the interleaved multi-agent run — two research agents
taking different paths inside otherwise identical runs are now
distinguishable, and the interleaved noise (messages, responses,
other agents' steps) stops diluting the alignment.
"""

import json

from approximately.align import similar_payload, tokens
from approximately.cli import main
from approximately.recorder import Recorder
from approximately.store import TraceStore


def _team(store, task, researcher_tools, booker_tools):
with Recorder(task, model="m/1", store=store) as rec:
for t in researcher_tools:
rec.tool(t, {}, agent="researcher")
for t in booker_tools:
rec.tool(t, {}, agent="booker")
rec.respond("done", success=True)
return rec.trace


def _store(tmp_path):
store = TraceStore(tmp_path / "s")
_team(store, "route A", ["search", "compare", "search"],
["book"])
_team(store, "route A again", ["search", "compare", "search"],
["book"])
_team(store, "route B", ["search", "guess", "book"],
["book"])
return store


def test_tokens_filter_by_agent(tmp_path):
store = _store(tmp_path)
trace = store.list_traces()[0]
assert len(tokens(trace)) == 4 # interleaved stream
assert len(tokens(trace, agent="researcher")) == 3
assert len(tokens(trace, agent="booker")) == 1
assert tokens(trace, agent="nobody") == []


def test_agent_scoped_ranking_distinguishes_routes(tmp_path):
store = _store(tmp_path)
by_task = {t.task: t for t in store.list_traces()}
target = by_task["route A"]
payload = similar_payload(target, store.list_traces(),
agent="researcher")
assert payload["agent"] == "researcher"
matches = {m["id"]: m["similarity"] for m in payload["matches"]}
# route A again aligns perfectly on the researcher stream
assert matches[by_task["route A again"].id] == 1.0
assert matches[by_task["route B"].id] < 1.0


def test_interleaved_stream_masks_the_difference(tmp_path):
store = _store(tmp_path)
by_task = {t.task: t for t in store.list_traces()}
target = by_task["route A"]
route_b = by_task["route B"].id
interleaved = {m["id"]: m["similarity"]
for m in similar_payload(target,
store.list_traces())
["matches"]}
scoped = {m["id"]: m["similarity"]
for m in similar_payload(target, store.list_traces(),
agent="researcher")["matches"]}
# without the agent filter the booker's identical step props up
# route B's score relative to the scoped ranking
assert interleaved[route_b] > scoped[route_b]


def test_cli_door_passes_the_filter(tmp_path, capsys):
store = _store(tmp_path)
trace = next(t for t in store.list_traces()
if t.task == "route A")
rc = main(["similar", trace.id, "--store", str(store.directory),
"--agent", "researcher", "--json"])
out = capsys.readouterr().out
assert rc == 0
payload = json.loads(out)
assert payload["agent"] == "researcher"
assert payload["matches"][0]["similarity"] == 1.0
Loading