fix(retrieval): the review drops records written after the call it re-runs (#4773)
CI & Build / Python lint (push) Successful in 4s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / TypeScript typecheck (push) Successful in 54s
CI & Build / integration (push) Successful in 1m8s
CI & Build / Python tests (push) Successful in 1m53s
CI & Build / Build & push image (push) Successful in 36s

The first live sample ranked records the call could never have been offered:
the session that made a call writes the decision, often quoting the message,
and that record tops the re-run. Judged, it inflates on_point exactly where
the budget is decided.

- _rerun over-fetches by POSTDATED_SLACK, drops records created after the
  call before ranking, and names them in `postdated`.
- each line carries `changed_since_call` (updated_at or a work log after
  the call) and `logged` (shown fresh then).
- judge_menu shares the re-run, so a post-dated record cannot be judged.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-10-03 19:42:20 -04:00
co-authored by Claude Opus 5.5
parent 6598c7fa85
commit 5d0e97576b
3 changed files with 109 additions and 14 deletions
+37 -6
View File
@@ -81,17 +81,22 @@ def test_the_re_run_searches_the_way_the_arm_does():
assert rerun.get(key) == arm[key], key
CALL = datetime(2026, 9, 20, 12, 0, tzinfo=timezone.utc)
BEFORE = CALL - timedelta(days=1)
@pytest.mark.asyncio
async def test_the_re_run_ranks_from_one_and_marks_the_budget_cut():
row = SimpleNamespace(query="q", threshold=0.55, limit_n=2, project_id=None)
row = SimpleNamespace(query="q", threshold=0.55, limit_n=2, project_id=None,
created_at=CALL)
hits = [
(0.81, fake_note(id=11, title="first")),
(0.74, fake_note(id=12, title="second")),
(0.66, fake_note(id=13, title="third")),
(0.81, fake_note(id=11, title="first", created_at=BEFORE, updated_at=BEFORE)),
(0.74, fake_note(id=12, title="second", created_at=BEFORE, updated_at=BEFORE)),
(0.66, fake_note(id=13, title="third", created_at=BEFORE, updated_at=BEFORE)),
]
async def search(user_id, query, **kw):
assert kw["limit"] == 3 and kw["threshold"] == 0.55
assert kw["limit"] == 3 + review.POSTDATED_SLACK and kw["threshold"] == 0.55
assert kw["project_id"] is None
kw["report"]["best_chunk"] = {12: {"text": "second\nthe passage that matched"}}
return hits
@@ -105,6 +110,31 @@ async def test_the_re_run_ranks_from_one_and_marks_the_budget_cut():
assert lines[0]["passage"] is None
@pytest.mark.asyncio
async def test_a_record_written_after_the_call_is_dropped_before_ranking():
"""The session that made a call goes on to write about it, and that record
outranks everything in a re-run. The call could never have been offered
it, so it must not take a rank — or a verdict — from what was."""
row = SimpleNamespace(query="q", threshold=0.55, limit_n=1, project_id=None,
created_at=CALL)
after = CALL + timedelta(hours=1)
hits = [
(0.90, fake_note(id=21, title="the decision", created_at=after, updated_at=after)),
(0.80, fake_note(id=22, title="held then", created_at=BEFORE, updated_at=BEFORE)),
(0.70, fake_note(id=23, title="edited since", created_at=BEFORE, updated_at=after)),
]
async def search(user_id, query, **kw):
return hits
rep: dict = {}
with patch("scribe.services.embeddings.semantic_search_notes", side_effect=search):
lines = await review._rerun(1, row, 2, report=rep)
assert rep["postdated"] == [21]
assert [(ln["rank"], ln["record_id"], ln["within_budget"], ln["changed_since_call"])
for ln in lines] == [(1, 22, True, False), (2, 23, False, True)]
# ── against Postgres ─────────────────────────────────────────────────────
@@ -155,7 +185,8 @@ def _lines(*specs):
"""A stand-in re-run: (record_id, rank, within_budget)."""
return AsyncMock(return_value=[
{"record_id": rid, "rank": rank, "score": 0.9 - rank / 10,
"within_budget": within, "kind": "note", "name": f"n{rid}", "passage": "p"}
"within_budget": within, "changed_since_call": False,
"kind": "note", "name": f"n{rid}", "passage": "p"}
for rid, rank, within in specs
])