fix(retrieval): the review drops records written after the call it re-runs (#4773)
CI & Build / Python lint (push) Successful in 4s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / TypeScript typecheck (push) Successful in 54s
CI & Build / integration (push) Successful in 1m8s
CI & Build / Python tests (push) Successful in 1m53s
CI & Build / Build & push image (push) Successful in 36s
CI & Build / Python lint (push) Successful in 4s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / TypeScript typecheck (push) Successful in 54s
CI & Build / integration (push) Successful in 1m8s
CI & Build / Python tests (push) Successful in 1m53s
CI & Build / Build & push image (push) Successful in 36s
The first live sample ranked records the call could never have been offered: the session that made a call writes the decision, often quoting the message, and that record tops the re-run. Judged, it inflates on_point exactly where the budget is decided. - _rerun over-fetches by POSTDATED_SLACK, drops records created after the call before ranking, and names them in `postdated`. - each line carries `changed_since_call` (updated_at or a work log after the call) and `logged` (shown fresh then). - judge_menu shares the re-run, so a post-dated record cannot be judged. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -81,17 +81,22 @@ def test_the_re_run_searches_the_way_the_arm_does():
|
||||
assert rerun.get(key) == arm[key], key
|
||||
|
||||
|
||||
CALL = datetime(2026, 9, 20, 12, 0, tzinfo=timezone.utc)
|
||||
BEFORE = CALL - timedelta(days=1)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_the_re_run_ranks_from_one_and_marks_the_budget_cut():
|
||||
row = SimpleNamespace(query="q", threshold=0.55, limit_n=2, project_id=None)
|
||||
row = SimpleNamespace(query="q", threshold=0.55, limit_n=2, project_id=None,
|
||||
created_at=CALL)
|
||||
hits = [
|
||||
(0.81, fake_note(id=11, title="first")),
|
||||
(0.74, fake_note(id=12, title="second")),
|
||||
(0.66, fake_note(id=13, title="third")),
|
||||
(0.81, fake_note(id=11, title="first", created_at=BEFORE, updated_at=BEFORE)),
|
||||
(0.74, fake_note(id=12, title="second", created_at=BEFORE, updated_at=BEFORE)),
|
||||
(0.66, fake_note(id=13, title="third", created_at=BEFORE, updated_at=BEFORE)),
|
||||
]
|
||||
|
||||
async def search(user_id, query, **kw):
|
||||
assert kw["limit"] == 3 and kw["threshold"] == 0.55
|
||||
assert kw["limit"] == 3 + review.POSTDATED_SLACK and kw["threshold"] == 0.55
|
||||
assert kw["project_id"] is None
|
||||
kw["report"]["best_chunk"] = {12: {"text": "second\nthe passage that matched"}}
|
||||
return hits
|
||||
@@ -105,6 +110,31 @@ async def test_the_re_run_ranks_from_one_and_marks_the_budget_cut():
|
||||
assert lines[0]["passage"] is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_record_written_after_the_call_is_dropped_before_ranking():
|
||||
"""The session that made a call goes on to write about it, and that record
|
||||
outranks everything in a re-run. The call could never have been offered
|
||||
it, so it must not take a rank — or a verdict — from what was."""
|
||||
row = SimpleNamespace(query="q", threshold=0.55, limit_n=1, project_id=None,
|
||||
created_at=CALL)
|
||||
after = CALL + timedelta(hours=1)
|
||||
hits = [
|
||||
(0.90, fake_note(id=21, title="the decision", created_at=after, updated_at=after)),
|
||||
(0.80, fake_note(id=22, title="held then", created_at=BEFORE, updated_at=BEFORE)),
|
||||
(0.70, fake_note(id=23, title="edited since", created_at=BEFORE, updated_at=after)),
|
||||
]
|
||||
|
||||
async def search(user_id, query, **kw):
|
||||
return hits
|
||||
|
||||
rep: dict = {}
|
||||
with patch("scribe.services.embeddings.semantic_search_notes", side_effect=search):
|
||||
lines = await review._rerun(1, row, 2, report=rep)
|
||||
assert rep["postdated"] == [21]
|
||||
assert [(ln["rank"], ln["record_id"], ln["within_budget"], ln["changed_since_call"])
|
||||
for ln in lines] == [(1, 22, True, False), (2, 23, False, True)]
|
||||
|
||||
|
||||
# ── against Postgres ─────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -155,7 +185,8 @@ def _lines(*specs):
|
||||
"""A stand-in re-run: (record_id, rank, within_budget)."""
|
||||
return AsyncMock(return_value=[
|
||||
{"record_id": rid, "rank": rank, "score": 0.9 - rank / 10,
|
||||
"within_budget": within, "kind": "note", "name": f"n{rid}", "passage": "p"}
|
||||
"within_budget": within, "changed_since_call": False,
|
||||
"kind": "note", "name": f"n{rid}", "passage": "p"}
|
||||
for rid, rank, within in specs
|
||||
])
|
||||
|
||||
|
||||
Reference in New Issue
Block a user