Judging live auto-inject menus (#4772) found "## Work log — 2026-08-21", "## Findings worth carrying forward" and "Two dev→main PRs this session." each embedded as a whole chunk. Text about nothing embeds close to everything, so they took ranks 1–3 on vague queries and pushed real records down. Three ways the chunker made them, each closed: - _split_paragraphs flushed a heading on its own when the paragraph under it was long. A thin lead is now carried into the paragraph that follows, and a hard cut never lands in the first quarter of the budget, where the newline it finds closes a heading. - A heading with no body (above its ### parts) became a section of its own when the section before it was full. A thin section now takes the next. - A short lead-in or a short last section stood alone. The lead-in joins what follows; a thin tail joins what precedes it. A continuation piece now repeats the LAST heading before it, since one section can now hold several. CHUNKER_VERSION 2 → 3, so the startup backfill re-embeds; tuned dials will report their shape as changed, which is true. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
416 lines
16 KiB
Python
416 lines
16 KiB
Python
"""chunk_document — the document shape a record is embedded as (#280).
|
||
|
||
The model window is 512 tokens and fastembed truncates silently, so before
|
||
chunking, everything past ~400 words of a record was PERMANENTLY invisible to
|
||
semantic search. These tests pin the two halves of the fix's contract: short
|
||
records keep the exact historical shape (the corpus's sharpest vectors are
|
||
untouched), and long records lose NOTHING — every line of the body lands in
|
||
some chunk, each chunk inside the window budget, each carrying the title as
|
||
its topical anchor.
|
||
"""
|
||
import re
|
||
|
||
from scribe.services.embeddings import (
|
||
_CHUNK_CHAR_BUDGET,
|
||
_MIN_CHUNK_CONTENT,
|
||
chunk_document,
|
||
embedding_text,
|
||
)
|
||
|
||
|
||
def _long_section(tag: str, paragraphs: int = 6, sentence: str = None) -> str:
|
||
sentence = sentence or f"This paragraph discusses {tag} in useful detail."
|
||
para = " ".join([sentence] * 6)
|
||
return "\n\n".join(f"{para} (p{i})" for i in range(paragraphs))
|
||
|
||
|
||
# --- the identity half: short records are byte-for-byte unaffected -----------
|
||
|
||
|
||
def test_a_short_record_yields_exactly_the_historical_shape():
|
||
"""Snippets and reference notes are the sharpest records in the corpus
|
||
(#2485) precisely because of this shape — chunking must not touch them."""
|
||
assert chunk_document("A title", "A short body") == [
|
||
embedding_text("A title", "A short body")
|
||
]
|
||
|
||
|
||
def test_a_bodyless_record_is_one_chunk_of_its_title():
|
||
assert chunk_document("Just a title", "") == ["Just a title"]
|
||
|
||
|
||
def test_an_empty_record_yields_no_chunks():
|
||
"""Callers gate on falsiness to skip embedding entirely."""
|
||
assert chunk_document("", "") == []
|
||
assert chunk_document(None, None) == []
|
||
|
||
|
||
def test_a_record_exactly_at_budget_stays_whole():
|
||
body = "x" * (_CHUNK_CHAR_BUDGET - len("T\n"))
|
||
assert chunk_document("T", body) == [embedding_text("T", body)]
|
||
|
||
|
||
# --- the no-lost-data half: this is what the build is FOR --------------------
|
||
|
||
|
||
def test_every_line_of_a_long_body_lands_in_some_chunk():
|
||
"""The point of #280. Before chunking, a 2,000-word dev-log's last three
|
||
quarters could not influence retrieval at all. Nothing may be dropped."""
|
||
sections = [
|
||
f"## Topic {i}\n\n{_long_section(f'topic-{i}')}" for i in range(8)
|
||
]
|
||
body = "Intro paragraph before any heading.\n\n" + "\n\n".join(sections)
|
||
chunks = chunk_document("A very long dev-log", body)
|
||
|
||
assert len(chunks) > 1
|
||
joined = "\n".join(chunks)
|
||
for line in body.splitlines():
|
||
if line.strip():
|
||
assert line.strip() in joined, f"content dropped: {line[:60]!r}"
|
||
|
||
|
||
def test_every_chunk_fits_the_window_budget():
|
||
body = "\n\n".join(_long_section(f"t{i}") for i in range(10))
|
||
for chunk in chunk_document("T", body):
|
||
assert len(chunk) <= _CHUNK_CHAR_BUDGET + len("T") + 1
|
||
|
||
|
||
def test_every_chunk_is_anchored_by_the_title():
|
||
"""Each vector must carry its own topical anchor — the property that makes
|
||
snippets discriminative. An unanchored mid-document chunk would embed as
|
||
free-floating prose about nothing in particular."""
|
||
body = "\n\n".join(
|
||
f"## Section {i}\n\n{_long_section(f'sec-{i}')}" for i in range(6)
|
||
)
|
||
chunks = chunk_document("Retrieval reference", body)
|
||
assert len(chunks) > 1
|
||
for chunk in chunks:
|
||
assert chunk.startswith("Retrieval reference\n")
|
||
|
||
|
||
# --- boundary behaviour ------------------------------------------------------
|
||
|
||
|
||
def test_sections_split_at_markdown_headings_and_stay_whole_when_they_fit():
|
||
a = "## Alpha\n\nShort alpha content."
|
||
b = "## Beta\n\n" + _long_section("beta")
|
||
c = "## Gamma\n\n" + _long_section("gamma")
|
||
chunks = chunk_document("T", f"{a}\n\n{b}\n\n{c}")
|
||
# Beta's content never shares a chunk with Gamma's heading-onward content:
|
||
# heading boundaries are chunk boundaries unless merging small sections.
|
||
for chunk in chunks:
|
||
assert not ("(p5)" in chunk and "## Gamma" in chunk and "beta" in chunk)
|
||
|
||
|
||
def test_small_adjacent_sections_merge_instead_of_each_spending_a_vector():
|
||
body = (
|
||
"\n\n".join(f"## S{i}\n\nTiny." for i in range(4))
|
||
+ "\n\n## Big\n\n"
|
||
+ _long_section("big", paragraphs=10)
|
||
)
|
||
chunks = chunk_document("T", body)
|
||
tiny_chunks = [c for c in chunks if "Tiny." in c]
|
||
assert len(tiny_chunks) == 1, "four tiny sections should share one chunk"
|
||
|
||
|
||
def test_a_heading_inside_a_code_fence_does_not_split():
|
||
"""A commented `# step` in a recorded shell snippet is content, not
|
||
structure."""
|
||
body = "Intro.\n\n```bash\n# not a heading\necho hi\n```\n\nOutro."
|
||
chunks = chunk_document("T", body)
|
||
assert chunks == [embedding_text("T", body)]
|
||
|
||
|
||
def test_pieces_subsplit_from_one_section_repeat_its_heading():
|
||
"""'Which part of which topic' must survive the split — a continuation
|
||
piece without its heading embeds as context-free prose."""
|
||
body = "## The Only Topic\n\n" + _long_section("only", paragraphs=40)
|
||
chunks = chunk_document("T", body)
|
||
assert len(chunks) > 1
|
||
for chunk in chunks:
|
||
assert "## The Only Topic" in chunk
|
||
|
||
|
||
def test_a_monster_single_paragraph_is_hard_split_not_dropped():
|
||
body = "word " * 2000 # one paragraph, no newlines to split at
|
||
chunks = chunk_document("T", body)
|
||
assert len(chunks) > 1
|
||
total_words = sum(chunk.count("word") for chunk in chunks)
|
||
assert total_words == 2000
|
||
|
||
|
||
# --- no chunk is a heading alone (#4784) -------------------------------------
|
||
#
|
||
# Found by judging live auto-inject menus: "## Work log — 2026-08-21",
|
||
# "## Findings worth carrying forward" and "Two dev→main PRs this session."
|
||
# were each a whole chunk, and took ranks 1–3 on vague queries — text about
|
||
# nothing embeds close to everything. Each case below is one of those shapes.
|
||
|
||
_HEADING = re.compile(r"^#{1,6}\s")
|
||
|
||
|
||
def _said(chunk: str, title: str) -> str:
|
||
"""A chunk's text once its title prefix and heading lines are set aside."""
|
||
body = chunk[len(title) + 1:] if chunk.startswith(title + "\n") else chunk
|
||
return "\n".join(ln for ln in body.splitlines() if not _HEADING.match(ln)).strip()
|
||
|
||
|
||
def _assert_nothing_thin_and_nothing_lost(title: str, body: str) -> list[str]:
|
||
chunks = chunk_document(title, body)
|
||
assert len(chunks) > 1, "the case must be long enough to split"
|
||
for chunk in chunks:
|
||
assert len(_said(chunk, title)) >= _MIN_CHUNK_CONTENT, (
|
||
f"a chunk says almost nothing: {chunk[:120]!r}")
|
||
assert len(chunk) <= _CHUNK_CHAR_BUDGET + len(title) + 1 + 80
|
||
joined = "\n".join(chunks)
|
||
for line in (ln.strip() for ln in body.splitlines()):
|
||
# A line longer than a chunk is cut across two, so only its ends can
|
||
# be looked for whole.
|
||
for part in ([line] if len(line) < 400 else [line[:60], line[-60:]]):
|
||
assert part in joined, f"content dropped: {part[:60]!r}"
|
||
return chunks
|
||
|
||
|
||
def test_a_heading_over_one_long_paragraph_is_not_a_chunk_of_its_own():
|
||
"""A work log whose entry is one long paragraph: the split used to flush
|
||
the heading as its own piece before cutting the paragraph."""
|
||
entry = " ".join(["The deploy was verified against the live instance."] * 40)
|
||
_assert_nothing_thin_and_nothing_lost(
|
||
"Step 6 — webhook drift-flagging",
|
||
"The step body.\n\n## Work log — 2026-08-16\n\n" + entry,
|
||
)
|
||
|
||
|
||
def test_a_heading_with_nothing_under_it_joins_what_follows():
|
||
"""'## Findings' directly above its '### …' parts is a section with no
|
||
body, and stood alone whenever the section before it was full."""
|
||
body = (
|
||
# Five paragraphs fill the section before it, so the heading cannot
|
||
# merge backwards and has to be carried forward.
|
||
"## Context\n\n" + _long_section("context", paragraphs=5)
|
||
+ "\n\n## Findings worth carrying forward\n\n"
|
||
+ "### One\n\n" + _long_section("one", paragraphs=5)
|
||
+ "\n\n### Two\n\n" + _long_section("two", paragraphs=5)
|
||
)
|
||
_assert_nothing_thin_and_nothing_lost("A milestone dev-log", body)
|
||
|
||
|
||
def test_a_lone_lead_in_line_joins_the_section_after_it():
|
||
_assert_nothing_thin_and_nothing_lost(
|
||
"2026-06-10 — UI batch",
|
||
"Two dev→main PRs this session.\n\n## PR 89\n\n"
|
||
+ _long_section("pr89", paragraphs=8),
|
||
)
|
||
|
||
|
||
def test_a_thin_last_section_joins_the_one_before_it():
|
||
_assert_nothing_thin_and_nothing_lost(
|
||
"T",
|
||
"## Body\n\n" + _long_section("body", paragraphs=8) + "\n\n## Next\n\nTBD.",
|
||
)
|
||
|
||
|
||
def test_a_continuation_piece_repeats_the_heading_it_sits_under():
|
||
"""A thin heading merged forward puts two headings in one section; the
|
||
pieces of the second must carry the second, not the first."""
|
||
body = "## Parent\n\n## Child\n\n" + _long_section("child", paragraphs=40)
|
||
chunks = chunk_document("T", body)
|
||
assert len(chunks) > 2
|
||
for chunk in chunks[1:]:
|
||
assert "## Child" in chunk
|
||
|
||
|
||
# --- the read path: best chunk wins (#280 step 4) ----------------------------
|
||
|
||
|
||
async def test_search_collapses_chunk_rows_to_best_chunk_per_note():
|
||
"""Rows arrive at CHUNK grain ordered by distance; a note appearing via
|
||
several chunks must come back ONCE, scored by its best chunk — otherwise a
|
||
long record fills the top-k with copies of itself."""
|
||
from unittest.mock import AsyncMock, MagicMock, patch
|
||
|
||
from scribe.services import embeddings as emb
|
||
|
||
note_a, note_b = MagicMock(id=1), MagicMock(id=2)
|
||
# Rows are (Note, distance, chunk_index, chunk_text) — the chunk columns
|
||
# ride along so the collapse can report WHICH passage won (#4243).
|
||
rows = [
|
||
(note_a, 0.10, 3, "the passage that actually matched"),
|
||
(note_b, 0.20, 0, "b's best"),
|
||
(note_a, 0.25, 7, "a worse chunk of a"),
|
||
(note_a, 0.30, 1, "a worse chunk of a"),
|
||
]
|
||
result = MagicMock()
|
||
result.all.return_value = rows
|
||
session, ctx = _session_ctx()
|
||
session.execute = AsyncMock(return_value=result)
|
||
|
||
report: dict = {}
|
||
with (
|
||
patch.object(emb, "async_session", return_value=ctx),
|
||
patch.object(emb, "get_embedding", AsyncMock(return_value=[0.0] * 384)),
|
||
):
|
||
out = await emb.semantic_search_notes(
|
||
1, "a query", limit=8, demote_superseded=False, report=report
|
||
)
|
||
|
||
assert [note.id for _s, note in out] == [1, 2]
|
||
assert out[0][0] == 1.0 - 0.10 # the BEST chunk's score, not a later one
|
||
|
||
# And the winning chunk is reported, not merely used for scoring. Without
|
||
# this a caller can only preview the head of the body — a span this query
|
||
# has already ranked lower than the one that won (#4243).
|
||
assert report["best_chunk"][1] == {
|
||
"index": 3, "text": "the passage that actually matched",
|
||
}
|
||
assert report["best_chunk"][2]["index"] == 0
|
||
|
||
|
||
# --- the write path: one row per chunk (#280 step 3) -------------------------
|
||
|
||
|
||
def _session_ctx(log_rows=()):
|
||
"""A session stand-in. `log_rows` answers the work-log read a task's
|
||
document now needs (#4251) — answered explicitly rather than left to
|
||
autovivify, because `list(result.all())` on a bare MagicMock raises and is
|
||
swallowed, which would quietly make every one of these a no-log test."""
|
||
from unittest.mock import AsyncMock, MagicMock
|
||
|
||
session = MagicMock()
|
||
result = MagicMock()
|
||
result.all.return_value = list(log_rows)
|
||
session.execute = AsyncMock(return_value=result)
|
||
session.commit = AsyncMock()
|
||
ctx = MagicMock()
|
||
ctx.__aenter__ = AsyncMock(return_value=session)
|
||
ctx.__aexit__ = AsyncMock(return_value=False)
|
||
return session, ctx
|
||
|
||
|
||
async def test_upsert_stores_one_versioned_row_per_chunk():
|
||
from unittest.mock import AsyncMock, patch
|
||
|
||
from scribe.services import embeddings as emb
|
||
|
||
body = "\n\n".join(
|
||
f"## Section {i}\n\n{_long_section(f'sec-{i}')}" for i in range(6)
|
||
)
|
||
chunks = chunk_document("T", body)
|
||
assert len(chunks) > 1
|
||
|
||
session, ctx = _session_ctx()
|
||
with (
|
||
patch.object(emb, "async_session", return_value=ctx),
|
||
patch.object(
|
||
emb, "get_embeddings",
|
||
AsyncMock(return_value=[[0.0] * 384 for _ in chunks]),
|
||
),
|
||
):
|
||
await emb.upsert_note_embedding(7, 42, "T", body)
|
||
|
||
rows = [call.args[0] for call in session.add.call_args_list]
|
||
assert [r.chunk_index for r in rows] == list(range(len(chunks)))
|
||
assert [r.chunk_text for r in rows] == chunks
|
||
assert {r.chunker_version for r in rows} == {emb.CHUNKER_VERSION}
|
||
assert {r.embedding_model for r in rows} == {emb.EMBEDDING_MODEL}
|
||
assert {r.user_id for r in rows} == {42}
|
||
session.execute.assert_awaited() # the delete that makes replacement atomic
|
||
|
||
|
||
async def test_a_tasks_work_logs_reach_the_rows_it_is_stored_as():
|
||
"""The wiring half of #4251: the shaper is pure and tested next door, so
|
||
what is pinned here is that the WRITER actually asks for the logs and
|
||
embeds what comes back. A log that is written, readable (#4241) and absent
|
||
from the index is the same half-surface one layer down."""
|
||
import datetime
|
||
from unittest.mock import AsyncMock, patch
|
||
|
||
from scribe.services import embeddings as emb
|
||
|
||
session, ctx = _session_ctx(
|
||
log_rows=[(datetime.datetime(2026, 9, 20), "ruled out the cache theory")]
|
||
)
|
||
with (
|
||
patch.object(emb, "async_session", return_value=ctx),
|
||
patch.object(emb, "get_embeddings", AsyncMock(return_value=[[0.0] * 384])),
|
||
):
|
||
await emb.upsert_note_embedding(7, 42, "A task", "Its own prose.")
|
||
|
||
rows = [call.args[0] for call in session.add.call_args_list]
|
||
stored = "\n".join(r.chunk_text for r in rows)
|
||
assert "ruled out the cache theory" in stored
|
||
assert "Its own prose." in stored
|
||
assert emb.WORK_LOG_HEADING in stored
|
||
|
||
|
||
async def test_upsert_of_an_emptied_record_clears_rows_instead_of_embedding():
|
||
"""An empty embedding is worse than none, and a STALE one is worse than
|
||
that — a record emptied of content must stop being findable by what it no
|
||
longer says."""
|
||
from unittest.mock import patch
|
||
|
||
from scribe.services import embeddings as emb
|
||
|
||
session, ctx = _session_ctx()
|
||
with (
|
||
patch.object(emb, "async_session", return_value=ctx),
|
||
patch.object(emb, "get_embeddings") as embedder,
|
||
):
|
||
await emb.upsert_note_embedding(7, 42, "", "")
|
||
|
||
embedder.assert_not_called()
|
||
session.execute.assert_awaited() # the delete
|
||
session.add.assert_not_called()
|
||
session.commit.assert_awaited()
|
||
|
||
|
||
async def test_backfill_reembeds_notes_with_a_stale_chunker_version():
|
||
"""The reason chunker_version exists: a shape change becomes a version
|
||
bump that re-embeds exactly the stale notes, instead of another 0077-style
|
||
table wipe. Only rows AT the current version count as done."""
|
||
from unittest.mock import AsyncMock, MagicMock, patch
|
||
|
||
from scribe.services import embeddings as emb
|
||
|
||
def _ids(*values):
|
||
r = MagicMock()
|
||
r.fetchall.return_value = [(v,) for v in values]
|
||
return r
|
||
|
||
session, ctx = _session_ctx()
|
||
# The opening scan, in order: ids already at the current version, every
|
||
# note id, vectors older than their note, tasks logged since their vectors.
|
||
# IDS ONLY — the text is re-read per note at embed time (#4262).
|
||
session.execute = AsyncMock(side_effect=[
|
||
_ids(1), # note 1 is current
|
||
_ids(1, 2), # the corpus
|
||
_ids(), # nothing stale by timestamp
|
||
_ids(), # no task logged since its vectors
|
||
])
|
||
|
||
with (
|
||
patch.object(emb, "async_session", return_value=ctx),
|
||
patch.object(emb, "_current_row",
|
||
AsyncMock(return_value=(42, "stale-version", "body", "note", None))),
|
||
patch.object(emb, "upsert_note_embedding", AsyncMock()) as upsert,
|
||
patch.object(emb.asyncio, "sleep", AsyncMock()),
|
||
):
|
||
await emb.backfill_note_embeddings()
|
||
|
||
embedded = [call.args[0] for call in upsert.call_args_list]
|
||
assert embedded == [2], "only the stale note is re-embedded"
|
||
|
||
|
||
def test_a_row_is_current_only_in_the_live_models_space():
|
||
"""#4132: the column is a width, not an identity, so a same-width model
|
||
swap writes a second geometry with no error. The backfill's "current"
|
||
predicate has to name BOTH halves of the stamp, on every embedding table."""
|
||
from scribe.models.embedding import (
|
||
MilestoneEmbedding, NoteEmbedding, RuleEmbedding, SystemEmbedding,
|
||
)
|
||
from scribe.services import embeddings as emb
|
||
|
||
for table in (NoteEmbedding, RuleEmbedding, MilestoneEmbedding, SystemEmbedding):
|
||
clause = str(emb.is_current_stamp(table))
|
||
assert "chunker_version" in clause and "embedding_model" in clause, table
|