Files
FabledScribe/tests/test_chunking.py
T
bvandeusenandClaude Opus 5.5 78fa01b025 fix(embeddings): no chunk is a heading alone (#4784)
Judging live auto-inject menus (#4772) found "## Work log — 2026-08-21",
"## Findings worth carrying forward" and "Two dev→main PRs this session."
each embedded as a whole chunk. Text about nothing embeds close to
everything, so they took ranks 1–3 on vague queries and pushed real records
down.

Three ways the chunker made them, each closed:
- _split_paragraphs flushed a heading on its own when the paragraph under it
  was long. A thin lead is now carried into the paragraph that follows, and
  a hard cut never lands in the first quarter of the budget, where the
  newline it finds closes a heading.
- A heading with no body (above its ### parts) became a section of its own
  when the section before it was full. A thin section now takes the next.
- A short lead-in or a short last section stood alone. The lead-in joins
  what follows; a thin tail joins what precedes it.

A continuation piece now repeats the LAST heading before it, since one
section can now hold several. CHUNKER_VERSION 2 → 3, so the startup backfill
re-embeds; tuned dials will report their shape as changed, which is true.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-03 21:25:37 -04:00

416 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""chunk_document — the document shape a record is embedded as (#280).
The model window is 512 tokens and fastembed truncates silently, so before
chunking, everything past ~400 words of a record was PERMANENTLY invisible to
semantic search. These tests pin the two halves of the fix's contract: short
records keep the exact historical shape (the corpus's sharpest vectors are
untouched), and long records lose NOTHING — every line of the body lands in
some chunk, each chunk inside the window budget, each carrying the title as
its topical anchor.
"""
import re
from scribe.services.embeddings import (
_CHUNK_CHAR_BUDGET,
_MIN_CHUNK_CONTENT,
chunk_document,
embedding_text,
)
def _long_section(tag: str, paragraphs: int = 6, sentence: str = None) -> str:
sentence = sentence or f"This paragraph discusses {tag} in useful detail."
para = " ".join([sentence] * 6)
return "\n\n".join(f"{para} (p{i})" for i in range(paragraphs))
# --- the identity half: short records are byte-for-byte unaffected -----------
def test_a_short_record_yields_exactly_the_historical_shape():
"""Snippets and reference notes are the sharpest records in the corpus
(#2485) precisely because of this shape — chunking must not touch them."""
assert chunk_document("A title", "A short body") == [
embedding_text("A title", "A short body")
]
def test_a_bodyless_record_is_one_chunk_of_its_title():
assert chunk_document("Just a title", "") == ["Just a title"]
def test_an_empty_record_yields_no_chunks():
"""Callers gate on falsiness to skip embedding entirely."""
assert chunk_document("", "") == []
assert chunk_document(None, None) == []
def test_a_record_exactly_at_budget_stays_whole():
body = "x" * (_CHUNK_CHAR_BUDGET - len("T\n"))
assert chunk_document("T", body) == [embedding_text("T", body)]
# --- the no-lost-data half: this is what the build is FOR --------------------
def test_every_line_of_a_long_body_lands_in_some_chunk():
"""The point of #280. Before chunking, a 2,000-word dev-log's last three
quarters could not influence retrieval at all. Nothing may be dropped."""
sections = [
f"## Topic {i}\n\n{_long_section(f'topic-{i}')}" for i in range(8)
]
body = "Intro paragraph before any heading.\n\n" + "\n\n".join(sections)
chunks = chunk_document("A very long dev-log", body)
assert len(chunks) > 1
joined = "\n".join(chunks)
for line in body.splitlines():
if line.strip():
assert line.strip() in joined, f"content dropped: {line[:60]!r}"
def test_every_chunk_fits_the_window_budget():
body = "\n\n".join(_long_section(f"t{i}") for i in range(10))
for chunk in chunk_document("T", body):
assert len(chunk) <= _CHUNK_CHAR_BUDGET + len("T") + 1
def test_every_chunk_is_anchored_by_the_title():
"""Each vector must carry its own topical anchor — the property that makes
snippets discriminative. An unanchored mid-document chunk would embed as
free-floating prose about nothing in particular."""
body = "\n\n".join(
f"## Section {i}\n\n{_long_section(f'sec-{i}')}" for i in range(6)
)
chunks = chunk_document("Retrieval reference", body)
assert len(chunks) > 1
for chunk in chunks:
assert chunk.startswith("Retrieval reference\n")
# --- boundary behaviour ------------------------------------------------------
def test_sections_split_at_markdown_headings_and_stay_whole_when_they_fit():
a = "## Alpha\n\nShort alpha content."
b = "## Beta\n\n" + _long_section("beta")
c = "## Gamma\n\n" + _long_section("gamma")
chunks = chunk_document("T", f"{a}\n\n{b}\n\n{c}")
# Beta's content never shares a chunk with Gamma's heading-onward content:
# heading boundaries are chunk boundaries unless merging small sections.
for chunk in chunks:
assert not ("(p5)" in chunk and "## Gamma" in chunk and "beta" in chunk)
def test_small_adjacent_sections_merge_instead_of_each_spending_a_vector():
body = (
"\n\n".join(f"## S{i}\n\nTiny." for i in range(4))
+ "\n\n## Big\n\n"
+ _long_section("big", paragraphs=10)
)
chunks = chunk_document("T", body)
tiny_chunks = [c for c in chunks if "Tiny." in c]
assert len(tiny_chunks) == 1, "four tiny sections should share one chunk"
def test_a_heading_inside_a_code_fence_does_not_split():
"""A commented `# step` in a recorded shell snippet is content, not
structure."""
body = "Intro.\n\n```bash\n# not a heading\necho hi\n```\n\nOutro."
chunks = chunk_document("T", body)
assert chunks == [embedding_text("T", body)]
def test_pieces_subsplit_from_one_section_repeat_its_heading():
"""'Which part of which topic' must survive the split — a continuation
piece without its heading embeds as context-free prose."""
body = "## The Only Topic\n\n" + _long_section("only", paragraphs=40)
chunks = chunk_document("T", body)
assert len(chunks) > 1
for chunk in chunks:
assert "## The Only Topic" in chunk
def test_a_monster_single_paragraph_is_hard_split_not_dropped():
body = "word " * 2000 # one paragraph, no newlines to split at
chunks = chunk_document("T", body)
assert len(chunks) > 1
total_words = sum(chunk.count("word") for chunk in chunks)
assert total_words == 2000
# --- no chunk is a heading alone (#4784) -------------------------------------
#
# Found by judging live auto-inject menus: "## Work log — 2026-08-21",
# "## Findings worth carrying forward" and "Two dev→main PRs this session."
# were each a whole chunk, and took ranks 1–3 on vague queries — text about
# nothing embeds close to everything. Each case below is one of those shapes.
_HEADING = re.compile(r"^#{1,6}\s")
def _said(chunk: str, title: str) -> str:
"""A chunk's text once its title prefix and heading lines are set aside."""
body = chunk[len(title) + 1:] if chunk.startswith(title + "\n") else chunk
return "\n".join(ln for ln in body.splitlines() if not _HEADING.match(ln)).strip()
def _assert_nothing_thin_and_nothing_lost(title: str, body: str) -> list[str]:
chunks = chunk_document(title, body)
assert len(chunks) > 1, "the case must be long enough to split"
for chunk in chunks:
assert len(_said(chunk, title)) >= _MIN_CHUNK_CONTENT, (
f"a chunk says almost nothing: {chunk[:120]!r}")
assert len(chunk) <= _CHUNK_CHAR_BUDGET + len(title) + 1 + 80
joined = "\n".join(chunks)
for line in (ln.strip() for ln in body.splitlines()):
# A line longer than a chunk is cut across two, so only its ends can
# be looked for whole.
for part in ([line] if len(line) < 400 else [line[:60], line[-60:]]):
assert part in joined, f"content dropped: {part[:60]!r}"
return chunks
def test_a_heading_over_one_long_paragraph_is_not_a_chunk_of_its_own():
"""A work log whose entry is one long paragraph: the split used to flush
the heading as its own piece before cutting the paragraph."""
entry = " ".join(["The deploy was verified against the live instance."] * 40)
_assert_nothing_thin_and_nothing_lost(
"Step 6 — webhook drift-flagging",
"The step body.\n\n## Work log — 2026-08-16\n\n" + entry,
)
def test_a_heading_with_nothing_under_it_joins_what_follows():
"""'## Findings' directly above its '### …' parts is a section with no
body, and stood alone whenever the section before it was full."""
body = (
# Five paragraphs fill the section before it, so the heading cannot
# merge backwards and has to be carried forward.
"## Context\n\n" + _long_section("context", paragraphs=5)
+ "\n\n## Findings worth carrying forward\n\n"
+ "### One\n\n" + _long_section("one", paragraphs=5)
+ "\n\n### Two\n\n" + _long_section("two", paragraphs=5)
)
_assert_nothing_thin_and_nothing_lost("A milestone dev-log", body)
def test_a_lone_lead_in_line_joins_the_section_after_it():
_assert_nothing_thin_and_nothing_lost(
"2026-06-10 — UI batch",
"Two dev→main PRs this session.\n\n## PR 89\n\n"
+ _long_section("pr89", paragraphs=8),
)
def test_a_thin_last_section_joins_the_one_before_it():
_assert_nothing_thin_and_nothing_lost(
"T",
"## Body\n\n" + _long_section("body", paragraphs=8) + "\n\n## Next\n\nTBD.",
)
def test_a_continuation_piece_repeats_the_heading_it_sits_under():
"""A thin heading merged forward puts two headings in one section; the
pieces of the second must carry the second, not the first."""
body = "## Parent\n\n## Child\n\n" + _long_section("child", paragraphs=40)
chunks = chunk_document("T", body)
assert len(chunks) > 2
for chunk in chunks[1:]:
assert "## Child" in chunk
# --- the read path: best chunk wins (#280 step 4) ----------------------------
async def test_search_collapses_chunk_rows_to_best_chunk_per_note():
"""Rows arrive at CHUNK grain ordered by distance; a note appearing via
several chunks must come back ONCE, scored by its best chunk — otherwise a
long record fills the top-k with copies of itself."""
from unittest.mock import AsyncMock, MagicMock, patch
from scribe.services import embeddings as emb
note_a, note_b = MagicMock(id=1), MagicMock(id=2)
# Rows are (Note, distance, chunk_index, chunk_text) — the chunk columns
# ride along so the collapse can report WHICH passage won (#4243).
rows = [
(note_a, 0.10, 3, "the passage that actually matched"),
(note_b, 0.20, 0, "b's best"),
(note_a, 0.25, 7, "a worse chunk of a"),
(note_a, 0.30, 1, "a worse chunk of a"),
]
result = MagicMock()
result.all.return_value = rows
session, ctx = _session_ctx()
session.execute = AsyncMock(return_value=result)
report: dict = {}
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(emb, "get_embedding", AsyncMock(return_value=[0.0] * 384)),
):
out = await emb.semantic_search_notes(
1, "a query", limit=8, demote_superseded=False, report=report
)
assert [note.id for _s, note in out] == [1, 2]
assert out[0][0] == 1.0 - 0.10 # the BEST chunk's score, not a later one
# And the winning chunk is reported, not merely used for scoring. Without
# this a caller can only preview the head of the body — a span this query
# has already ranked lower than the one that won (#4243).
assert report["best_chunk"][1] == {
"index": 3, "text": "the passage that actually matched",
}
assert report["best_chunk"][2]["index"] == 0
# --- the write path: one row per chunk (#280 step 3) -------------------------
def _session_ctx(log_rows=()):
"""A session stand-in. `log_rows` answers the work-log read a task's
document now needs (#4251) — answered explicitly rather than left to
autovivify, because `list(result.all())` on a bare MagicMock raises and is
swallowed, which would quietly make every one of these a no-log test."""
from unittest.mock import AsyncMock, MagicMock
session = MagicMock()
result = MagicMock()
result.all.return_value = list(log_rows)
session.execute = AsyncMock(return_value=result)
session.commit = AsyncMock()
ctx = MagicMock()
ctx.__aenter__ = AsyncMock(return_value=session)
ctx.__aexit__ = AsyncMock(return_value=False)
return session, ctx
async def test_upsert_stores_one_versioned_row_per_chunk():
from unittest.mock import AsyncMock, patch
from scribe.services import embeddings as emb
body = "\n\n".join(
f"## Section {i}\n\n{_long_section(f'sec-{i}')}" for i in range(6)
)
chunks = chunk_document("T", body)
assert len(chunks) > 1
session, ctx = _session_ctx()
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(
emb, "get_embeddings",
AsyncMock(return_value=[[0.0] * 384 for _ in chunks]),
),
):
await emb.upsert_note_embedding(7, 42, "T", body)
rows = [call.args[0] for call in session.add.call_args_list]
assert [r.chunk_index for r in rows] == list(range(len(chunks)))
assert [r.chunk_text for r in rows] == chunks
assert {r.chunker_version for r in rows} == {emb.CHUNKER_VERSION}
assert {r.embedding_model for r in rows} == {emb.EMBEDDING_MODEL}
assert {r.user_id for r in rows} == {42}
session.execute.assert_awaited() # the delete that makes replacement atomic
async def test_a_tasks_work_logs_reach_the_rows_it_is_stored_as():
"""The wiring half of #4251: the shaper is pure and tested next door, so
what is pinned here is that the WRITER actually asks for the logs and
embeds what comes back. A log that is written, readable (#4241) and absent
from the index is the same half-surface one layer down."""
import datetime
from unittest.mock import AsyncMock, patch
from scribe.services import embeddings as emb
session, ctx = _session_ctx(
log_rows=[(datetime.datetime(2026, 9, 20), "ruled out the cache theory")]
)
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(emb, "get_embeddings", AsyncMock(return_value=[[0.0] * 384])),
):
await emb.upsert_note_embedding(7, 42, "A task", "Its own prose.")
rows = [call.args[0] for call in session.add.call_args_list]
stored = "\n".join(r.chunk_text for r in rows)
assert "ruled out the cache theory" in stored
assert "Its own prose." in stored
assert emb.WORK_LOG_HEADING in stored
async def test_upsert_of_an_emptied_record_clears_rows_instead_of_embedding():
"""An empty embedding is worse than none, and a STALE one is worse than
that — a record emptied of content must stop being findable by what it no
longer says."""
from unittest.mock import patch
from scribe.services import embeddings as emb
session, ctx = _session_ctx()
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(emb, "get_embeddings") as embedder,
):
await emb.upsert_note_embedding(7, 42, "", "")
embedder.assert_not_called()
session.execute.assert_awaited() # the delete
session.add.assert_not_called()
session.commit.assert_awaited()
async def test_backfill_reembeds_notes_with_a_stale_chunker_version():
"""The reason chunker_version exists: a shape change becomes a version
bump that re-embeds exactly the stale notes, instead of another 0077-style
table wipe. Only rows AT the current version count as done."""
from unittest.mock import AsyncMock, MagicMock, patch
from scribe.services import embeddings as emb
def _ids(*values):
r = MagicMock()
r.fetchall.return_value = [(v,) for v in values]
return r
session, ctx = _session_ctx()
# The opening scan, in order: ids already at the current version, every
# note id, vectors older than their note, tasks logged since their vectors.
# IDS ONLY — the text is re-read per note at embed time (#4262).
session.execute = AsyncMock(side_effect=[
_ids(1), # note 1 is current
_ids(1, 2), # the corpus
_ids(), # nothing stale by timestamp
_ids(), # no task logged since its vectors
])
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(emb, "_current_row",
AsyncMock(return_value=(42, "stale-version", "body", "note", None))),
patch.object(emb, "upsert_note_embedding", AsyncMock()) as upsert,
patch.object(emb.asyncio, "sleep", AsyncMock()),
):
await emb.backfill_note_embeddings()
embedded = [call.args[0] for call in upsert.call_args_list]
assert embedded == [2], "only the stale note is re-embedded"
def test_a_row_is_current_only_in_the_live_models_space():
"""#4132: the column is a width, not an identity, so a same-width model
swap writes a second geometry with no error. The backfill's "current"
predicate has to name BOTH halves of the stamp, on every embedding table."""
from scribe.models.embedding import (
MilestoneEmbedding, NoteEmbedding, RuleEmbedding, SystemEmbedding,
)
from scribe.services import embeddings as emb
for table in (NoteEmbedding, RuleEmbedding, MilestoneEmbedding, SystemEmbedding):
clause = str(emb.is_current_stamp(table))
assert "chunker_version" in clause and "embedding_model" in clause, table