Files
FabledScribe/tests/test_chunking.py
T
bvandeusenandClaude Opus 5 6abedb0168
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 11s
CI & Build / integration (push) Successful in 47s
CI & Build / TypeScript typecheck (push) Successful in 56s
CI & Build / Python tests (push) Successful in 1m36s
CI & Build / Build & push image (push) Successful in 30s
test: two tests that had to move with the chunk-carrying select (#4243)
test_chunking::test_search_collapses_chunk_rows_to_best_chunk_per_note feeds
rows straight into the real semantic_search_notes, so widening the select to
carry chunk_index/chunk_text broke its 2-tuple fakes. I had claimed no unit
test did this after grepping test_embeddings.py, which was the wrong file.
Rows are 4-tuples now, and the test additionally asserts the winning chunk is
REPORTED and not merely used for scoring — the property #4243 exists for, and
this is where it belongs, beside the collapse it comes from.

test_task_work_log_surface's ACL test asserted "task_logs.user_id" was absent
from the compiled statement. It is present in every SELECT as a projected
column; the claim was about the WHERE clause. Asserts on stmt.whereclause now,
which is what was actually meant and still fails if an owner filter returns.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy
2026-09-21 08:54:15 -04:00

277 lines
10 KiB
Python

"""chunk_document — the document shape a record is embedded as (#280).
The model window is 512 tokens and fastembed truncates silently, so before
chunking, everything past ~400 words of a record was PERMANENTLY invisible to
semantic search. These tests pin the two halves of the fix's contract: short
records keep the exact historical shape (the corpus's sharpest vectors are
untouched), and long records lose NOTHING — every line of the body lands in
some chunk, each chunk inside the window budget, each carrying the title as
its topical anchor.
"""
from scribe.services.embeddings import (
_CHUNK_CHAR_BUDGET,
chunk_document,
embedding_text,
)
def _long_section(tag: str, paragraphs: int = 6, sentence: str = None) -> str:
sentence = sentence or f"This paragraph discusses {tag} in useful detail."
para = " ".join([sentence] * 6)
return "\n\n".join(f"{para} (p{i})" for i in range(paragraphs))
# --- the identity half: short records are byte-for-byte unaffected -----------
def test_a_short_record_yields_exactly_the_historical_shape():
"""Snippets and reference notes are the sharpest records in the corpus
(#2485) precisely because of this shape — chunking must not touch them."""
assert chunk_document("A title", "A short body") == [
embedding_text("A title", "A short body")
]
def test_a_bodyless_record_is_one_chunk_of_its_title():
assert chunk_document("Just a title", "") == ["Just a title"]
def test_an_empty_record_yields_no_chunks():
"""Callers gate on falsiness to skip embedding entirely."""
assert chunk_document("", "") == []
assert chunk_document(None, None) == []
def test_a_record_exactly_at_budget_stays_whole():
body = "x" * (_CHUNK_CHAR_BUDGET - len("T\n"))
assert chunk_document("T", body) == [embedding_text("T", body)]
# --- the no-lost-data half: this is what the build is FOR --------------------
def test_every_line_of_a_long_body_lands_in_some_chunk():
"""The point of #280. Before chunking, a 2,000-word dev-log's last three
quarters could not influence retrieval at all. Nothing may be dropped."""
sections = [
f"## Topic {i}\n\n{_long_section(f'topic-{i}')}" for i in range(8)
]
body = "Intro paragraph before any heading.\n\n" + "\n\n".join(sections)
chunks = chunk_document("A very long dev-log", body)
assert len(chunks) > 1
joined = "\n".join(chunks)
for line in body.splitlines():
if line.strip():
assert line.strip() in joined, f"content dropped: {line[:60]!r}"
def test_every_chunk_fits_the_window_budget():
body = "\n\n".join(_long_section(f"t{i}") for i in range(10))
for chunk in chunk_document("T", body):
assert len(chunk) <= _CHUNK_CHAR_BUDGET + len("T") + 1
def test_every_chunk_is_anchored_by_the_title():
"""Each vector must carry its own topical anchor — the property that makes
snippets discriminative. An unanchored mid-document chunk would embed as
free-floating prose about nothing in particular."""
body = "\n\n".join(
f"## Section {i}\n\n{_long_section(f'sec-{i}')}" for i in range(6)
)
chunks = chunk_document("Retrieval reference", body)
assert len(chunks) > 1
for chunk in chunks:
assert chunk.startswith("Retrieval reference\n")
# --- boundary behaviour ------------------------------------------------------
def test_sections_split_at_markdown_headings_and_stay_whole_when_they_fit():
a = "## Alpha\n\nShort alpha content."
b = "## Beta\n\n" + _long_section("beta")
c = "## Gamma\n\n" + _long_section("gamma")
chunks = chunk_document("T", f"{a}\n\n{b}\n\n{c}")
# Beta's content never shares a chunk with Gamma's heading-onward content:
# heading boundaries are chunk boundaries unless merging small sections.
for chunk in chunks:
assert not ("(p5)" in chunk and "## Gamma" in chunk and "beta" in chunk)
def test_small_adjacent_sections_merge_instead_of_each_spending_a_vector():
body = (
"\n\n".join(f"## S{i}\n\nTiny." for i in range(4))
+ "\n\n## Big\n\n"
+ _long_section("big", paragraphs=10)
)
chunks = chunk_document("T", body)
tiny_chunks = [c for c in chunks if "Tiny." in c]
assert len(tiny_chunks) == 1, "four tiny sections should share one chunk"
def test_a_heading_inside_a_code_fence_does_not_split():
"""A commented `# step` in a recorded shell snippet is content, not
structure."""
body = "Intro.\n\n```bash\n# not a heading\necho hi\n```\n\nOutro."
chunks = chunk_document("T", body)
assert chunks == [embedding_text("T", body)]
def test_pieces_subsplit_from_one_section_repeat_its_heading():
"""'Which part of which topic' must survive the split — a continuation
piece without its heading embeds as context-free prose."""
body = "## The Only Topic\n\n" + _long_section("only", paragraphs=40)
chunks = chunk_document("T", body)
assert len(chunks) > 1
for chunk in chunks:
assert "## The Only Topic" in chunk
def test_a_monster_single_paragraph_is_hard_split_not_dropped():
body = "word " * 2000 # one paragraph, no newlines to split at
chunks = chunk_document("T", body)
assert len(chunks) > 1
total_words = sum(chunk.count("word") for chunk in chunks)
assert total_words == 2000
# --- the read path: best chunk wins (#280 step 4) ----------------------------
async def test_search_collapses_chunk_rows_to_best_chunk_per_note():
"""Rows arrive at CHUNK grain ordered by distance; a note appearing via
several chunks must come back ONCE, scored by its best chunk — otherwise a
long record fills the top-k with copies of itself."""
from unittest.mock import AsyncMock, MagicMock, patch
from scribe.services import embeddings as emb
note_a, note_b = MagicMock(id=1), MagicMock(id=2)
# Rows are (Note, distance, chunk_index, chunk_text) — the chunk columns
# ride along so the collapse can report WHICH passage won (#4243).
rows = [
(note_a, 0.10, 3, "the passage that actually matched"),
(note_b, 0.20, 0, "b's best"),
(note_a, 0.25, 7, "a worse chunk of a"),
(note_a, 0.30, 1, "a worse chunk of a"),
]
result = MagicMock()
result.all.return_value = rows
session, ctx = _session_ctx()
session.execute = AsyncMock(return_value=result)
report: dict = {}
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(emb, "get_embedding", AsyncMock(return_value=[0.0] * 384)),
):
out = await emb.semantic_search_notes(
1, "a query", limit=8, demote_superseded=False, report=report
)
assert [note.id for _s, note in out] == [1, 2]
assert out[0][0] == 1.0 - 0.10 # the BEST chunk's score, not a later one
# And the winning chunk is reported, not merely used for scoring. Without
# this a caller can only preview the head of the body — a span this query
# has already ranked lower than the one that won (#4243).
assert report["best_chunk"][1] == {
"index": 3, "text": "the passage that actually matched",
}
assert report["best_chunk"][2]["index"] == 0
# --- the write path: one row per chunk (#280 step 3) -------------------------
def _session_ctx():
from unittest.mock import AsyncMock, MagicMock
session = MagicMock()
session.execute = AsyncMock()
session.commit = AsyncMock()
ctx = MagicMock()
ctx.__aenter__ = AsyncMock(return_value=session)
ctx.__aexit__ = AsyncMock(return_value=False)
return session, ctx
async def test_upsert_stores_one_versioned_row_per_chunk():
from unittest.mock import AsyncMock, patch
from scribe.services import embeddings as emb
body = "\n\n".join(
f"## Section {i}\n\n{_long_section(f'sec-{i}')}" for i in range(6)
)
chunks = chunk_document("T", body)
assert len(chunks) > 1
session, ctx = _session_ctx()
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(
emb, "get_embeddings",
AsyncMock(return_value=[[0.0] * 384 for _ in chunks]),
),
):
await emb.upsert_note_embedding(7, 42, "T", body)
rows = [call.args[0] for call in session.add.call_args_list]
assert [r.chunk_index for r in rows] == list(range(len(chunks)))
assert [r.chunk_text for r in rows] == chunks
assert {r.chunker_version for r in rows} == {emb.CHUNKER_VERSION}
assert {r.user_id for r in rows} == {42}
session.execute.assert_awaited() # the delete that makes replacement atomic
async def test_upsert_of_an_emptied_record_clears_rows_instead_of_embedding():
"""An empty embedding is worse than none, and a STALE one is worse than
that — a record emptied of content must stop being findable by what it no
longer says."""
from unittest.mock import patch
from scribe.services import embeddings as emb
session, ctx = _session_ctx()
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(emb, "get_embeddings") as embedder,
):
await emb.upsert_note_embedding(7, 42, "", "")
embedder.assert_not_called()
session.execute.assert_awaited() # the delete
session.add.assert_not_called()
session.commit.assert_awaited()
async def test_backfill_reembeds_notes_with_a_stale_chunker_version():
"""The reason chunker_version exists: a shape change becomes a version
bump that re-embeds exactly the stale notes, instead of another 0077-style
table wipe. Only rows AT the current version count as done."""
from unittest.mock import AsyncMock, MagicMock, patch
from scribe.services import embeddings as emb
current_rows = MagicMock()
current_rows.fetchall.return_value = [(1,)] # note 1 is current
note_rows = MagicMock()
note_rows.fetchall.return_value = [
(1, 42, "current", "body"),
(2, 42, "stale-version", "body"),
]
session, ctx = _session_ctx()
session.execute = AsyncMock(side_effect=[current_rows, note_rows])
with (
patch.object(emb, "async_session", return_value=ctx),
patch.object(emb, "upsert_note_embedding", AsyncMock()) as upsert,
patch.object(emb.asyncio, "sleep", AsyncMock()),
):
await emb.backfill_note_embeddings()
embedded = [call.args[0] for call in upsert.call_args_list]
assert embedded == [2], "only the stale note is re-embedded"