feat(retrieval): a record the operator names by number reaches the menu (#4796)
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / TypeScript typecheck (push) Successful in 52s
CI & Build / integration (push) Successful in 54s
CI & Build / Python tests (push) Successful in 1m47s
CI & Build / Build & push image (push) Successful in 37s

#4796 "A record the operator names by number reaches auto-inject only if
its wording happens to match". "yes go ahead with 4448" says which record
is meant, but a number means nothing to an embedding, so the prompt menu
filled with records resembling the words around it.

- record_refs.named_record_ids reads the operator's raw prompt, never the
  reply-enriched query:
  - `#N`, unless the word before it marks another numbering (PR, CI, rule,
    milestone, system, log...);
  - a bare number of 3 or more digits that opens the message, follows a
    reference word or continues a list one started;
  - never a quantity ("300 seconds"), a date, version, path or fenced code.
- Each id is resolved through the ACL check; trashed or inaccessible ids
  are dropped.
- A "Named in your message" block leads the menu: the kind, the System,
  the name and the opening of the body, or the seen pointer if the record
  is already on the ledger. It takes no share of top_k, and the semantic
  lines leave those ids out.
- The block is booked under the new `named_ref` source, registered as an
  unbidden lookup that is allowed to be quiet. A named id that also ranked
  counts as suppressed in the auto_inject row, so #3668's identity holds.
- Named records now arrive even when the search finds nothing.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-10-03 22:10:14 -04:00
co-authored by Claude Opus 5.5
parent 5c6d9a7b55
commit f95ffae972
4 changed files with 469 additions and 41 deletions
+125 -1
View File
@@ -1,4 +1,4 @@
"""Record references written into prose: refusing guessed ids, resolving placeholders.
"""Record references written into prose: refusing guessed ids, resolving placeholders, reading the ids a message names.
THE DEFECT THIS EXISTS FOR (#4016)
@@ -129,3 +129,127 @@ def resolve_placeholders(text: str | None, refs: dict[str, str]) -> str | None:
if not text:
return text
return PLACEHOLDER_RE.sub(lambda m: refs[m.group(1)], text)
# ── Records an operator names in a message (#4796) ─────────────────────────
#
# "yes go ahead with 4448" says exactly which record is meant, and the prompt
# menu could not use it: a number carries no meaning to an embedding, so the
# menu filled with whatever resembled the words AROUND it. These ids are looked
# up directly instead (plugin_context.build_autoinject_hint).
#
# A guard against a false READ, not a false lookup. A number that is not a
# record id still has to resolve to a record this user can open before anything
# is shown, and the line says it was found by number — but every wrong id is a
# record dropped in front of the reader for no reason, so the parser takes a
# number only where the sentence makes it a reference.
# How many named records one message can bring in. A real message names one or
# two; past a handful the numbers are a pasted list or a log, not a reference.
NAMED_LIMIT = 5
# A number, with its `#` when it has one. The lookbehind keeps out a number
# inside a word, a path or URL (`pulls/198`), a version or date (`1.2`,
# `2026-10-03`), money and entities; the lookahead keeps out the same on the
# other side, a unit glued on (`300ms`, `40%`) and a thousands separator.
_NAMED_RE = re.compile(
r"(?<![\w#&$€£.:/\\=@~-])(#?)(\d{1,7})(?![\w%]|[.:/-]\d|,\d{3}(?!\d))"
)
# Words after which `#N` belongs to a different numbering: a forge's (a PR, a
# CI run, a job) or one of Scribe's own id spaces that is not a note's. Rules,
# milestones, Systems, projects and work-logs each draw from their own
# sequence, so "rule #153" is not note 153 — and note 153 is somebody's record.
_OTHER_SPACE = frozenset({
"pr", "prs", "pull", "pulls", "request", "requests", "mr", "mrs",
"run", "runs", "ci", "build", "builds", "job", "jobs", "pipeline",
"action", "actions", "commit", "github", "gitlab", "gitea", "gh",
"rule", "rules", "preference", "preferences", "milestone", "milestones",
"system", "systems", "project", "projects", "log", "logs", "topic",
"topics", "rulebook", "rulebooks", "step", "steps", "rank", "line",
"lines", "page", "item", "option", "version", "port",
})
# Words after which a BARE number is a reference. Bare numbers are most of what
# a message holds — sizes, counts, durations — so a bare one is taken only
# after a word that points at something, at the very start of the message, or
# continuing a list that started as a reference.
_REF_WORDS = frozenset({
"with", "for", "on", "at", "about", "re", "regarding", "into", "up",
"task", "tasks", "issue", "issues", "note", "notes", "record", "records",
"snippet", "snippets", "lesson", "lessons", "spike", "spikes", "process",
"continue", "resume", "start", "reopen", "open", "close", "finish",
"see", "check", "read", "do", "id", "ids",
})
# A word after a bare number that makes it a quantity: "for 300 seconds",
# "about 500 records". Not applied after `#`, which is never a quantity.
_UNITS = frozenset({
"s", "sec", "secs", "second", "seconds", "ms", "min", "mins", "minute",
"minutes", "h", "hr", "hrs", "hour", "hours", "day", "days", "week",
"weeks", "month", "months", "year", "years", "char", "chars", "character",
"characters", "token", "tokens", "line", "lines", "word", "words", "byte",
"bytes", "kb", "mb", "gb", "px", "row", "rows", "record", "records",
"note", "notes", "task", "tasks", "item", "items", "file", "files",
"test", "tests", "call", "calls", "time", "times", "x", "percent",
"result", "results", "hit", "hits", "query", "queries", "message",
"messages", "prompt", "prompts", "step", "steps", "point", "points",
})
_FENCE_RE = re.compile(r"```.*?(?:```|\Z)", re.S)
_PREV_WORD_RE = re.compile(r"([A-Za-z]+)[\s\"'(\[*_:]*\Z")
_NEXT_WORD_RE = re.compile(r"\s*([A-Za-z]+)")
_OPENS_RE = re.compile(r"[\s\"'(\[*_>-]*")
_LIST_JOIN_RE = re.compile(r"\s*(?:,|,?\s*(?:and|or|&|\+|/))\s*", re.I)
def _prev_word(before: str) -> str:
"""The word right before a number, past a space, a quote or a bracket."""
m = _PREV_WORD_RE.search(before)
return m.group(1).lower() if m else ""
def _next_word(after: str) -> str:
m = _NEXT_WORD_RE.match(after)
return m.group(1).lower() if m else ""
def named_record_ids(prompt: str | None, limit: int = NAMED_LIMIT) -> list[int]:
"""The record ids a message names, in the order it names them.
`#4448` unless the word before it is another numbering ("PR #198"); a bare
number of three or more digits when it opens the message ("4417 sounds
right"), follows a word that points at something ("go ahead with 4448") or
continues a list one of those started ("4448, 4449 and 4450"), and is not
a quantity ("for 300 seconds"). Fenced code is skipped: a pasted log is
full of numbers and names nothing.
Read the operator's OWN words, never a query enriched with the assistant's
reply — a number the agent wrote is not one the operator named.
"""
text = _FENCE_RE.sub("\n", prompt or "")
found: list[int] = []
last_end, last_ok = -1, False
for m in _NAMED_RE.finditer(text):
hashed, digits = bool(m.group(1)), m.group(2)
before, after = text[:m.start()], text[m.end():]
# A list inherits its head's reading, either way: "PR #198 and #199"
# is two PRs, "with 4448, 4449" is two records.
if last_end >= 0 and _LIST_JOIN_RE.fullmatch(text[last_end:m.start()]):
ok = last_ok
elif hashed:
ok = _prev_word(before) not in _OTHER_SPACE
else:
ok = (bool(_OPENS_RE.fullmatch(before))
or _prev_word(before) in _REF_WORDS)
last_end, last_ok = m.end(), ok
if not ok or digits[0] == "0" or len(digits) < (2 if hashed else 3):
continue
if not hashed and _next_word(after) in _UNITS:
continue
nid = int(digits)
if nid not in found:
found.append(nid)
if len(found) >= limit:
break
return found