From 5ae60734bbfe9738a72fc13dc6b53a5f628f3f5d Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Wed, 16 Sep 2026 13:47:34 -0400 Subject: [PATCH] fix(retrieval): the completion-report arm gets its own bar, not the prompt arm's (#3860) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `report_preference` shipped reading PROMPTRULE_THRESHOLD_KEY, so the two arms were one dial: tuning the bar for an operator's prose silently retuned the lookup that runs when a task closes. That coupling is worse on this arm than it would be anywhere else. Every other retrieval arm scores a query that varies per call, so a mis-set bar shows up as a changed clear-rate. COMPLETION_QUERY is a fixed string, so this arm's best score for a given corpus is a CONSTANT — and a constant sitting under the bar is a dead arm rather than a quiet one. No volume of traffic reveals it. Found by the first live read for milestone 394 step 9: 69 calls, 69 declines, every one naming the same record at the same score (0.7194 against a 0.72 bar). Reading `best_available_id` (#3807) showed the record was about interpreting a REQUEST, not about report shape — so the declines were correct and the arm is healthy. The percentile alone would have said "lower the bar", which would have delivered a false positive on every completion report ever written. The bar does not move; the key does. Both defaults stay 0.72, so this changes no behaviour on any install — it makes "leave this one where it is" expressible, which it was not before. Settings grows the control (rules 25, 27), and the default-agreement check grows a row. Deliberately NOT included: a change to PROMPTRULE_DEFAULT_THRESHOLD. The evidence for moving it is this install's near-miss table, and rule 115 keeps a shipped default from being justified by one instance's corpus. That bar is a per-user setting and belongs in the operator's Settings, not in the product. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy --- frontend/src/views/SettingsView.vue | 34 ++++++++++++++++++++ src/scribe/services/plugin_context.py | 19 +++++++++++ src/scribe/services/reply_preferences.py | 34 +++++++++++++------- tests/test_services_reply_preferences.py | 40 ++++++++++++++++++++++++ tests/test_settings_defaults_agree.py | 1 + 5 files changed, 117 insertions(+), 11 deletions(-) diff --git a/frontend/src/views/SettingsView.vue b/frontend/src/views/SettingsView.vue index 73fa61c..5ecaa00 100644 --- a/frontend/src/views/SettingsView.vue +++ b/frontend/src/views/SettingsView.vue @@ -96,6 +96,7 @@ const kbToolRuleThreshold = ref("0.68"); // And the prompt boundary is a third query shape again — the operator's own // prose rather than anything a tool produced (#3852). const kbPromptRuleThreshold = ref("0.72"); +const kbReportPrefThreshold = ref("0.72"); // Near-duplicate report floors, one per record kind (services/dedup.py). // Snippets are single-chunk, so their floor sits below the 0.90 write-time // gate and catches what it lets through. Notes/tasks are scored at chunk @@ -171,6 +172,7 @@ async function saveKbInject() { // Bash call, so a fallback of 0 would put a rule in front of every command. const trT = Math.min(1, Math.max(0, Number(kbToolRuleThreshold.value) || 0.68)); const prT = Math.min(1, Math.max(0, Number(kbPromptRuleThreshold.value) || 0.72)); + const rpT = Math.min(1, Math.max(0, Number(kbReportPrefThreshold.value) || 0.72)); kbInjectThreshold.value = String(t); kbInjectTopK.value = String(k); kbDupThresholdSnippet.value = String(dupSnip); @@ -181,6 +183,7 @@ async function saveKbInject() { kbRuleHintThreshold.value = String(rhT); kbToolRuleThreshold.value = String(trT); kbPromptRuleThreshold.value = String(prT); + kbReportPrefThreshold.value = String(rpT); savingKbInject.value = true; kbInjectSaved.value = false; try { @@ -202,6 +205,12 @@ async function saveKbInject() { // queries are different shapes. Moving one must not move the others. kb_toolrule_threshold: String(trT), kb_promptrule_threshold: String(prT), + // A SIXTH, and the one that most needed its own key: this arm's query + // is a fixed string, so its score is a constant for a given corpus. + // While it borrowed the prompt bar, tuning prose silently retuned it — + // and a constant that lands under the bar is a dead arm, not a quiet + // one (#3860). + kb_reportpref_threshold: String(rpT), kb_duplicate_threshold_snippet: String(dupSnip), kb_duplicate_threshold_note: String(dupNote), kb_duplicate_threshold_task: String(dupTask), @@ -657,6 +666,9 @@ onMounted(async () => { if (allSettings.kb_promptrule_threshold !== undefined) { kbPromptRuleThreshold.value = allSettings.kb_promptrule_threshold; } + if (allSettings.kb_reportpref_threshold !== undefined) { + kbReportPrefThreshold.value = allSettings.kb_reportpref_threshold; + } if (allSettings.kb_writepath_threshold !== undefined) { kbWritePathThreshold.value = allSettings.kb_writepath_threshold; } @@ -1568,6 +1580,28 @@ async function deleteUser(userId: number) { not a command or a file — which is why it carries its own number.

+
+ + +

+ The bar for a preference about how a completion report should be + written, looked up when a task closes. Unlike every other bar + here, the question this arm asks never changes — so its score is + fixed by your preferences alone, and it will either always find one + or never find one. If you have written a preference for report shape + and it is not arriving, lower this; there is no run of calls that + will reveal the problem on its own. +

+