"""The completion-report preference lookup (milestone 409 step 4). What it pins: the lookup asks for PREFERENCES only, logs every call under its own source (the empty ones too), counts only what it showed as surfaced, and fails open. The query stays domain-neutral, because every kind of project closes tasks. """ import re from unittest.mock import AsyncMock, MagicMock, patch M = "scribe.services.reply_preferences" def _rule(rid, kind="preference"): return MagicMock(id=rid, kind=kind, title=f"t{rid}", statement=f"s{rid}") async def _run(hits, *, searched=True, raises=None): from scribe.services.reply_preferences import completion_preferences async def search(user_id, query, **kw): if raises: raise raises kw["report"].update({"searched": searched, "best_available_score": 0.7}) return hits search_mock = AsyncMock(side_effect=search) with patch(f"{M}.semantic_search_rules", search_mock), \ patch(f"{M}.get_setting", AsyncMock(return_value="0.72")), \ patch(f"{M}.record_retrieval") as logged, \ patch(f"{M}.record_rule_surfaced") as surfaced: out = await completion_preferences(7, project_id=3) return out, search_mock, logged, surfaced async def test_asks_for_preferences_only_and_returns_them_best_first(): out, search, logged, surfaced = await _run([(0.9, _rule(1)), (0.8, _rule(2))]) assert search.await_args.kwargs["kind"] == "preference" assert [p["id"] for p in out] == [1, 2] assert all(p["kind"] == "preference" for p in out) assert logged.call_args.kwargs["source"] == "report_preference" assert surfaced.call_args.kwargs == {"user_id": 7, "rule_ids": [1, 2], "source": "report_preference"} async def test_a_rule_that_slips_through_is_not_handed_back_as_a_preference(): out, _, _, surfaced = await _run([(0.9, _rule(1, kind="rule")), (0.8, _rule(2))]) assert [p["id"] for p in out] == [2] assert surfaced.call_args.kwargs["rule_ids"] == [2] async def test_an_empty_call_is_still_logged_and_surfaces_nothing(): out, _, logged, surfaced = await _run([]) assert out == [] logged.assert_called_once() assert logged.call_args.kwargs["results"] == [] surfaced.assert_not_called() async def test_a_search_that_never_ran_says_so_to_the_log(): _, _, logged, _ = await _run([], searched=False) assert logged.call_args.kwargs["searched"] is False async def test_fails_open(): out, _, _, surfaced = await _run([], raises=RuntimeError("embedder down")) assert out == [] surfaced.assert_not_called() def test_it_is_a_ranked_source(): from scribe.services.rule_usage import is_ambient assert not is_ambient("report_preference") def test_its_bar_is_its_own_key_not_the_prompt_arm_s(monkeypatch): """Regression on the coupling #3860 found (and on the fix being real). This arm shipped reading PROMPTRULE_THRESHOLD_KEY, so an operator tuning the bar for their own prose moved this one with it and was never told. That is worse here than anywhere else: every other arm scores a query that varies per call, while COMPLETION_QUERY is fixed — so this arm's score for a given corpus is a CONSTANT, and a constant that lands under the bar is a dead arm rather than a quiet one. No amount of traffic reveals it. Asserted on the key the lookup actually asks for, which is the thing that broke, rather than on the constant being defined somewhere. """ import asyncio from scribe.services import plugin_context from scribe.services.reply_preferences import completion_preferences asked: list[str] = [] async def get_setting(user_id, key, default): asked.append(key) return default async def search(user_id, query, **kw): kw["report"].update({"searched": True}) return [] with patch(f"{M}.get_setting", AsyncMock(side_effect=get_setting)), \ patch(f"{M}.semantic_search_rules", AsyncMock(side_effect=search)), \ patch(f"{M}.record_retrieval"): asyncio.run(completion_preferences(7)) assert asked == [plugin_context.REPORTPREF_THRESHOLD_KEY] assert plugin_context.REPORTPREF_THRESHOLD_KEY != plugin_context.PROMPTRULE_THRESHOLD_KEY, ( "the two keys are the same string again, so the settings form has one " "dial driving two arms — which is the defect, whatever the value is" ) def test_the_query_assumes_no_particular_domain(): from scribe.services.reply_preferences import COMPLETION_QUERY dev_only = [w for w in (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest") if re.search(w, COMPLETION_QUERY, re.IGNORECASE)] assert not dev_only, f"software-only vocabulary in a query every project runs: {dev_only}"