import time from quart import Blueprint, jsonify, request from scribe.auth import login_required, get_current_user_id from scribe.services.access import owner_names_for from scribe.services.embeddings import ( INTERACTIVE_SEARCH_THRESHOLD as _REST_SEARCH_THRESHOLD, ) from scribe.services.embeddings import semantic_search_notes from scribe.services.knowledge import content_type_filters from scribe.services.retrieval_telemetry import record_retrieval # The interactive floor lives in embeddings.py now, shared with Browse search # and the list views' semantic `q` — one number, one rationale (#2463). search_bp = Blueprint("search", __name__, url_prefix="/api/search") @search_bp.route("", methods=["GET"]) @login_required async def search_route(): uid = get_current_user_id() q = (request.args.get("q") or "").strip() if not q: return jsonify({"error": "q is required"}), 400 limit = min(request.args.get("limit", 10, type=int), 50) # Every kind the facet table declares, derived rather than mapped here — # this route used to know exactly two and read anything else as "no # filter", so `?content_type=snippets` silently returned the whole corpus # (#4250). An unknown kind is now a 400 naming the ones that exist: a # result set is an answer, and it should not be able to answer a question # nobody asked. try: filters = content_type_filters(request.args.get("content_type", "all")) except ValueError as exc: return jsonify({"error": str(exc)}), 400 is_task = filters.get("is_task") # Same association filters the MCP tool takes (#33). Optional, default # global: this route has NO frontend consumer today (measured 2026-08-08 — # the web UI searches through /api/knowledge), so it serves API callers, # and an API caller states its scope explicitly. system_id = request.args.get("system_id", type=int) project_id = request.args.get("project_id", type=int) t0 = time.perf_counter() report: dict = {} results = await semantic_search_notes( uid, q, limit=limit, **filters, threshold=_REST_SEARCH_THRESHOLD, project_id=project_id, system_id=system_id, # The user typed this, so it reaches everything they may read. scope="read", report=report, ) record_retrieval( user_id=uid, source="rest_search", query=q, threshold=_REST_SEARCH_THRESHOLD, limit=limit, project_id=project_id, is_task=is_task, results=results, duration_ms=(time.perf_counter() - t0) * 1000.0, best_available=report.get("best_available_score"), best_available_id=report.get("best_available_id"), searched=bool(report.get("searched", True)), ) owners = await owner_names_for( {int(note.user_id) for _s, note in results if note.user_id != uid} ) return jsonify({ "results": [ { "id": note.id, "title": note.title, "body": note.body or "", "is_task": note.is_task, "tags": note.tags or [], "similarity": score, **( {"shared": True, "owner": owners.get(int(note.user_id))} if note.user_id != uid else {} ), } for score, note in results # semantic_search_notes returns list[tuple[float, Note]] ], "total": len(results), })