diff --git a/engraphis/core/recall.py b/engraphis/core/recall.py index 428b9444..f4803fe5 100644 --- a/engraphis/core/recall.py +++ b/engraphis/core/recall.py @@ -195,6 +195,18 @@ def __init__(self, store: Store, embedder, vector_index, reranker: Optional[Rera max(1, int(resolved_cap)) if resolved_cap is not None else None ) self._planner_slot = threading.BoundedSemaphore(1) + # Latency knob: an operator may opt in to a narrower prompt-only first + # arm for small-k callers (k <= 20) where the B2 P2 latency tier + # widened the search by candidate_k + min(250, candidate_k*3). Setting + # ``ENGRAPHIS_RECALL_NARROW_ARM`` to any non-empty, non-zero value + # switches the formula to ``min(50, candidate_k * 2)`` which is ~2x + # less work at k=8 (16 vs 32). This is a recall-quality trade-off: + # the default wider arm remains untouched unless the operator opts in. + # The narrow arm is gated on k <= 20 because the regression that + # motivated the wider arm only showed up on small-k callers, and the + # larger-k callers need the full widening to find approved evidence. + narrow_raw = os.environ.get("ENGRAPHIS_RECALL_NARROW_ARM", "").strip() + self._narrow_arm_opt_in = bool(narrow_raw) and narrow_raw not in ("0", "false", "no") # "ppr" (default) = Personalized PageRank over entities+links (multi-hop); # "1hop" = the Phase-1 entity expansion, kept for fallback and ablation. self.graph_mode = graph_mode @@ -315,28 +327,24 @@ def recall(self, query: str, flt: Optional[SearchFilter] = None, *, k: int = 8, candidate_ceiling = candidate_k arm_candidate_k = candidate_k if prompt_only: - arm_candidate_k = candidate_k + min(250, candidate_k * 3) - # Latency knob (see __init__ and ``ARM_CANDIDATE_K_DEFAULT``). - # The effective cap clamps the widened first page so the common - # case stays cheap; the escalation ceiling below is intentionally - # NOT clamped by it. Escalation only fires when the first page came - # back saturated yet short on prompt-eligible records (the - # untrusted-heavy vault), so the extra scan is paid only when the - # trusted evidence would otherwise be unreachable. The ceiling - # itself stays bounded by PROMPT_ONLY_MAX_CANDIDATES. - # Clamp the widened first arm to the effective cap, but never - # below the caller's requested candidate_k so a small scope - # still searches at least as deep as requested. + narrow_arm_active = bool( + self._narrow_arm_opt_in and max(1, int(k)) <= 20 + ) + if narrow_arm_active: + arm_candidate_k = max(max(1, int(k)), min(50, candidate_k * 2)) + ceiling_bound = min( + PROMPT_ONLY_MAX_CANDIDATES, candidate_k * 4 + ) + else: + arm_candidate_k = candidate_k + min(250, candidate_k * 3) + ceiling_bound = min( + PROMPT_ONLY_MAX_CANDIDATES, + max(PROMPT_ONLY_MIN_CANDIDATES, candidate_k * 16), + ) arm_candidate_k = max( candidate_k, min(arm_cap, arm_candidate_k) ) - ceiling_bound = min( - PROMPT_ONLY_MAX_CANDIDATES, - max(PROMPT_ONLY_MIN_CANDIDATES, candidate_k * 16), - ) if self._arm_candidate_k_cap is not None: - # Explicit operator override: every index query, including the - # escalation pass, respects it. ceiling_bound = min(ceiling_bound, self._arm_candidate_k_cap) candidate_ceiling = max(arm_candidate_k, ceiling_bound) run_configs = [ diff --git a/tests/test_recall_narrow_arm_opt_in.py b/tests/test_recall_narrow_arm_opt_in.py new file mode 100644 index 00000000..c2263c55 --- /dev/null +++ b/tests/test_recall_narrow_arm_opt_in.py @@ -0,0 +1,195 @@ +"""Tests for the opt-in ``ENGRAPHIS_RECALL_NARROW_ARM`` latency knob. + +B2's 4th-pass review found the PR #171 prompt-only widening +(``candidate_k + min(250, candidate_k*3)``) costs ~5x more matrix-vector +work on a 49-fact corpus at k=8 (32ms -> 175ms). That regression was the +user-accepted trade-off for keeping recall quality on small-k callers, +but operators who care more about latency than top-of-list precision can +opt in to a narrower arm by setting ``ENGRAPHIS_RECALL_NARROW_ARM=1``. + +When the knob is enabled, the first prompt-only arm is clamped to +``max(k, min(50, candidate_k*2))`` and the escalation ceiling is clamped +to ``min(PROMPT_ONLY_MAX_CANDIDATES, candidate_k*4)`` so the second page +does not silently undo the savings. The narrow arm is gated on k <= 20 +because larger-k callers still need the full widening to find approved +evidence. + +Default behavior (env var unset / 0 / empty) is unchanged — the wider +arm from PR #171 is preserved verbatim. + +These tests prove both halves of the contract: the env var is honored +when set, and the wider arm is the default when it is not. +""" +from __future__ import annotations + +from engraphis.backends import DeterministicEmbedder +from engraphis.backends.reranker import IdentityReranker +from engraphis.core.interfaces import MemoryRecord, SearchFilter +from engraphis.core.recall import RecallEngine +from engraphis.core.store import Store + + +class _SemanticTestEmbedder(DeterministicEmbedder): + supports_semantic_search = True + embedding_mode = "semantic" + + +class _RecordingIndex: + """Vector-index double that records every arm size it was queried with.""" + + def __init__(self): + self.requested: list[int] = [] + + def search(self, query, k, *, filter=None): + self.requested.append(int(k)) + return [(f"mem_{i}", float(k - i)) for i in range(min(k, 4))] + + +def _add(store, emb, wid, rid, text, **kw): + provenance = dict(kw.get("provenance") or { + "source": "test", "trusted": True, "review_state": "approved", + }) + if provenance.get("trusted") is True: + provenance.setdefault("review_state", "approved") + kw["provenance"] = provenance + return store.add_memory(MemoryRecord( + id="", content=text, workspace_id=wid, repo_id=rid, + embedding=emb.embed([text])[0], **kw, + )) + + +def test_narrow_arm_opt_in_default_is_false(monkeypatch): + """Without the env var the opt-in flag is False — the wider arm stays the + default for every caller, including small-k ones.""" + monkeypatch.delenv("ENGRAPHIS_RECALL_NARROW_ARM", raising=False) + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + _RecordingIndex(), IdentityReranker()) + assert eng._narrow_arm_opt_in is False + + +def test_narrow_arm_opt_in_reads_env_var(monkeypatch): + """Any non-empty, non-zero env value flips the opt-in flag; bad/zero + values are ignored so a typo cannot silently change retrieval behavior.""" + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", "1") + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + _RecordingIndex(), IdentityReranker()) + assert eng._narrow_arm_opt_in is True + + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", "true") + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + _RecordingIndex(), IdentityReranker()) + assert eng._narrow_arm_opt_in is True + + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", "0") + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + _RecordingIndex(), IdentityReranker()) + assert eng._narrow_arm_opt_in is False + + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", "false") + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + _RecordingIndex(), IdentityReranker()) + assert eng._narrow_arm_opt_in is False + + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", " ") + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + _RecordingIndex(), IdentityReranker()) + assert eng._narrow_arm_opt_in is False + + +def test_narrow_arm_opt_in_changes_first_arm_at_k_8(monkeypatch): + """With the env var set, the first prompt-only arm at k=8 shrinks from + 8 + min(250, 24) = 32 to min(50, 8*2) = 16. This is the latency win + the knob is meant to unlock.""" + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", "1") + index = _RecordingIndex() + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + index, IdentityReranker()) + store = eng.store + wid = store.get_or_create_workspace("w") + for i in range(60): + _add(store, eng.embedder, wid, None, f"fact {i}") + + result = eng.recall("fact 5", SearchFilter(workspace_id=wid), k=8, + candidate_k=8, prompt_only=True) + + assert index.requested[0] == 16 # 8 + 24 (default) would have been 32 + assert result.candidate_k_used == 16 + # The result must still be non-empty: the narrow arm must not crash + # recall on a trusted-only corpus. + assert result.count >= 1 + + +def test_narrow_arm_does_not_change_default(monkeypatch): + """Without the env var the wider arm is preserved. This is the + user-accepted trade-off: 5x latency for top-of-list recall.""" + monkeypatch.delenv("ENGRAPHIS_RECALL_NARROW_ARM", raising=False) + index = _RecordingIndex() + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + index, IdentityReranker()) + store = eng.store + wid = store.get_or_create_workspace("w") + for i in range(60): + _add(store, eng.embedder, wid, None, f"fact {i}") + + result = eng.recall("fact 5", SearchFilter(workspace_id=wid), k=8, + candidate_k=8, prompt_only=True) + + # PR #171 default: 8 + min(250, 8*3) = 8 + 24 = 32 + assert index.requested[0] == 32 + assert result.candidate_k_used == 32 + + +def test_narrow_arm_only_applies_to_small_k(monkeypatch): + """The narrow arm is gated on k <= 20. A k=50 caller must still see + the wider arm even with the env var set — that is the caller class + that motivated the widening in the first place.""" + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", "1") + index = _RecordingIndex() + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + index, IdentityReranker()) + store = eng.store + wid = store.get_or_create_workspace("w") + for i in range(60): + _add(store, eng.embedder, wid, None, f"fact {i}") + + eng.recall("fact 5", SearchFilter(workspace_id=wid), k=50, + candidate_k=50, prompt_only=True) + + # k=50 is above the gate; wider arm preserved: 50 + min(250, 150) = 200. + assert index.requested[0] == 200 + + # k=20 is at the boundary and should still be narrow: 20*2 = 40. + index2 = _RecordingIndex() + eng2 = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + index2, IdentityReranker()) + eng2.recall("fact 5", SearchFilter(workspace_id=wid), k=20, + candidate_k=20, prompt_only=True) + assert index2.requested[0] == 40 + + # k=21 is just past the gate; wider arm preserved: 21 + min(250, 63) = 84. + index3 = _RecordingIndex() + eng3 = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + index3, IdentityReranker()) + eng3.recall("fact 5", SearchFilter(workspace_id=wid), k=21, + candidate_k=21, prompt_only=True) + assert index3.requested[0] == 84 + + +def test_narrow_arm_does_not_apply_to_non_prompt_only(monkeypatch): + """The narrow arm is a prompt-only knob. A non-prompt recall with the + env var set must behave exactly like the default, because the wider + widening lives inside the ``if prompt_only`` block.""" + monkeypatch.setenv("ENGRAPHIS_RECALL_NARROW_ARM", "1") + index = _RecordingIndex() + eng = RecallEngine(Store(":memory:"), _SemanticTestEmbedder(256), + index, IdentityReranker()) + store = eng.store + wid = store.get_or_create_workspace("w") + for i in range(60): + _add(store, eng.embedder, wid, None, f"fact {i}") + + # include_untrusted=True forces prompt_only=False regardless of the + # env var; the first arm must be the raw candidate_k with no widening. + eng.recall("fact 5", SearchFilter(workspace_id=wid), k=8, + candidate_k=8, prompt_only=False, include_untrusted=True) + assert index.requested[0] == 8