[object Object]

← back to A2a Lab

add blind paraphrase eval harness (eval_find.py) — FALSIFIES lexical find() at 52.6% top-1

c5493c800cf0c98b223fc1ef8144003a9ed887f6 · 2026-08-01 22:28:31 -0700 · Steve

38 held-out everyday-phrasing queries; scorer misroutes 18/38 on blind phrasing.
Confirms Cody's contrarian objection: the 'every tested query passes' premise was circular.
Committed with the failing baseline on purpose; NOT tuned-to-pass (would re-create circularity).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Files touched

Diff

commit c5493c800cf0c98b223fc1ef8144003a9ed887f6
Author: Steve <steve@designerwallcoverings.com>
Date:   Sat Aug 1 22:28:31 2026 -0700

    add blind paraphrase eval harness (eval_find.py) — FALSIFIES lexical find() at 52.6% top-1
    
    38 held-out everyday-phrasing queries; scorer misroutes 18/38 on blind phrasing.
    Confirms Cody's contrarian objection: the 'every tested query passes' premise was circular.
    Committed with the failing baseline on purpose; NOT tuned-to-pass (would re-create circularity).
    
    Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---
 eval_find.py | 128 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
 1 file changed, 128 insertions(+)

diff --git a/eval_find.py b/eval_find.py
new file mode 100644
index 0000000..327483c
--- /dev/null
+++ b/eval_find.py
@@ -0,0 +1,128 @@
+"""Blind paraphrase eval harness for the cabinet-directory `find` router.
+
+Purpose (DTD verdict A, 2026-08-01, TK-10124): the case for keeping the lexical-semantic
+scorer rested on "every TESTED query routes correctly" — but that was circular (the
+queries and the synonym map shared an author). This harness is the demanded validation:
+held-out, everyday-phrasing queries scored blind against find(), reporting top-1 accuracy
+and every misroute so a real failure surfaces BEFORE prod.
+
+Honesty note: the built-in EVAL_SET is authored to AVOID the synonym-map / trigger
+vocabulary (stress test), but it is not authored by a fully independent party. For a
+truly blind run, drop in an external set:  python eval_find.py --queries blind.json
+(JSON: [{"q": "...", "accept": ["vp-x", "vp-y"]}, ...]).
+
+Exit code: 0 if top-1 accuracy >= THRESHOLD, else 1 (so it can gate).
+"""
+from __future__ import annotations
+
+import json
+import sys
+
+import cabinet_directory as d
+
+THRESHOLD = 0.90
+
+# Each case: (query, [acceptable top-1 officers]). Most have ONE right answer; a few
+# genuinely span two domains (marked with two) — a router hint is allowed to pick either.
+# Queries use plain user phrasing and deliberately dodge the synonym-map tokens.
+EVAL_SET: list[tuple[str, list[str]]] = [
+    # vp-operations — infra/monitoring/secrets/dns
+    ("my website suddenly stopped responding, is the server alive", ["vp-operations"]),
+    ("point a new web address at my box and get https working", ["vp-operations"]),
+    ("something keeps dying overnight, can we auto-restart it", ["vp-operations"]),
+    ("where do i store a new access credential so everything picks it up", ["vp-operations", "vp-security"]),
+    # vp-dw-commerce — catalog/scrapers/shopify/skus
+    ("pull a wallpaper maker's whole product line into our store", ["vp-dw-commerce"]),
+    ("the prices on our storefront look wrong, fix the listings", ["vp-dw-commerce"]),
+    ("two products accidentally share the same item number", ["vp-dw-commerce"]),
+    ("add a new supplier's patterns to the shop", ["vp-dw-commerce"]),
+    # vp-dw-marketing — brand marketing/social/video
+    ("write me a short promo clip to post on instagram for the brand", ["vp-dw-marketing", "vp-research-content"]),
+    ("draft an email blast announcing the new collection", ["vp-dw-marketing"]),
+    ("help our products show up higher in google searches", ["vp-dw-marketing"]),
+    ("plan this month's posting schedule across our channels", ["vp-dw-marketing"]),
+    # vp-engineering — code/db/perf/test
+    ("this database query is painfully slow, speed it up", ["vp-engineering"]),
+    ("review my pull request for bugs before i merge", ["vp-engineering"]),
+    ("design a clean api for the new feature", ["vp-engineering"]),
+    ("the test suite is broken, figure out why", ["vp-engineering"]),
+    # vp-security — incident/rotation/firewall/breach
+    ("i think someone broke into one of our machines", ["vp-security"]),
+    ("lock down the open ports on the server", ["vp-security", "vp-operations"]),
+    ("a password may have leaked, change all the keys", ["vp-security"]),
+    ("are there any known vulnerabilities in our dependencies", ["vp-security", "vp-engineering"]),
+    # vp-research-content — research/records/video/mockups/voice
+    ("look up who owns a property in los angeles", ["vp-research-content"]),
+    ("dig up everything on this competitor company", ["vp-research-content"]),
+    ("make a narrated walkthrough video of this app", ["vp-research-content"]),
+    ("show me a few different homepage design directions", ["vp-research-content"]),
+    # vp-directories — lawyer/doctor/animals verticals
+    ("build out the attorney listings site", ["vp-directories"]),
+    ("add more physicians to the medical directory", ["vp-directories"]),
+    ("who is running paid ads in our restaurant directory", ["vp-directories"]),
+    # vp-compliance-policy — comms compliance/legal
+    ("is this email campaign legal to send to our list", ["vp-compliance-policy"]),
+    ("check whether we can text customers under the rules", ["vp-compliance-policy"]),
+    ("scrub this contact list against do-not-call", ["vp-compliance-policy"]),
+    # vp-cncp — flow/chief-of-staff
+    ("clear out the pending approvals and keep things moving", ["vp-cncp"]),
+    ("what is stalled across all my projects right now", ["vp-cncp"]),
+    # vp-special-projects — wallpapersback / apartmentwallpaper / site-factory
+    ("check on the ai-generated wallpaper storefront", ["vp-special-projects"]),
+    ("the peel and stick wallpaper site needs attention", ["vp-special-projects"]),
+    # vp-abramsego — command center / revenue engines / stripe
+    ("wire up test payments on the abrams command center", ["vp-abramsego"]),
+    ("check the waitlist signups on the ego dashboard", ["vp-abramsego"]),
+    # vp-consulting — client portals / intake
+    ("onboard a new consulting client and build their portal", ["vp-consulting"]),
+    ("run the intake questionnaire for a business with no website", ["vp-consulting"]),
+]
+
+
+def load_cases(path: str | None):
+    if not path:
+        return EVAL_SET
+    raw = json.load(open(path))
+    return [(c["q"], c["accept"] if isinstance(c["accept"], list) else [c["accept"]]) for c in raw]
+
+
+def main() -> int:
+    path = None
+    if "--queries" in sys.argv:
+        path = sys.argv[sys.argv.index("--queries") + 1]
+    cases = load_cases(path)
+
+    total = len(cases)
+    top1_ok = 0
+    top3_ok = 0
+    misroutes = []
+    for q, accept in cases:
+        ranked = d.find(q, top=3)
+        names = [o["vp"] for o, _s, _w in ranked]
+        top1 = names[0] if names else "(none)"
+        s1 = ranked[0][1] if ranked else 0.0
+        if top1 in accept:
+            top1_ok += 1
+        else:
+            # margin between the winner and the best acceptable officer, for triage
+            best_accept = next(((n, s) for (o, s, _w) in ranked for n in [o["vp"]] if n in accept), None)
+            misroutes.append((q, accept, top1, s1, best_accept))
+        if any(n in accept for n in names):
+            top3_ok += 1
+
+    acc = top1_ok / total if total else 0.0
+    r3 = top3_ok / total if total else 0.0
+    print(f"cabinet find() eval — {total} held-out queries ({'external' if path else 'built-in'})")
+    print(f"  top-1 accuracy : {top1_ok}/{total} = {acc:.1%}")
+    print(f"  top-3 recall   : {top3_ok}/{total} = {r3:.1%}")
+    print(f"  threshold      : {THRESHOLD:.0%}  → {'PASS' if acc >= THRESHOLD else 'FAIL'}")
+    if misroutes:
+        print(f"\n  {len(misroutes)} misroute(s):")
+        for q, accept, top1, s1, ba in misroutes:
+            ba_str = f"{ba[0]} @ {ba[1]:.3f}" if ba else "not in top-3"
+            print(f"    ✗ {q!r}\n        got {top1} @ {s1:.3f} · wanted {accept} (best acceptable: {ba_str})")
+    return 0 if acc >= THRESHOLD else 1
+
+
+if __name__ == "__main__":
+    raise SystemExit(main())

← 714e492 add SDK-native client (client_sdk.py) + lexical-semantic fin  ·  back to A2a Lab  ·  scorer fix (field-weight + trigram-downweight + synonyms) + ac91d16 →