← back to A2a Lab
add blind paraphrase eval harness (eval_find.py) — FALSIFIES lexical find() at 52.6% top-1
c5493c800cf0c98b223fc1ef8144003a9ed887f6 · 2026-08-01 22:28:31 -0700 · Steve
38 held-out everyday-phrasing queries; scorer misroutes 18/38 on blind phrasing.
Confirms Cody's contrarian objection: the 'every tested query passes' premise was circular.
Committed with the failing baseline on purpose; NOT tuned-to-pass (would re-create circularity).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Files touched
Diff
commit c5493c800cf0c98b223fc1ef8144003a9ed887f6
Author: Steve <steve@designerwallcoverings.com>
Date: Sat Aug 1 22:28:31 2026 -0700
add blind paraphrase eval harness (eval_find.py) — FALSIFIES lexical find() at 52.6% top-1
38 held-out everyday-phrasing queries; scorer misroutes 18/38 on blind phrasing.
Confirms Cody's contrarian objection: the 'every tested query passes' premise was circular.
Committed with the failing baseline on purpose; NOT tuned-to-pass (would re-create circularity).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
---
eval_find.py | 128 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 128 insertions(+)
diff --git a/eval_find.py b/eval_find.py
new file mode 100644
index 0000000..327483c
--- /dev/null
+++ b/eval_find.py
@@ -0,0 +1,128 @@
+"""Blind paraphrase eval harness for the cabinet-directory `find` router.
+
+Purpose (DTD verdict A, 2026-08-01, TK-10124): the case for keeping the lexical-semantic
+scorer rested on "every TESTED query routes correctly" — but that was circular (the
+queries and the synonym map shared an author). This harness is the demanded validation:
+held-out, everyday-phrasing queries scored blind against find(), reporting top-1 accuracy
+and every misroute so a real failure surfaces BEFORE prod.
+
+Honesty note: the built-in EVAL_SET is authored to AVOID the synonym-map / trigger
+vocabulary (stress test), but it is not authored by a fully independent party. For a
+truly blind run, drop in an external set: python eval_find.py --queries blind.json
+(JSON: [{"q": "...", "accept": ["vp-x", "vp-y"]}, ...]).
+
+Exit code: 0 if top-1 accuracy >= THRESHOLD, else 1 (so it can gate).
+"""
+from __future__ import annotations
+
+import json
+import sys
+
+import cabinet_directory as d
+
+THRESHOLD = 0.90
+
+# Each case: (query, [acceptable top-1 officers]). Most have ONE right answer; a few
+# genuinely span two domains (marked with two) — a router hint is allowed to pick either.
+# Queries use plain user phrasing and deliberately dodge the synonym-map tokens.
+EVAL_SET: list[tuple[str, list[str]]] = [
+ # vp-operations — infra/monitoring/secrets/dns
+ ("my website suddenly stopped responding, is the server alive", ["vp-operations"]),
+ ("point a new web address at my box and get https working", ["vp-operations"]),
+ ("something keeps dying overnight, can we auto-restart it", ["vp-operations"]),
+ ("where do i store a new access credential so everything picks it up", ["vp-operations", "vp-security"]),
+ # vp-dw-commerce — catalog/scrapers/shopify/skus
+ ("pull a wallpaper maker's whole product line into our store", ["vp-dw-commerce"]),
+ ("the prices on our storefront look wrong, fix the listings", ["vp-dw-commerce"]),
+ ("two products accidentally share the same item number", ["vp-dw-commerce"]),
+ ("add a new supplier's patterns to the shop", ["vp-dw-commerce"]),
+ # vp-dw-marketing — brand marketing/social/video
+ ("write me a short promo clip to post on instagram for the brand", ["vp-dw-marketing", "vp-research-content"]),
+ ("draft an email blast announcing the new collection", ["vp-dw-marketing"]),
+ ("help our products show up higher in google searches", ["vp-dw-marketing"]),
+ ("plan this month's posting schedule across our channels", ["vp-dw-marketing"]),
+ # vp-engineering — code/db/perf/test
+ ("this database query is painfully slow, speed it up", ["vp-engineering"]),
+ ("review my pull request for bugs before i merge", ["vp-engineering"]),
+ ("design a clean api for the new feature", ["vp-engineering"]),
+ ("the test suite is broken, figure out why", ["vp-engineering"]),
+ # vp-security — incident/rotation/firewall/breach
+ ("i think someone broke into one of our machines", ["vp-security"]),
+ ("lock down the open ports on the server", ["vp-security", "vp-operations"]),
+ ("a password may have leaked, change all the keys", ["vp-security"]),
+ ("are there any known vulnerabilities in our dependencies", ["vp-security", "vp-engineering"]),
+ # vp-research-content — research/records/video/mockups/voice
+ ("look up who owns a property in los angeles", ["vp-research-content"]),
+ ("dig up everything on this competitor company", ["vp-research-content"]),
+ ("make a narrated walkthrough video of this app", ["vp-research-content"]),
+ ("show me a few different homepage design directions", ["vp-research-content"]),
+ # vp-directories — lawyer/doctor/animals verticals
+ ("build out the attorney listings site", ["vp-directories"]),
+ ("add more physicians to the medical directory", ["vp-directories"]),
+ ("who is running paid ads in our restaurant directory", ["vp-directories"]),
+ # vp-compliance-policy — comms compliance/legal
+ ("is this email campaign legal to send to our list", ["vp-compliance-policy"]),
+ ("check whether we can text customers under the rules", ["vp-compliance-policy"]),
+ ("scrub this contact list against do-not-call", ["vp-compliance-policy"]),
+ # vp-cncp — flow/chief-of-staff
+ ("clear out the pending approvals and keep things moving", ["vp-cncp"]),
+ ("what is stalled across all my projects right now", ["vp-cncp"]),
+ # vp-special-projects — wallpapersback / apartmentwallpaper / site-factory
+ ("check on the ai-generated wallpaper storefront", ["vp-special-projects"]),
+ ("the peel and stick wallpaper site needs attention", ["vp-special-projects"]),
+ # vp-abramsego — command center / revenue engines / stripe
+ ("wire up test payments on the abrams command center", ["vp-abramsego"]),
+ ("check the waitlist signups on the ego dashboard", ["vp-abramsego"]),
+ # vp-consulting — client portals / intake
+ ("onboard a new consulting client and build their portal", ["vp-consulting"]),
+ ("run the intake questionnaire for a business with no website", ["vp-consulting"]),
+]
+
+
+def load_cases(path: str | None):
+ if not path:
+ return EVAL_SET
+ raw = json.load(open(path))
+ return [(c["q"], c["accept"] if isinstance(c["accept"], list) else [c["accept"]]) for c in raw]
+
+
+def main() -> int:
+ path = None
+ if "--queries" in sys.argv:
+ path = sys.argv[sys.argv.index("--queries") + 1]
+ cases = load_cases(path)
+
+ total = len(cases)
+ top1_ok = 0
+ top3_ok = 0
+ misroutes = []
+ for q, accept in cases:
+ ranked = d.find(q, top=3)
+ names = [o["vp"] for o, _s, _w in ranked]
+ top1 = names[0] if names else "(none)"
+ s1 = ranked[0][1] if ranked else 0.0
+ if top1 in accept:
+ top1_ok += 1
+ else:
+ # margin between the winner and the best acceptable officer, for triage
+ best_accept = next(((n, s) for (o, s, _w) in ranked for n in [o["vp"]] if n in accept), None)
+ misroutes.append((q, accept, top1, s1, best_accept))
+ if any(n in accept for n in names):
+ top3_ok += 1
+
+ acc = top1_ok / total if total else 0.0
+ r3 = top3_ok / total if total else 0.0
+ print(f"cabinet find() eval — {total} held-out queries ({'external' if path else 'built-in'})")
+ print(f" top-1 accuracy : {top1_ok}/{total} = {acc:.1%}")
+ print(f" top-3 recall : {top3_ok}/{total} = {r3:.1%}")
+ print(f" threshold : {THRESHOLD:.0%} → {'PASS' if acc >= THRESHOLD else 'FAIL'}")
+ if misroutes:
+ print(f"\n {len(misroutes)} misroute(s):")
+ for q, accept, top1, s1, ba in misroutes:
+ ba_str = f"{ba[0]} @ {ba[1]:.3f}" if ba else "not in top-3"
+ print(f" ✗ {q!r}\n got {top1} @ {s1:.3f} · wanted {accept} (best acceptable: {ba_str})")
+ return 0 if acc >= THRESHOLD else 1
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
← 714e492 add SDK-native client (client_sdk.py) + lexical-semantic fin
·
back to A2a Lab
·
scorer fix (field-weight + trigram-downweight + synonyms) + ac91d16 →