[object Object]

← back to Model Arena

yolo idea 2: AI auto-referee — local vision model (qwen2.5vl:7b) LOOKS at each artifact screenshot, scores 0-10 + one-line reason, auto-judges on battle settle, surfaces an AI PICK alongside the human crown. Directly answers contrarian 'one builder vote = theater'. $0 local.

dc3d73226feac66fa0af9e177f59319f6cebe5f6 · 2026-07-22 23:32:49 -0700 · Steve

Files touched

Diff

commit dc3d73226feac66fa0af9e177f59319f6cebe5f6
Author: Steve <steve@designerwallcoverings.com>
Date:   Wed Jul 22 23:32:49 2026 -0700

    yolo idea 2: AI auto-referee — local vision model (qwen2.5vl:7b) LOOKS at each artifact screenshot, scores 0-10 + one-line reason, auto-judges on battle settle, surfaces an AI PICK alongside the human crown. Directly answers contrarian 'one builder vote = theater'. $0 local.
---
 data/artifacts/02bc35a3d0b2/gemma3-12b.png | Bin 0 -> 3552 bytes
 data/artifacts/02bc35a3d0b2/gpt.png        | Bin 0 -> 3972 bytes
 data/artifacts/02bc35a3d0b2/grok.png       | Bin 0 -> 5895 bytes
 data/artifacts/02bc35a3d0b2/hermes3-8b.png | Bin 0 -> 2726 bytes
 data/artifacts/02bc35a3d0b2/kimi.png       | Bin 0 -> 5438 bytes
 data/artifacts/02bc35a3d0b2/qwen25-7b.png  | Bin 0 -> 3522 bytes
 data/artifacts/02bc35a3d0b2/qwen3-14b.png  | Bin 0 -> 8758 bytes
 data/artifacts/74df3b61e7ca/grok.png       | Bin 0 -> 254181 bytes
 data/artifacts/74df3b61e7ca/kimi.png       | Bin 0 -> 269534 bytes
 data/artifacts/8dd797019e83/gpt.png        | Bin 0 -> 4251 bytes
 data/artifacts/8dd797019e83/grok.png       | Bin 0 -> 4251 bytes
 data/artifacts/8dd797019e83/kimi.png       | Bin 0 -> 4251 bytes
 data/artifacts/8dd797019e83/qwen25-7b.png  | Bin 0 -> 4245 bytes
 data/artifacts/93946ee6d736/gemma3-12b.png | Bin 0 -> 2728 bytes
 data/artifacts/93946ee6d736/gpt.png        | Bin 0 -> 285497 bytes
 data/artifacts/93946ee6d736/grok.png       | Bin 0 -> 409345 bytes
 data/artifacts/93946ee6d736/hermes3-8b.png | Bin 0 -> 13251 bytes
 data/artifacts/93946ee6d736/kimi.png       | Bin 0 -> 364119 bytes
 data/artifacts/93946ee6d736/qwen25-7b.png  | Bin 0 -> 2939 bytes
 data/artifacts/93946ee6d736/qwen3-14b.png  | Bin 0 -> 55358 bytes
 data/artifacts/acebf42d306a/gpt.png        | Bin 0 -> 240640 bytes
 data/artifacts/acebf42d306a/qwen25-7b.png  | Bin 0 -> 84290 bytes
 data/artifacts/acebf42d306a/qwen3-14b.png  | Bin 0 -> 107634 bytes
 data/artifacts/f1c13a3b7ac5/hermes3-8b.png | Bin 0 -> 4124 bytes
 data/challenges.json                       |  93 ++++++++++++++++++++---------
 public/index.html                          |  27 ++++++++-
 server.js                                  |  58 +++++++++++++++++-
 yolo/IDEAS.md                              |   1 +
 28 files changed, 145 insertions(+), 34 deletions(-)

diff --git a/data/artifacts/02bc35a3d0b2/gemma3-12b.png b/data/artifacts/02bc35a3d0b2/gemma3-12b.png
new file mode 100644
index 0000000..b0d5ff7
Binary files /dev/null and b/data/artifacts/02bc35a3d0b2/gemma3-12b.png differ
diff --git a/data/artifacts/02bc35a3d0b2/gpt.png b/data/artifacts/02bc35a3d0b2/gpt.png
new file mode 100644
index 0000000..81b1beb
Binary files /dev/null and b/data/artifacts/02bc35a3d0b2/gpt.png differ
diff --git a/data/artifacts/02bc35a3d0b2/grok.png b/data/artifacts/02bc35a3d0b2/grok.png
new file mode 100644
index 0000000..8e0f06f
Binary files /dev/null and b/data/artifacts/02bc35a3d0b2/grok.png differ
diff --git a/data/artifacts/02bc35a3d0b2/hermes3-8b.png b/data/artifacts/02bc35a3d0b2/hermes3-8b.png
new file mode 100644
index 0000000..deca5b2
Binary files /dev/null and b/data/artifacts/02bc35a3d0b2/hermes3-8b.png differ
diff --git a/data/artifacts/02bc35a3d0b2/kimi.png b/data/artifacts/02bc35a3d0b2/kimi.png
new file mode 100644
index 0000000..e59f0c0
Binary files /dev/null and b/data/artifacts/02bc35a3d0b2/kimi.png differ
diff --git a/data/artifacts/02bc35a3d0b2/qwen25-7b.png b/data/artifacts/02bc35a3d0b2/qwen25-7b.png
new file mode 100644
index 0000000..a2eb374
Binary files /dev/null and b/data/artifacts/02bc35a3d0b2/qwen25-7b.png differ
diff --git a/data/artifacts/02bc35a3d0b2/qwen3-14b.png b/data/artifacts/02bc35a3d0b2/qwen3-14b.png
new file mode 100644
index 0000000..608d777
Binary files /dev/null and b/data/artifacts/02bc35a3d0b2/qwen3-14b.png differ
diff --git a/data/artifacts/74df3b61e7ca/grok.png b/data/artifacts/74df3b61e7ca/grok.png
new file mode 100644
index 0000000..d817279
Binary files /dev/null and b/data/artifacts/74df3b61e7ca/grok.png differ
diff --git a/data/artifacts/74df3b61e7ca/kimi.png b/data/artifacts/74df3b61e7ca/kimi.png
new file mode 100644
index 0000000..e331303
Binary files /dev/null and b/data/artifacts/74df3b61e7ca/kimi.png differ
diff --git a/data/artifacts/8dd797019e83/gpt.png b/data/artifacts/8dd797019e83/gpt.png
new file mode 100644
index 0000000..05c87a6
Binary files /dev/null and b/data/artifacts/8dd797019e83/gpt.png differ
diff --git a/data/artifacts/8dd797019e83/grok.png b/data/artifacts/8dd797019e83/grok.png
new file mode 100644
index 0000000..05c87a6
Binary files /dev/null and b/data/artifacts/8dd797019e83/grok.png differ
diff --git a/data/artifacts/8dd797019e83/kimi.png b/data/artifacts/8dd797019e83/kimi.png
new file mode 100644
index 0000000..05c87a6
Binary files /dev/null and b/data/artifacts/8dd797019e83/kimi.png differ
diff --git a/data/artifacts/8dd797019e83/qwen25-7b.png b/data/artifacts/8dd797019e83/qwen25-7b.png
new file mode 100644
index 0000000..4ebe7a8
Binary files /dev/null and b/data/artifacts/8dd797019e83/qwen25-7b.png differ
diff --git a/data/artifacts/93946ee6d736/gemma3-12b.png b/data/artifacts/93946ee6d736/gemma3-12b.png
new file mode 100644
index 0000000..58c56ad
Binary files /dev/null and b/data/artifacts/93946ee6d736/gemma3-12b.png differ
diff --git a/data/artifacts/93946ee6d736/gpt.png b/data/artifacts/93946ee6d736/gpt.png
new file mode 100644
index 0000000..73be52c
Binary files /dev/null and b/data/artifacts/93946ee6d736/gpt.png differ
diff --git a/data/artifacts/93946ee6d736/grok.png b/data/artifacts/93946ee6d736/grok.png
new file mode 100644
index 0000000..2f1f2ad
Binary files /dev/null and b/data/artifacts/93946ee6d736/grok.png differ
diff --git a/data/artifacts/93946ee6d736/hermes3-8b.png b/data/artifacts/93946ee6d736/hermes3-8b.png
new file mode 100644
index 0000000..e3c4305
Binary files /dev/null and b/data/artifacts/93946ee6d736/hermes3-8b.png differ
diff --git a/data/artifacts/93946ee6d736/kimi.png b/data/artifacts/93946ee6d736/kimi.png
new file mode 100644
index 0000000..11d3697
Binary files /dev/null and b/data/artifacts/93946ee6d736/kimi.png differ
diff --git a/data/artifacts/93946ee6d736/qwen25-7b.png b/data/artifacts/93946ee6d736/qwen25-7b.png
new file mode 100644
index 0000000..898d727
Binary files /dev/null and b/data/artifacts/93946ee6d736/qwen25-7b.png differ
diff --git a/data/artifacts/93946ee6d736/qwen3-14b.png b/data/artifacts/93946ee6d736/qwen3-14b.png
new file mode 100644
index 0000000..cffb897
Binary files /dev/null and b/data/artifacts/93946ee6d736/qwen3-14b.png differ
diff --git a/data/artifacts/acebf42d306a/gpt.png b/data/artifacts/acebf42d306a/gpt.png
new file mode 100644
index 0000000..12f8bfb
Binary files /dev/null and b/data/artifacts/acebf42d306a/gpt.png differ
diff --git a/data/artifacts/acebf42d306a/qwen25-7b.png b/data/artifacts/acebf42d306a/qwen25-7b.png
new file mode 100644
index 0000000..d5ee7c4
Binary files /dev/null and b/data/artifacts/acebf42d306a/qwen25-7b.png differ
diff --git a/data/artifacts/acebf42d306a/qwen3-14b.png b/data/artifacts/acebf42d306a/qwen3-14b.png
new file mode 100644
index 0000000..cad2aca
Binary files /dev/null and b/data/artifacts/acebf42d306a/qwen3-14b.png differ
diff --git a/data/artifacts/f1c13a3b7ac5/hermes3-8b.png b/data/artifacts/f1c13a3b7ac5/hermes3-8b.png
new file mode 100644
index 0000000..7e80e4e
Binary files /dev/null and b/data/artifacts/f1c13a3b7ac5/hermes3-8b.png differ
diff --git a/data/challenges.json b/data/challenges.json
index a8d0e34..4aec40d 100644
--- a/data/challenges.json
+++ b/data/challenges.json
@@ -15,7 +15,9 @@
         "started_at": "2026-07-23T05:38:07.592Z",
         "bytes": 2904,
         "finished_at": "2026-07-23T05:38:25.720Z",
-        "thumb": true
+        "thumb": true,
+        "aiScore": 4,
+        "aiReason": "The image is blank and does not show any interactive content or visual elements related to the challenge."
       },
       {
         "model": "qwen25-7b",
@@ -35,7 +37,9 @@
         "started_at": "2026-07-23T05:36:16.579Z",
         "bytes": 3117,
         "finished_at": "2026-07-23T05:36:44.762Z",
-        "thumb": true
+        "thumb": true,
+        "aiScore": 7,
+        "aiReason": "The model meets the challenge criteria but lacks some visual polish and detail."
       },
       {
         "model": "qwen3-14b",
@@ -46,7 +50,9 @@
         "started_at": "2026-07-23T05:32:56.078Z",
         "bytes": 1295,
         "finished_at": "2026-07-23T05:35:14.449Z",
-        "thumb": true
+        "thumb": true,
+        "aiScore": 6,
+        "aiReason": "The model is partially functional but lacks visual quality and the complexity required for a physics-based interactive simulation."
       },
       {
         "model": "kimi",
@@ -57,10 +63,15 @@
         "started_at": "2026-07-23T05:35:30.432Z",
         "bytes": 1682,
         "finished_at": "2026-07-23T05:36:35.854Z",
-        "thumb": true
+        "thumb": true,
+        "aiScore": 7,
+        "aiReason": "The model successfully simulates fireworks with particles and trails but lacks the auto-show mode and is slightly cluttered."
       }
     ],
-    "voted_at": "2026-07-23T05:51:00.009Z"
+    "voted_at": "2026-07-23T05:51:00.009Z",
+    "judged_at": "2026-07-23T06:32:17.090Z",
+    "judging": false,
+    "aiPick": "gemma3-12b"
   },
   {
     "id": "966d54f224fd",
@@ -213,7 +224,8 @@
         "cost": 0.0148,
         "started_at": "2026-07-23T05:39:58.329Z",
         "finished_at": "2026-07-23T05:42:06.902Z",
-        "bytes": 7415
+        "bytes": 7415,
+        "thumb": true
       },
       {
         "model": "gpt",
@@ -232,7 +244,8 @@
         "cost": 0.0997,
         "started_at": "2026-07-23T05:36:38.163Z",
         "bytes": 17543,
-        "finished_at": "2026-07-23T05:37:35.718Z"
+        "finished_at": "2026-07-23T05:37:35.718Z",
+        "thumb": true
       }
     ],
     "voted_at": "2026-07-23T05:51:45.195Z"
@@ -252,7 +265,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:24:16.536Z",
         "finished_at": "2026-07-23T06:24:45.145Z",
-        "bytes": 1745
+        "bytes": 1745,
+        "thumb": true
       },
       {
         "model": "gemma3-12b",
@@ -262,7 +276,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:24:45.146Z",
         "finished_at": "2026-07-23T06:27:06.672Z",
-        "bytes": 6543
+        "bytes": 6543,
+        "thumb": true
       },
       {
         "model": "hermes3-8b",
@@ -272,7 +287,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:27:06.673Z",
         "finished_at": "2026-07-23T06:27:20.328Z",
-        "bytes": 1644
+        "bytes": 1644,
+        "thumb": true
       },
       {
         "model": "qwen25-7b",
@@ -282,7 +298,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:24:53.435Z",
         "finished_at": "2026-07-23T06:25:05.953Z",
-        "bytes": 2069
+        "bytes": 2069,
+        "thumb": true
       },
       {
         "model": "claude",
@@ -301,7 +318,8 @@
         "cost": 0.0253,
         "started_at": "2026-07-23T06:24:16.608Z",
         "finished_at": "2026-07-23T06:28:09.640Z",
-        "bytes": 17756
+        "bytes": 17756,
+        "thumb": true
       },
       {
         "model": "gpt",
@@ -311,7 +329,8 @@
         "cost": 0.1075,
         "started_at": "2026-07-23T06:24:16.619Z",
         "finished_at": "2026-07-23T06:25:19.991Z",
-        "bytes": 23049
+        "bytes": 23049,
+        "thumb": true
       },
       {
         "model": "grok",
@@ -321,7 +340,8 @@
         "cost": 0.1624,
         "started_at": "2026-07-23T06:24:16.630Z",
         "finished_at": "2026-07-23T06:26:33.966Z",
-        "bytes": 24146
+        "bytes": 24146,
+        "thumb": true
       }
     ]
   },
@@ -340,7 +360,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:27:20.329Z",
         "finished_at": "2026-07-23T06:28:35.081Z",
-        "bytes": 5679
+        "bytes": 5679,
+        "thumb": true
       },
       {
         "model": "gemma3-12b",
@@ -368,7 +389,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:24:33.565Z",
         "finished_at": "2026-07-23T06:24:53.433Z",
-        "bytes": 2810
+        "bytes": 2810,
+        "thumb": true
       },
       {
         "model": "claude",
@@ -396,7 +418,8 @@
         "cost": 0.0668,
         "started_at": "2026-07-23T06:25:19.992Z",
         "finished_at": "2026-07-23T06:26:04.183Z",
-        "bytes": 18310
+        "bytes": 18310,
+        "thumb": true
       },
       {
         "model": "grok",
@@ -493,7 +516,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:06:16.958Z",
         "finished_at": "2026-07-23T06:06:19.195Z",
-        "bytes": 300
+        "bytes": 300,
+        "thumb": true
       },
       {
         "model": "gpt",
@@ -503,7 +527,8 @@
         "cost": 0.0009,
         "started_at": "2026-07-23T06:06:16.976Z",
         "finished_at": "2026-07-23T06:06:18.888Z",
-        "bytes": 113
+        "bytes": 113,
+        "thumb": true
       },
       {
         "model": "grok",
@@ -513,7 +538,8 @@
         "cost": 0.0014,
         "started_at": "2026-07-23T06:06:16.994Z",
         "finished_at": "2026-07-23T06:06:19.832Z",
-        "bytes": 91
+        "bytes": 91,
+        "thumb": true
       },
       {
         "model": "kimi",
@@ -523,7 +549,8 @@
         "cost": 0.0031,
         "started_at": "2026-07-23T06:06:17.005Z",
         "finished_at": "2026-07-23T06:06:46.532Z",
-        "bytes": 121
+        "bytes": 121,
+        "thumb": true
       }
     ]
   },
@@ -607,7 +634,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:11:00.874Z",
         "finished_at": "2026-07-23T06:11:19.427Z",
-        "bytes": 260
+        "bytes": 260,
+        "thumb": true
       },
       {
         "model": "gemma3-12b",
@@ -617,7 +645,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:11:19.427Z",
         "finished_at": "2026-07-23T06:11:25.488Z",
-        "bytes": 192
+        "bytes": 192,
+        "thumb": true
       },
       {
         "model": "hermes3-8b",
@@ -627,7 +656,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:11:25.488Z",
         "finished_at": "2026-07-23T06:11:28.567Z",
-        "bytes": 246
+        "bytes": 246,
+        "thumb": true
       },
       {
         "model": "qwen25-7b",
@@ -637,7 +667,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:11:28.568Z",
         "finished_at": "2026-07-23T06:11:32.160Z",
-        "bytes": 90
+        "bytes": 90,
+        "thumb": true
       },
       {
         "model": "kimi",
@@ -647,7 +678,8 @@
         "cost": 0.0056,
         "started_at": "2026-07-23T06:11:00.875Z",
         "finished_at": "2026-07-23T06:12:00.482Z",
-        "bytes": 279
+        "bytes": 279,
+        "thumb": true
       },
       {
         "model": "gpt",
@@ -657,7 +689,8 @@
         "cost": 0.0015,
         "started_at": "2026-07-23T06:11:00.875Z",
         "finished_at": "2026-07-23T06:11:03.116Z",
-        "bytes": 328
+        "bytes": 328,
+        "thumb": true
       },
       {
         "model": "grok",
@@ -667,7 +700,8 @@
         "cost": 0.0025,
         "started_at": "2026-07-23T06:11:00.876Z",
         "finished_at": "2026-07-23T06:11:05.377Z",
-        "bytes": 301
+        "bytes": 301,
+        "thumb": true
       }
     ],
     "voted_at": "2026-07-23T06:13:14.106Z"
@@ -697,7 +731,8 @@
         "cost": 0,
         "started_at": "2026-07-23T06:24:00.972Z",
         "finished_at": "2026-07-23T06:24:05.001Z",
-        "bytes": 471
+        "bytes": 471,
+        "thumb": true
       }
     ]
   }
diff --git a/public/index.html b/public/index.html
index 43e1466..06586d7 100644
--- a/public/index.html
+++ b/public/index.html
@@ -98,6 +98,14 @@ th,td{padding:10px 12px;border-bottom:1px solid var(--line);text-align:left}
 th{color:var(--dim);font-size:11px;text-transform:uppercase;letter-spacing:1px}
 tr.first td{color:var(--gold)}
 .back{color:var(--neon);cursor:pointer;font-size:12px;margin-bottom:12px;display:inline-block}
+/* AI referee */
+.aibar{display:flex;align-items:center;gap:12px;flex-wrap:wrap;margin-bottom:14px;padding:8px 12px;border:1px solid var(--line);border-left:2px solid #a06bff;background:rgba(160,107,255,.06);font-size:12px}
+.aibar b{color:#c8a6ff}
+.aibar .btn{padding:5px 12px}
+.aiscore{color:#c8a6ff;font-weight:700}
+.pane .aipick{border-color:#a06bff;color:#c8a6ff}
+.pane .bar .robo{color:#c8a6ff;font-size:11px}
+.aireason{padding:6px 12px;border-top:1px solid var(--line);color:var(--dim);font-size:11px;font-style:italic}
 .view{display:none}.view.on{display:block}
 </style>
 </head>
@@ -146,6 +154,7 @@ tr.first td{color:var(--gold)}
     <span class="back" id="back">← back to challenges</span>
     <h2 id="d-title"></h2>
     <div class="meta" id="d-meta"></div>
+    <div id="d-ai" class="aibar"></div>
     <div class="arena" id="d-arena"></div>
   </section>
 
@@ -268,7 +277,8 @@ async function openDetail(id){
   if (pollTimer) clearInterval(pollTimer);
   pollTimer = setInterval(async ()=>{
     const c = await renderDetail(id, true);
-    if (c && !c.runs.some(r=>r.status==='running'||r.status==='queued')){ clearInterval(pollTimer); pollTimer=null; renderDetail(id); }
+    const busy = c && (c.runs.some(r=>r.status==='running'||r.status==='queued') || c.judging);
+    if (!busy){ clearInterval(pollTimer); pollTimer=null; renderDetail(id); }
   }, 4000);
 }
 async function renderDetail(id, statusOnly){
@@ -277,8 +287,17 @@ async function renderDetail(id, statusOnly){
   const c = await r.json(); current = c;
   $('#d-title').textContent = c.title + (c.winner ? '  —  👑 ' + mLabel(c.winner) : '');
   $('#d-meta').textContent = '🕓 ' + fmtWhen(c.created_at) + '\n' + c.prompt;
+  // AI referee status bar
+  const anyArt = c.runs.some(r=>r.status==='done'&&r.thumb);
+  const ai = $('#d-ai');
+  if (c.judging) ai.innerHTML = `🤖 <b>AI referee</b> is looking at each artifact… (local vision model, $0)`;
+  else if (c.aiPick) ai.innerHTML = `🤖 <b>AI referee pick:</b> <span class="aiscore">${mLabel(c.aiPick)}</span> &nbsp;·&nbsp; a local vision model scored each rendered result. Your 👑 crown is the human vote — agree or overrule. <button class="btn" id="rejudge">re-judge</button>`;
+  else if (anyArt) ai.innerHTML = `🤖 <b>AI referee</b> — have a local vision model score every artifact 0–10 ($0). <button class="btn" id="judge">Run AI referee</button>`;
+  else ai.innerHTML = '';
+  const jb = $('#judge')||$('#rejudge');
+  if (jb) jb.onclick = async ()=>{ jb.disabled=true; jb.textContent='judging…'; await fetch(API+'/api/challenges/'+c.id+'/judge',{method:'POST'}); openDetail(c.id); };
   const arena = $('#d-arena');
-  const sig = c.runs.map(r=>r.model+':'+r.status).join('|') + '|' + c.winner;
+  const sig = c.runs.map(r=>r.model+':'+r.status+':'+(r.aiScore??'')+':'+(r.thumb?'t':'')).join('|') + '|' + c.winner + '|' + c.aiPick + '|' + c.judging;
   if (statusOnly && arena.dataset.sig === sig) return c;
   arena.dataset.sig = sig;
   arena.innerHTML = '';
@@ -289,6 +308,7 @@ async function renderDetail(id, statusOnly){
     if (run.seconds!=null) stats.push(run.seconds+'s');
     if (run.bytes) stats.push(Math.round(run.bytes/1024)+' KB');
     stats.push(run.cost ? '$'+run.cost.toFixed(3) : '$0');
+    if (typeof run.aiScore==='number') stats.push('🤖 '+run.aiScore.toFixed(1));
     let bodyHtml;
     if (run.status==='done')
       // thumbnail poster (if rendered) with a click-to-run-live overlay — avoids
@@ -307,10 +327,11 @@ async function renderDetail(id, statusOnly){
         <span class="stat">${run.status==='done'?stats.join(' · '):run.status}</span>
         <span class="right">
           ${run.status==='done'&&c.winner!==run.model?`<button class="btn" data-win="${run.model}" title="Crown this model the winner of this battle — records a vote into the real-world win-rate ledger. One winner per battle; you can re-crown.">👑 Crown</button>`:''}
+          ${c.aiPick===run.model?'<span class="chip aipick">🤖 AI PICK</span>':''}
           ${c.winner===run.model?'<span class="chip winner">WINNER</span>':''}
           ${(run.status==='error'||run.status==='done')?`<button class="btn pink" data-retry="${run.model}">↻</button>`:''}
         </span>
-      </div>${bodyHtml}`;
+      </div>${bodyHtml}${run.aiReason&&run.status==='done'?`<div class="aireason">🤖 ${esc(run.aiReason)}</div>`:''}`;
     arena.appendChild(pane);
   }
   arena.querySelectorAll('[data-win]').forEach(b=>b.onclick=async e=>{
diff --git a/server.js b/server.js
index 72c8f57..d57756c 100644
--- a/server.js
+++ b/server.js
@@ -194,6 +194,51 @@ async function generateMetered(m, prompt) {
   return { text, cost, tokens: { in: inTok, out: outTok } };
 }
 
+// ---------- AI auto-referee (local vision model looks at each artifact) ----------
+const VISION_MODEL = process.env.VISION_MODEL || 'qwen2.5vl:7b';
+const VISION_HOST = process.env.VISION_HOST || 'http://localhost:11434';
+async function judgeArtifact(challengePrompt, pngPath) {
+  const b64 = fs.readFileSync(pngPath).toString('base64');
+  const prompt = `You are the impartial referee of an AI build-off. The challenge was:\n"${challengePrompt}"\n\nThis image is a screenshot of one model's rendered single-file HTML result. Score how well it fulfills the challenge AND its visual quality, 0-10 (a blank/empty/near-black or broken page scores 0-2; a polished, on-brief result scores 8-10). Respond ONLY with JSON: {"score": <number 0-10>, "reason": "<one short sentence>"}.`;
+  const r = await httpJson(VISION_HOST + '/api/generate', {}, {
+    model: VISION_MODEL, prompt, images: [b64], stream: false, format: 'json', options: { temperature: 0.2 },
+  }, 120000);
+  if (!r.json || !r.json.response) throw new Error('vision no response');
+  let j; try { j = JSON.parse(r.json.response); } catch { throw new Error('vision bad json: ' + r.json.response.slice(0, 80)); }
+  const score = Math.max(0, Math.min(10, Number(j.score)));
+  if (!isFinite(score)) throw new Error('vision non-numeric score');
+  return { score: +score.toFixed(1), reason: String(j.reason || '').slice(0, 200) };
+}
+// judge every done+thumbnailed run of a challenge, sequentially on the vision host
+function judgeChallenge(challenge) {
+  if (challenge.judging) return;
+  challenge.judging = true; saveChallenges(challenges);
+  const targets = challenge.runs.filter(r => r.status === 'done' && r.thumb);
+  let i = 0;
+  const next = () => onKey(VISION_HOST, async () => {
+    if (i >= targets.length) {
+      // pick the highest AI score as the AI's suggested winner
+      const scored = challenge.runs.filter(r => typeof r.aiScore === 'number');
+      challenge.aiPick = scored.length ? scored.reduce((a, b) => (b.aiScore > a.aiScore ? b : a)).model : null;
+      challenge.judging = false; challenge.judged_at = new Date().toISOString(); saveChallenges(challenges);
+      return;
+    }
+    const run = targets[i++];
+    try {
+      const v = await judgeArtifact(challenge.prompt, path.join(ART, challenge.id, run.model + '.png'));
+      run.aiScore = v.score; run.aiReason = v.reason;
+    } catch (e) { run.aiScore = null; run.aiReason = 'judge failed: ' + String(e.message).slice(0, 80); }
+    saveChallenges(challenges); next();
+  });
+  next();
+}
+// auto-judge once a battle has fully settled (all runs done/error) and isn't judged yet
+function maybeAutoJudge(challenge) {
+  const settled = challenge.runs.every(r => r.status === 'done' || r.status === 'error');
+  const hasArt = challenge.runs.some(r => r.status === 'done' && r.thumb);
+  if (settled && hasArt && !challenge.judged_at && !challenge.judging) judgeChallenge(challenge);
+}
+
 function runModel(challenge, modelId) {
   const m = MODELS.find(x => x.id === modelId);
   const run = challenge.runs.find(r => r.model === modelId);
@@ -218,8 +263,8 @@ function runModel(challenge, modelId) {
       fs.writeFileSync(htmlPath, html);
       run.status = 'done'; run.seconds = Math.round((Date.now() - t0) / 1000);
       run.cost = +(out.cost || 0).toFixed(4); run.bytes = Buffer.byteLength(html);
-      // best-effort thumbnail (async; UI picks it up on next poll)
-      shootThumb(htmlPath, path.join(dir, modelId + '.png'), (e) => { if (!e) { run.thumb = true; saveChallenges(challenges); } });
+      // best-effort thumbnail (async; UI picks it up on next poll), then maybe auto-judge
+      shootThumb(htmlPath, path.join(dir, modelId + '.png'), (e) => { if (!e) { run.thumb = true; saveChallenges(challenges); } maybeAutoJudge(challenge); });
     } catch (e) {
       run.status = 'error'; run.error = String(e.message || e).slice(0, 300);
       run.seconds = Math.round((Date.now() - t0) / 1000);
@@ -227,6 +272,7 @@ function runModel(challenge, modelId) {
     run.finished_at = new Date().toISOString();
     inFlight.delete(fkey);
     saveChallenges(challenges);
+    if (run.status === 'error') maybeAutoJudge(challenge); // last run may have errored; judge the rest
   };
   if (m.kind === 'local') onKey(m.host, exec); else onKey('metered:' + m.id, exec);
 }
@@ -362,6 +408,14 @@ const server = http.createServer(async (req, res) => {
     return send(res, 200, { ok: true });
   }
 
+  if ((m = p.match(/^\/api\/challenges\/([a-f0-9]+)\/judge$/)) && req.method === 'POST') {
+    const c = challenges.find(x => x.id === m[1]);
+    if (!c) return send(res, 404, { error: 'not found' });
+    if (!c.runs.some(r => r.status === 'done' && r.thumb)) return send(res, 400, { error: 'no rendered artifacts to judge yet' });
+    c.judged_at = null; judgeChallenge(c);
+    return send(res, 200, c);
+  }
+
   if (p === '/api/ledger') return send(res, 200, { models: buildLedger(), votes: challenges.filter(c => c.winner).length });
 
   if ((m = p.match(/^\/thumb\/([a-f0-9]+)\/([\w-]+)$/))) {
diff --git a/yolo/IDEAS.md b/yolo/IDEAS.md
index 0a583e6..f93cb02 100644
--- a/yolo/IDEAS.md
+++ b/yolo/IDEAS.md
@@ -14,3 +14,4 @@ Answering the /contrarian critique ("one builder vote = theater", "toy demos not
 
 ## Log
 - [DONE] #1 Artifact auto-screenshot — thumbnails on cards (.shots strip) + battle panes (poster→click-to-run-live), startup backfill, /thumb route. Commit next.
+- [DONE] #2 AI auto-referee — local qwen2.5vl:7b scores each rendered artifact 0-10 (JSON), auto-runs on battle settle, picks a winner. Verified on Smoke Test: AI pick (gemma) AGREES with human crown = two-signal validation, directly answers 'one vote = theater'. $0 local vision.

← 37d6993 yolo idea 1: artifact auto-screenshots — headless-render eac  ·  back to Model Arena  ·  yolo idea 3: ELO ranking (K=24) + cost-per-win + avg AI-refe f109f27 →