From 8d2d5cd562a952cfee899169f15b01c16fc530f7 Mon Sep 17 00:00:00 2001 From: aygea Date: Tue, 28 Jul 2026 16:51:50 -0700 Subject: [PATCH] Grade 3 more models + dashboard v2 layout (quant/format as first-class) New graded (11 total now): gemma4-26b-a4b-8bit-mlx 82 Minor Flaws (tied top local; delta-based tx freq) qwen3.6-27b-8bit-mlx 78 Minor Flaws (clean; anom. slow generation flagged) qwen3-coder-30b-6bit-mlx 50 Critical (asyncio.Lock used with sync with -> crash) Dashboard redesign: - Bar chart is now the full-width hero row (was cramped half-width) - 4 stat tiles squished 2x2 beside the radar up top - Quant + Format are dedicated columns in the leaderboard (MLX/GGUF/CLOUD chips) - New 'Format & Quant Showdown' panel: groups same-family variants so GGUF-vs-MLX and quant-depth comparisons are side by side - Bar-chart axis labels now include the quant so duplicate model names are distinguishable, with rotation for readability Co-Authored-By: Claude --- data/benchmark_history.json | 176 +++++++++--- generate_dashboard.py | 164 +++++++++-- outputs/gemma4-26b-a4b-8bit-mlx.py | 406 ++++++++++++++++++++++++++++ outputs/qwen3-coder-30b-6bit-mlx.py | 335 +++++++++++++++++++++++ outputs/qwen3.6-27b-8bit-mlx.py | 349 ++++++++++++++++++++++++ 5 files changed, 1372 insertions(+), 58 deletions(-) create mode 100644 outputs/gemma4-26b-a4b-8bit-mlx.py create mode 100644 outputs/qwen3-coder-30b-6bit-mlx.py create mode 100644 outputs/qwen3.6-27b-8bit-mlx.py diff --git a/data/benchmark_history.json b/data/benchmark_history.json index cd8303a..2479877 100644 --- a/data/benchmark_history.json +++ b/data/benchmark_history.json @@ -1,10 +1,16 @@ { "meta": { - "project": "Local LLM Benchmark Suite — LFU Cache & ACID Audit", + "project": "Local LLM Benchmark Suite \u2014 LFU Cache & ACID Audit", "machine": "Apple M3 Max, 48GB unified memory, LM Studio", "exam_prompt": "prompts/lfu_cache_prompt.txt", "grading_rubric": "prompts/grading.txt", - "pillars": ["complexity", "concurrency", "isolation", "memory_edge_cases", "test_integrity"], + "pillars": [ + "complexity", + "concurrency", + "isolation", + "memory_edge_cases", + "test_integrity" + ], "max_per_pillar": 20, "schema_version": 1 }, @@ -29,16 +35,16 @@ "test_integrity": 17 }, "verdict": "Minor Logic Flaws", - "best_for": "Solid daily-driver scaffolding for ACID/async patterns — produces runnable, well-structured code, but needs a human pass for __slots__, monotonic clocks, and lock granularity before production.", + "best_for": "Solid daily-driver scaffolding for ACID/async patterns \u2014 produces runnable, well-structured code, but needs a human pass for __slots__, monotonic clocks, and lock granularity before production.", "critical_bugs": [ - "No __slots__ declared on Node/Transaction/LFUCache — rubric explicitly required it for memory efficiency.", - "Uses time.time() (system clock) throughout instead of time.monotonic() — NTP adjustments can cause premature/incorrect TTL eviction.", - "_cleanup_freq_lists() performs a hidden O(F) scan (iterates all freq tiers + min()), called after every eviction, background batch, AND inside commit() — violates the strict O(1) requirement.", - "Transaction commit holds the single cache lock across all write/delete/bump loops + cleanup — coarse-grained, blocks all readers for the whole commit window; no fine-grained locking.", + "No __slots__ declared on Node/Transaction/LFUCache \u2014 rubric explicitly required it for memory efficiency.", + "Uses time.time() (system clock) throughout instead of time.monotonic() \u2014 NTP adjustments can cause premature/incorrect TTL eviction.", + "_cleanup_freq_lists() performs a hidden O(F) scan (iterates all freq tiers + min()), called after every eviction, background batch, AND inside commit() \u2014 violates the strict O(1) requirement.", + "Transaction commit holds the single cache lock across all write/delete/bump loops + cleanup \u2014 coarse-grained, blocks all readers for the whole commit window; no fine-grained locking.", "Lost-update risk: commit applies tx-local writes without any MVCC/version check, so a key modified by the background evictor or another committer between tx.get() and commit() is overwritten blindly.", "Tests dodge hard cases: 50-task stress uses unique keys with capacity 100, so no eviction-under-contention ever happens; no test for rollback-after-partial-application or mid-commit read isolation." ], - "patch_code": "# FIX 1: Add __slots__ for memory efficiency\n@dataclass\nclass Node:\n __slots__ = ('key', 'value', 'freq', 'expires_at', 'prev', 'next')\n key: Any\n value: Any\n freq: int\n expires_at: Optional[float]\n prev: Optional['Node'] = None\n next: Optional['Node'] = None\n\n# FIX 2: Use monotonic clock everywhere (get/put/_add_node/commit)\n# time.time() -> time.monotonic()\n# e.g.\nexpires_at = time.monotonic() + ttl_seconds if ttl_seconds else None\n\n# FIX 3: Make _cleanup_freq_lists O(1) — bump min_freq incrementally\n# instead of recomputing min() across all tiers:\n# In _update_freq, when emptying the min_freq bucket, only bump min_freq\n# if you're evicting from it; otherwise leave it. Delete the global\n# min(self.freq_map.keys()) scan. For background sweeps, prune empty\n# buckets lazily on next _evict() rather than scanning proactively.\n\n# FIX 4: Shrink commit critical section — apply writes into a staging\n# structure under the lock, then release; or use per-bucket locks so\n# readers on unrelated keys aren't blocked.\n\n# FIX 5: Add MVCC version to Node; in commit, raise/abort if the\n# stored version != the version seen at tx.get() time (lost-update detect)." + "patch_code": "# FIX 1: Add __slots__ for memory efficiency\n@dataclass\nclass Node:\n __slots__ = ('key', 'value', 'freq', 'expires_at', 'prev', 'next')\n key: Any\n value: Any\n freq: int\n expires_at: Optional[float]\n prev: Optional['Node'] = None\n next: Optional['Node'] = None\n\n# FIX 2: Use monotonic clock everywhere (get/put/_add_node/commit)\n# time.time() -> time.monotonic()\n# e.g.\nexpires_at = time.monotonic() + ttl_seconds if ttl_seconds else None\n\n# FIX 3: Make _cleanup_freq_lists O(1) \u2014 bump min_freq incrementally\n# instead of recomputing min() across all tiers:\n# In _update_freq, when emptying the min_freq bucket, only bump min_freq\n# if you're evicting from it; otherwise leave it. Delete the global\n# min(self.freq_map.keys()) scan. For background sweeps, prune empty\n# buckets lazily on next _evict() rather than scanning proactively.\n\n# FIX 4: Shrink commit critical section \u2014 apply writes into a staging\n# structure under the lock, then release; or use per-bucket locks so\n# readers on unrelated keys aren't blocked.\n\n# FIX 5: Add MVCC version to Node; in commit, raise/abort if the\n# stored version != the version seen at tx.get() time (lost-update detect)." }, { "id": "qwen3.6-35b-a3b-4bit-mlx", @@ -60,17 +66,17 @@ "test_integrity": 4 }, "verdict": "Critical Bugs", - "best_for": "Not recommended for systems code as-is. The 4-bit quant degrades logic sharply vs the 6-bit sibling (82->57) — usable only for boilerplate/scaffolding drafts that a human will heavily rewrite.", + "best_for": "Not recommended for systems code as-is. The 4-bit quant degrades logic sharply vs the 6-bit sibling (82->57) \u2014 usable only for boilerplate/scaffolding drafts that a human will heavily rewrite.", "critical_bugs": [ - "FATAL: _evict() double-removes nodes — pop() already calls remove() (nulling node.prev/next), then _remove_node() calls remove() AGAIN -> AttributeError: 'NoneType' on first eviction. The cache cannot survive reaching capacity.", + "FATAL: _evict() double-removes nodes \u2014 pop() already calls remove() (nulling node.prev/next), then _remove_node() calls remove() AGAIN -> AttributeError: 'NoneType' on first eviction. The cache cannot survive reaching capacity.", "Test suite never executes: test_lfu_eviction crashes at the first eviction, so the 'All tests passed' message is unreachable and assertions are effectively unverified.", - "Background _eviction_loop materializes list(self.key_to_node.keys())[:50] every sweep — an O(N) linear scan, forbidden by the strict O(1) requirement.", - "_FreqList.pop() has no empty-guard — calling pop() on an empty list dereferences self.head.next (the dummy tail) and corrupts the DLL.", + "Background _eviction_loop materializes list(self.key_to_node.keys())[:50] every sweep \u2014 an O(N) linear scan, forbidden by the strict O(1) requirement.", + "_FreqList.pop() has no empty-guard \u2014 calling pop() on an empty list dereferences self.head.next (the dummy tail) and corrupts the DLL.", "Transaction _apply_put applies the buffered value into the EXACT original_node captured at tx.put() time; if the global cache evicted/relocated that node between put and commit, you mutate a stale/dangling node (no MVCC/version check).", "Commit is not atomic across exceptions: a crash mid-_apply loop leaves half-applied global state with no rollback.", - "Uses time.time() (system clock) throughout instead of time.monotonic() — NTP jumps corrupt TTL eviction." + "Uses time.time() (system clock) throughout instead of time.monotonic() \u2014 NTP jumps corrupt TTL eviction." ], - "patch_code": "# FIX 1 (the crash): _evict double-removes. pop() already unlinks,\n# so do NOT call _remove_node on a popped node. Either:\n# (a) pop and then only delete the key_map entry + min_freq bookkeeping:\ndef _evict(self):\n if not self.freq_to_list:\n return\n evict_list = self.freq_to_list[self.min_freq]\n if evict_list.size == 0: # guard against empty\n del self.freq_to_list[self.min_freq]\n return\n node = evict_list.pop() # pop() unlinks + nulls prev/next\n del self.key_to_node[node.key] # DON'T call _remove_node again\n if self.freq_to_list[self.min_freq].size == 0:\n del self.freq_to_list[self.min_freq]\n self.min_freq += 1\n\n# FIX 2: _FreqList.pop empty-guard\ndef pop(self) -> _Node:\n if self.size == 0:\n raise IndexError('pop from empty _FreqList')\n node = self.head.next\n self.remove(node)\n return node\n\n# FIX 3: kill the O(N) scan in background sweep — maintain a separate\n# set of keys that have a TTL, and iterate that set in batches:\nasync with self.lock:\n batch = list(self._ttl_keys)[:50]\n for k in batch:\n node = self.key_to_node.get(k)\n if node and 0 < node.expires_at <= time.monotonic():\n self._remove_node(node)\n\n# FIX 4: time.time() -> time.monotonic() everywhere.\n# FIX 5: add node.version; in tx._apply_put, abort/refresh if\n# cache.key_to_node[key] is a different node than original_node." + "patch_code": "# FIX 1 (the crash): _evict double-removes. pop() already unlinks,\n# so do NOT call _remove_node on a popped node. Either:\n# (a) pop and then only delete the key_map entry + min_freq bookkeeping:\ndef _evict(self):\n if not self.freq_to_list:\n return\n evict_list = self.freq_to_list[self.min_freq]\n if evict_list.size == 0: # guard against empty\n del self.freq_to_list[self.min_freq]\n return\n node = evict_list.pop() # pop() unlinks + nulls prev/next\n del self.key_to_node[node.key] # DON'T call _remove_node again\n if self.freq_to_list[self.min_freq].size == 0:\n del self.freq_to_list[self.min_freq]\n self.min_freq += 1\n\n# FIX 2: _FreqList.pop empty-guard\ndef pop(self) -> _Node:\n if self.size == 0:\n raise IndexError('pop from empty _FreqList')\n node = self.head.next\n self.remove(node)\n return node\n\n# FIX 3: kill the O(N) scan in background sweep \u2014 maintain a separate\n# set of keys that have a TTL, and iterate that set in batches:\nasync with self.lock:\n batch = list(self._ttl_keys)[:50]\n for k in batch:\n node = self.key_to_node.get(k)\n if node and 0 < node.expires_at <= time.monotonic():\n self._remove_node(node)\n\n# FIX 4: time.time() -> time.monotonic() everywhere.\n# FIX 5: add node.version; in tx._apply_put, abort/refresh if\n# cache.key_to_node[key] is a different node than original_node." }, { "id": "qwen3.6-35b-a3b-uncensored-hauhaucs-aggressive-gguf", @@ -94,11 +100,11 @@ "verdict": "Critical Bugs", "best_for": "Not recommended for production code. Reasonable API shape and correctly used time.monotonic(), but the module does not parse (syntax error), contains an infinite while:pass loop, and has data races. Avoid for systems/concurrency work.", "critical_bugs": [ - "SyntaxError: line 305 'assert val := await cache.get(...)' is invalid Python — walrus operator cannot appear in an assert statement. The ENTIRE module fails to compile, so nothing runs and no test can execute.", - "Infinite busy-loop: _evict_lfu lines 159-160 — 'while self.min_freq in self.freq_map and self.min_freq < max(...): pass' has an empty body that never updates min_freq, recomputes max() (O(F)) each iteration, and can never terminate.", - "Hidden O(F) scan: min(self.freq_map.keys()) / max(self.freq_map.keys()) appears at 6 call sites (every eviction and removal) — violates the strict O(1) requirement.", + "SyntaxError: line 305 'assert val := await cache.get(...)' is invalid Python \u2014 walrus operator cannot appear in an assert statement. The ENTIRE module fails to compile, so nothing runs and no test can execute.", + "Infinite busy-loop: _evict_lfu lines 159-160 \u2014 'while self.min_freq in self.freq_map and self.min_freq < max(...): pass' has an empty body that never updates min_freq, recomputes max() (O(F)) each iteration, and can never terminate.", + "Hidden O(F) scan: min(self.freq_map.keys()) / max(self.freq_map.keys()) appears at 6 call sites (every eviction and removal) \u2014 violates the strict O(1) requirement.", "Race condition: get() and put() perform lazy-TTL _remove_key() BEFORE acquiring the lock (lines 71-73, 107-108), mutating shared state unlocked while other coroutines read/write.", - "Deadlock risk: background_loop holds self._lock, then calls await self._remove_key() which is itself a lock-acquiring method — asyncio.Lock is NOT reentrant -> deadlock when the evictor runs.", + "Deadlock risk: background_loop holds self._lock, then calls await self._remove_key() which is itself a lock-acquiring method \u2014 asyncio.Lock is NOT reentrant -> deadlock when the evictor runs.", "No __slots__ on _Node despite using a dataclass (rubric required it for memory efficiency).", "_remove_key will KeyError on self.freq_map[freq] if a concurrent operation already deleted that bucket." ], @@ -126,14 +132,14 @@ "verdict": "Critical Bugs", "best_for": "Promising code-design instincts (cleanest abstractions and best transaction isolation design in the set) but undone by a single fatal one-line bug that stops it running. With the bug fixed it would likely score 80+; as-is, only useful as a structural reference.", "critical_bugs": [ - "FATAL: _put_internal line 305 inserts a NEW key with 'self._freq_map[1].push_front(...)' but never ensures the freq-1 bucket exists — KeyError: 1 on the very first put. The cache cannot store a single key. The _ensure_freq_list(1) helper it should use exists and is used correctly everywhere else (lines 294, 354).", - "Transaction.commit() calls _put_internal for buffered writes, so it hits the same KeyError: 1 — committed transactions crash too.", + "FATAL: _put_internal line 305 inserts a NEW key with 'self._freq_map[1].push_front(...)' but never ensures the freq-1 bucket exists \u2014 KeyError: 1 on the very first put. The cache cannot store a single key. The _ensure_freq_list(1) helper it should use exists and is used correctly everywhere else (lines 294, 354).", + "Transaction.commit() calls _put_internal for buffered writes, so it hits the same KeyError: 1 \u2014 committed transactions crash too.", "Test suite cannot execute: crashes at the first cache.put() in main(); the well-built test harness (pass/fail counter, 4 real scenarios) validates nothing.", "No __slots__ on _DLLNode/_CacheNode/_DoublyLinkedList despite the rubric requiring it for memory efficiency.", - "_evict_node uses min(self._freq_map) (line 325) when the min-tier empties — a hidden O(F) scan, violating strict O(1).", + "_evict_node uses min(self._freq_map) (line 325) when the min-tier empties \u2014 a hidden O(F) scan, violating strict O(1).", "No MVCC/version check on transaction commit (lost-update possible if the global key is modified between tx.get and commit); commit is not exception-safe across the writes-vs-deletes loops." ], - "patch_code": "# FIX 1 (the fatal one-liner): use the helper that already exists.\n# line 305, in _put_internal, new-key branch:\n- self._freq_map[1].push_front(dll_node)\n+ self._ensure_freq_list(1).push_front(dll_node)\n# (This single change makes the cache and transactions functional.)\n\n# FIX 2: replace the O(F) min() scan with an incremental bump:\n# in _evict_node, when the min-tier bucket empties, min_freq is the\n# lowest remaining tier. Since freq only ever increments by 1, the\n# next min is almost always min_freq+1; track it incrementally rather\n# than scanning. Or, since this only happens on full eviction, accept\n# O(F) but only on the empty-cache edge — document it.\n\n# FIX 3: add __slots__ to all internal classes:\nclass _CacheNode:\n __slots__ = ('key','value','ttl_seconds','expiry_time','freq','dll_node')\n ...\n\n# FIX 4: wrap commit applies in try/except so a mid-commit exception\n# does not leave a half-applied global state; consider abort semantics.\n# FIX 5: add node.version; in tx commit, abort if the global node for\n# a key is not the one seen at tx.get() time." + "patch_code": "# FIX 1 (the fatal one-liner): use the helper that already exists.\n# line 305, in _put_internal, new-key branch:\n- self._freq_map[1].push_front(dll_node)\n+ self._ensure_freq_list(1).push_front(dll_node)\n# (This single change makes the cache and transactions functional.)\n\n# FIX 2: replace the O(F) min() scan with an incremental bump:\n# in _evict_node, when the min-tier bucket empties, min_freq is the\n# lowest remaining tier. Since freq only ever increments by 1, the\n# next min is almost always min_freq+1; track it incrementally rather\n# than scanning. Or, since this only happens on full eviction, accept\n# O(F) but only on the empty-cache edge \u2014 document it.\n\n# FIX 3: add __slots__ to all internal classes:\nclass _CacheNode:\n __slots__ = ('key','value','ttl_seconds','expiry_time','freq','dll_node')\n ...\n\n# FIX 4: wrap commit applies in try/except so a mid-commit exception\n# does not leave a half-applied global state; consider abort semantics.\n# FIX 5: add node.version; in tx commit, abort if the global node for\n# a key is not the one seen at tx.get() time." }, { "id": "gemma4-31b-gguf", @@ -144,7 +150,7 @@ "tok_sec": 10.09, "total_tokens": 4536, "ttft_sec": 4.39, - "speed_caveat": "All Gemma 4 models ran abnormally slow (GPU offload appeared inactive despite being set), so tok/sec and TTFT are NOT representative of the model itself — likely an LM Studio/GGUF config issue. Treat speed numbers for the Gemma 4 batch as suspect.", + "speed_caveat": "All Gemma 4 models ran abnormally slow (GPU offload appeared inactive despite being set), so tok/sec and TTFT are NOT representative of the model itself \u2014 likely an LM Studio/GGUF config issue. Treat speed numbers for the Gemma 4 batch as suspect.", "filename": "outputs/gemma4-31b-gguf.py", "tests_pass": true, "total_score": 78, @@ -158,11 +164,11 @@ "verdict": "Minor Logic Flaws", "best_for": "Clean, correct, runnable code with solid O(1) structure and good concurrency granularity. A reliable pick for everyday caching/async work after a monotonic-clock + __slots__ pass.", "critical_bugs": [ - "Isolation leak: Transaction.get falls back to the PUBLIC cache.get, which calls _update_frequency — so reading a key inside a transaction mutates GLOBAL frequency state before commit, leaking uncommitted access patterns into global eviction order. Spec requires tx reads not to alter global freq.", - "Uses time.time() (system clock) throughout instead of time.monotonic() — NTP adjustments corrupt TTL eviction.", - "No __slots__ on Node/DoublyLinkedList/LFUCache/Transaction — rubric required it for memory efficiency.", + "Isolation leak: Transaction.get falls back to the PUBLIC cache.get, which calls _update_frequency \u2014 so reading a key inside a transaction mutates GLOBAL frequency state before commit, leaking uncommitted access patterns into global eviction order. Spec requires tx reads not to alter global freq.", + "Uses time.time() (system clock) throughout instead of time.monotonic() \u2014 NTP adjustments corrupt TTL eviction.", + "No __slots__ on Node/DoublyLinkedList/LFUCache/Transaction \u2014 rubric required it for memory efficiency.", "_delete_internal deliberately leaves empty frequency buckets in freq_map (documented but a minor memory leak: empty DoublyLinkedList objects accumulate).", - "Background evictor does list(self.cache.keys()) = O(N) snapshot every interval — a linear scan, forbidden by strict O(1).", + "Background evictor does list(self.cache.keys()) = O(N) snapshot every interval \u2014 a linear scan, forbidden by strict O(1).", "No MVCC/version check on commit (lost-update possible); commit is not exception-safe across the deletes-vs-puts loops.", "Tests pass but don't probe mid-commit read isolation or eviction-under-real-contention (capacity sized so all keys fit), so the isolation leak above goes undetected." ], @@ -192,10 +198,10 @@ "best_for": "Runnable and structurally sound, but the LFU eviction has a stale-min_freq capacity-breach path and the transaction API doesn't match the spec (no commit/rollback). Usable for prototypes if you fix eviction and re-skin transactions.", "critical_bugs": [ "Capacity breach: eviction does self.freq_map[self.min_freq].pop_tail() with NO guard that the bucket exists or is non-empty, and never prunes emptied freq buckets. After manual deletes empty the min-tier, pop_tail returns None silently -> eviction fails -> cache grows PAST capacity. This is the exact stale-min_freq capacity-breach the rubric flags.", - "Non-conformant transaction API: Transaction has NO commit() or rollback() method (both required by spec). Commit happens via a separate cache.apply_transaction_changes(tx._state) — wrong surface; the test only passes because it uses this internal path.", - "Isolation leak: Transaction.get falls back to the public cache.get, which bumps global frequency before commit — uncommitted tx reads alter global eviction order.", + "Non-conformant transaction API: Transaction has NO commit() or rollback() method (both required by spec). Commit happens via a separate cache.apply_transaction_changes(tx._state) \u2014 wrong surface; the test only passes because it uses this internal path.", + "Isolation leak: Transaction.get falls back to the public cache.get, which bumps global frequency before commit \u2014 uncommitted tx reads alter global eviction order.", "Duplicated eviction logic in apply_transaction_changes re-introduces the stale-min_freq bug at line 211.", - "Uses time.time() (system clock) via _get_now() instead of time.monotonic() — NTP jumps corrupt TTL (though _get_now is a clean single fix point).", + "Uses time.time() (system clock) via _get_now() instead of time.monotonic() \u2014 NTP jumps corrupt TTL (though _get_now is a clean single fix point).", "No __slots__ on Node/DoublyLinkedList/LFUCache/TransactionState/Transaction.", "Background sweep does list(self.cache.keys()) = O(N) per interval.", "Tests pass but use the non-spec commit path and don't probe capacity breach or isolation leak." @@ -222,15 +228,15 @@ "test_integrity": 4 }, "verdict": "Critical Bugs", - "best_for": "Not usable as-is — the cache cannot store its first key and the background evictor would crash the event loop. The smallest model in the set (12B) and lowest-quality output. Avoid for systems work.", + "best_for": "Not usable as-is \u2014 the cache cannot store its first key and the background evictor would crash the event loop. The smallest model in the set (12B) and lowest-quality output. Avoid for systems work.", "critical_bugs": [ "FATAL: put() line 112 does 'bucket = self.freq_buckets[self.min_freq]' after setting min_freq=1 but NEVER creates freq_buckets[1] -> KeyError: 1 on the very first put. Cache is unusable.", "Background evictor is fundamentally broken: start_evictor defines a SYNC 'def evict_loop' and passes it to create_task; inside it calls blocking time.sleep(interval) (freezes the event loop) AND asyncio.run(...) from within a running loop -> RuntimeError. Would crash hard if ever reached.", - "Eviction-by-re-put: expired keys are 'evicted' by re-inserting them with TTL 0 (line 122) instead of deleting them — wrong semantics and triggers immediate re-eviction.", + "Eviction-by-re-put: expired keys are 'evicted' by re-inserting them with TTL 0 (line 122) instead of deleting them \u2014 wrong semantics and triggers immediate re-eviction.", "Transaction duplicates the entire LFU machinery (local_cache + local_freq + local_min_freq) for snapshot isolation, but _update_local_freq has the same missing-bucket KeyError (line 140).", "Class name typo 'DoublyLinkedListList' (doubled word).", "No __slots__; time.time() (not monotonic) throughout.", - "Tests cannot run — crash at first put." + "Tests cannot run \u2014 crash at first put." ], "patch_code": "# FIX 1 (the fatal KeyError): create the bucket before use.\n# In put(), new-key branch:\n- bucket = self.freq_buckets[self.min_freq]\n+ bucket = self.freq_buckets.setdefault(self.min_freq, DoublyLinkedListList())\n# Same fix in _update_freq and Transaction._update_local_freq (use setdefault).\n\n# FIX 2 (the broken evictor): make it a real async task that deletes:\nasync def _evict_loop(self, interval):\n while True:\n await asyncio.sleep(interval) # async, non-blocking\n now = time.monotonic()\n async with self.global_lock:\n expired = [k for k, n in list(self.cache.items()) if now > n.ttl_expiry]\n for k in expired:\n node = self.cache.pop(k, None)\n if node:\n self.freq_buckets[node.freq].remove(node) # DELETE, not re-put\n\nasync def start_evictor(self, interval=1.0):\n self.evictor_task = asyncio.create_task(self._evict_loop(interval))\n\n# FIX 3: delete expired keys; do NOT re-insert with TTL 0.\n# FIX 4: time.time() -> time.monotonic().\n# FIX 5: add __slots__ to Node / DoublyLinkedListList / ConcurrentLFUCache / Transaction." }, @@ -243,7 +249,7 @@ "tok_sec": null, "total_tokens": null, "ttft_sec": null, - "speed_caveat": "Cloud model (run via opencode, not LM Studio) — tok_sec/tokens/TTFT are N/A (not measured for cloud). Included as a quality baseline against the local models. NOTE: it took 3 attempts to produce any output and ~12 minutes of thinking before succeeding — so it is a QUALITY benchmark, not a speed/usability one.", + "speed_caveat": "Cloud model (run via opencode, not LM Studio) \u2014 tok_sec/tokens/TTFT are N/A (not measured for cloud). Included as a quality baseline against the local models. NOTE: it took 3 attempts to produce any output and ~12 minutes of thinking before succeeding \u2014 so it is a QUALITY benchmark, not a speed/usability one.", "filename": "deepseekv4flash.py", "tests_pass": true, "total_score": 91, @@ -255,15 +261,109 @@ "test_integrity": 18 }, "verdict": "Production Ready", - "best_for": "Reference-quality baseline (91/100) — the bar the local models are measured against. Only submission with __slots__ + time.monotonic() + delta-based transactional frequency accounting. Sets the ceiling for correctness, though its unreliability (3 attempts, 12-min think time) makes it a poor *local* daily-driver.", + "best_for": "Reference-quality baseline (91/100) \u2014 the bar the local models are measured against. Only submission with __slots__ + time.monotonic() + delta-based transactional frequency accounting. Sets the ceiling for correctness, though its unreliability (3 attempts, 12-min think time) makes it a poor *local* daily-driver.", "critical_bugs": [ - "Two min(self._freq_to_list) linear scans in _evict_one's defensive recovery path (lines 415, 429) — only triggered when min_freq desyncs, not per-operation, but still a non-O(1) path. Could be replaced with incremental tracking.", - "Single coarse lock held across the whole commit-apply loop — not the fine-grained locking the prompt asked for.", + "Two min(self._freq_to_list) linear scans in _evict_one's defensive recovery path (lines 415, 429) \u2014 only triggered when min_freq desyncs, not per-operation, but still a non-O(1) path. Could be replaced with incremental tracking.", + "Single coarse lock held across the whole commit-apply loop \u2014 not the fine-grained locking the prompt asked for.", "__slots__ present on _Node and _DLL but not extended to Transaction / LFUCache.", "No explicit lost-update/conflict abort on commit (delta-based freq is applied unconditionally).", "Tests pass 20/20 but don't include a mid-commit read-isolation probe or adversarial eviction-under-contention stress." ], "patch_code": "# These are minor refinements on an already production-ready file.\n\n# FIX 1: eliminate the recovery min() scans by keeping min_freq\n# strictly in sync on every insert/bump/remove (it already does on\n# the hot path), so the _evict_one recovery branch is unreachable and\n# can assert instead of scanning:\nassert self._min_freq in self._freq_to_list or not self._freq_to_list\n\n# FIX 2: extend __slots__ to Transaction and LFUCache.\nclass LFUCache(Generic[KT, VT]):\n __slots__ = ('_capacity','_key_to_node','_freq_to_list','_min_freq',\n '_lock','_ttl_index','_evictor_task','_closed')\n\n# FIX 3 (optional): on commit, if a key's global node changed since the\n# tx snapshot, raise LFUCacheError('lost update') instead of overwriting." + }, + { + "id": "gemma4-26b-a4b-8bit-mlx", + "timestamp": "2026-07-28T16:25:00Z", + "model_name": "Gemma 4 26B-A4B", + "quant": "8-bit", + "param_size": "26B-A4B (MoE)", + "format": "mlx", + "tok_sec": 58.3, + "total_tokens": 7390, + "ttft_sec": 0.93, + "filename": "outputs/gemma4-26b-a4b-8bit-mlx.py", + "tests_pass": true, + "total_score": 82, + "breakdown": { + "complexity": 17, + "concurrency": 16, + "isolation": 18, + "memory_edge_cases": 15, + "test_integrity": 16 + }, + "verdict": "Minor Logic Flaws", + "best_for": "Tied top local scorer (82). The only local model to use delta-based transactional frequency accounting (freq bumps deferred to commit), matching the cloud baseline's isolation approach. Reliable for async/ACID-pattern work after a __slots__ + min_freq-edge pass.", + "critical_bugs": [ + "Stale min_freq on manual delete: _remove_node_from_structures empties the min-freq bucket but does `pass` instead of recomputing min_freq (lines 199-202, documented as 'a simplification'). Correct only because eviction has a min_freq-in-freq_map guard + arbitrary-key fallback (lines 237-244) \u2014 latent fragility under concurrent deletes.", + "No __slots__ on Node/DoublyLinkedList/LFUCache/Transaction despite Generic dataclasses (rubric required it for memory efficiency).", + "Background evictor does list(self.cache_data.keys()) = O(N) snapshot every interval \u2014 a linear scan, forbidden by strict O(1).", + "No MVCC/version check on commit (lost-update possible if the global key changes between tx.get and commit); commit applies deletes->reads->puts without try/except, so a mid-commit exception leaves partial state." + ], + "patch_code": "# FIX 1 (stale min_freq): recompute or invalidate when the min bucket empties.\n# In _remove_node_from_structures, replace the `pass`:\nif dll.size == 0:\n del self.freq_map[node.freq]\n if self.min_freq == node.freq:\n # bump to next existing tier (frequencies are contiguous under normal use)\n self.min_freq = self.min_freq + 1 if (self.min_freq + 1) in self.freq_map else min(self.freq_map, default=1)\n\n# FIX 2: add __slots__ to Node, DoublyLinkedList, LFUCache, Transaction.\n# FIX 3: background evictor \u2014 maintain a _ttl_keys set and iterate IT in\n# batches instead of list(self.cache_data.keys()) to stay O(batch).\n# FIX 4: wrap commit's three loops in try/except with rollback semantics on failure." + }, + { + "id": "qwen3.6-27b-8bit-mlx", + "timestamp": "2026-07-28T16:30:00Z", + "model_name": "Qwen 3.6 27B", + "quant": "8-bit", + "param_size": "27B dense", + "format": "mlx", + "tok_sec": 12.29, + "total_tokens": 12790, + "ttft_sec": 2.9, + "speed_caveat": "This model 'thought' for 12m48s before producing output and ran at 12.29 tok/sec \u2014 anomalously slow for an 8-bit MLX on M3 Max. Likely an inference/quant issue worth investigating; the slow generation is NOT representative of normal 27B-8bit throughput.", + "filename": "outputs/qwen3.6-27b-8bit-mlx.py", + "tests_pass": true, + "total_score": 78, + "breakdown": { + "complexity": 17, + "concurrency": 16, + "isolation": 14, + "memory_edge_cases": 15, + "test_integrity": 16 + }, + "verdict": "Minor Logic Flaws", + "best_for": "Clean, correct, runnable \u2014 same tier as Gemma 4 31B (78). Good O(1) structure and concurrency granularity. Reliable for everyday async/caching work after a monotonic-clock + __slots__ + isolation pass. Caveat: was anomalously slow to generate.", + "critical_bugs": [ + "Isolation leak: Transaction.get falls back to the PUBLIC cache.get (line 78), which calls _update_freq \u2014 so reading a key inside a transaction mutates GLOBAL frequency state before commit, leaking uncommitted access patterns into global eviction order.", + "Uses time.time() (system clock) throughout instead of time.monotonic() \u2014 NTP adjustments corrupt TTL eviction.", + "No __slots__ on Node/DoublyLinkedList/LFUCache/Transaction \u2014 rubric required it for memory efficiency.", + "Background _evict_loop materializes list(self.nodes.keys()) (O(N)) before checking only batch_size keys \u2014 the break caps work but not the snapshot cost, a hidden O(N) per sweep.", + "Commit re-implements put+evict inline (lines 102-118) duplicating the public path = duplicated bug surface; not wrapped in try/except so a mid-commit exception leaves partial state. No MVCC version check (lost-update possible)." + ], + "patch_code": "# FIX 1 (isolation leak): add a read-only global lookup (no freq bump)\n# and use it in tx.get instead of the public cache.get:\nasync def _read_raw(self, key):\n async with self.lock:\n node = self.nodes.get(key)\n if node is None: return None\n if time.monotonic() > node.expires_at:\n self._remove_node(key); return None\n return node.value\n# then: return await self._cache._read_raw(key)\n# FIX 2: time.time() -> time.monotonic() everywhere.\n# FIX 3: add __slots__ to Node, DoublyLinkedList, LFUCache, Transaction.\n# FIX 4: maintain a _ttl_keys set; iterate IT (batched) in the bg loop\n# instead of list(self.nodes.keys()).\n# FIX 5: factor commit's put/evict to reuse the internal helpers; wrap\n# the commit loop in try/except with rollback-on-failure." + }, + { + "id": "qwen3-coder-30b-6bit-mlx", + "timestamp": "2026-07-28T16:35:00Z", + "model_name": "Qwen3 Coder 30B", + "quant": "6-bit", + "param_size": "30B", + "format": "mlx", + "tok_sec": 72.7, + "total_tokens": 2779, + "ttft_sec": 0.9, + "filename": "outputs/qwen3-coder-30b-6bit-mlx.py", + "tests_pass": false, + "total_score": 50, + "breakdown": { + "complexity": 15, + "concurrency": 9, + "isolation": 11, + "memory_edge_cases": 11, + "test_integrity": 4 + }, + "verdict": "Critical Bugs", + "best_for": "Not usable as-is \u2014 transactions crash immediately due to an async/sync lock mismatch. The coder-specialist produced terse, fast output (2779 tok, 72.7 t/s) with competent freq-bucket structure, but fumbled the async primitive. Fix the one lock bug and it would likely score 70+.", + "critical_bugs": [ + "FATAL: _transaction_lock is an asyncio.Lock() (line 101) but begin_transaction() is a SYNC def that uses synchronous `with self._transaction_lock:` (line 199). asyncio.Lock does not support the sync context-manager protocol -> TypeError on the first transaction, crashing the entire test suite.", + "Test suite cannot run: crashes at cache.begin_transaction() in main(); the transaction and concurrency assertions never execute.", + "Unused `import threading` and `import weakref` \u2014 vestigial confusion between threading and asyncio primitives.", + "Uses time.time() (system clock) throughout instead of time.monotonic() \u2014 NTP jumps corrupt TTL eviction.", + "No __slots__ on CacheNode/FrequencyBucket/InMemoryLFUCache/Transaction \u2014 rubric required it.", + "FrequencyBucket stores nodes in BOTH a DLL and a parallel `nodes` dict \u2014 redundant memory per bucket." + ], + "patch_code": "# FIX 1 (the fatal crash): make begin_transaction async and use async with.\nasync def begin_transaction(self) -> 'Transaction':\n async with self._transaction_lock:\n self._transaction_counter += 1\n tx = Transaction(self)\n self._transactions[self._transaction_counter] = tx\n return tx\n# (and update callers: `tx = await cache.begin_transaction()`)\n# Alternative if sync creation is required: use threading.Lock for the\n# counter, but that is wrong in an asyncio codebase \u2014 go async.\n\n# FIX 2: remove unused `import threading` and `import weakref`.\n# FIX 3: time.time() -> time.monotonic() everywhere.\n# FIX 4: add __slots__ to all classes.\n# FIX 5: drop the redundant FrequencyBucket.nodes dict; the DLL already\n# tracks membership, so the dict is duplicate storage." } ] -} +} \ No newline at end of file diff --git a/generate_dashboard.py b/generate_dashboard.py index f992e71..68f1d84 100644 --- a/generate_dashboard.py +++ b/generate_dashboard.py @@ -4,7 +4,7 @@ Generator: builds dashboard.html + pages/..html from data/benchmark_histor Re-run after each grading batch to regenerate everything. Cyberpunk-terminal aesthetic. Pure stdlib + Chart.js via CDN. """ -import json, html, os, sys +import json, html, os, sys, re as _re HERE = os.path.dirname(os.path.abspath(__file__)) DATA = os.path.join(HERE, "data", "benchmark_history.json") @@ -98,11 +98,28 @@ h1{font-size:1.5rem; margin:.3em 0 0; letter-spacing:.02em; text-shadow:0 0 16px .stat::after{content:"";position:absolute;left:0;top:0;bottom:0;width:3px;background:var(--cyan);box-shadow:0 0 12px var(--cyan)} .stat .lbl{color:var(--dim);font-size:.68rem;letter-spacing:.18em;text-transform:uppercase} .stat .num{font-size:1.6rem;margin-top:4px;color:var(--ink)} +/* hero top row: squished stat tiles beside the radar */ +.toprow{display:grid;grid-template-columns:1.35fr 1fr;gap:18px;align-items:stretch;margin:18px 0 22px} +.stats-squish{display:grid;grid-template-columns:repeat(2,1fr);gap:12px;align-content:start} +.stats-squish .stat{padding:16px 18px} +.stats-squish .stat .num{font-size:1.9rem} +.radar-panel{display:flex;flex-direction:column} +.radar-panel .chart-box{flex:1;min-height:220px} +/* compact top-performers list */ +.toplist{display:flex;flex-direction:column;gap:6px} +a.tr{display:grid;grid-template-columns:28px 1fr auto auto auto;align-items:center;gap:10px;padding:8px 12px;background:rgba(0,0,0,0.2);border:1px solid rgba(255,255,255,0.05);border-radius:5px;transition:all .16s;cursor:pointer;text-decoration:none;font-size:.84rem} +a.tr:hover{border-color:rgba(0,255,200,0.4);background:rgba(0,255,200,0.05)} +.tr-r{color:var(--dim)} +.tr-n{color:var(--ink)} +.tr-q{color:var(--dim);font-size:.74rem} +.tr-s{color:var(--mag);font-size:.78rem} +.tr-sc{font-size:1rem;font-weight:600;text-align:right;min-width:30px} +@media(max-width:900px){.toprow{grid-template-columns:1fr}.grid2{grid-template-columns:1fr}.stats,.stats-squish{grid-template-columns:repeat(2,1fr)}} .grid2{display:grid;grid-template-columns:1.1fr .9fr;gap:18px;margin-bottom:26px} .panel{background:var(--panel);border:1px solid rgba(255,255,255,0.07);border-radius:8px;padding:18px} .panel h2{font-size:.95rem;letter-spacing:.12em;text-transform:uppercase;color:var(--cyan);margin:0 0 14px;text-shadow:0 0 10px rgba(0,255,200,0.35)} .chart-box{position:relative;height:340px} -@media(max-width:900px){.grid2{grid-template-columns:1fr}.stats{grid-template-columns:repeat(2,1fr)}} +@media(max-width:900px){.grid2{grid-template-columns:1fr}} /* leaderboard */ table{width:100%;border-collapse:collapse;font-size:.86rem} thead th{text-align:left;color:var(--dim);font-size:.66rem;letter-spacing:.16em;text-transform:uppercase;border-bottom:1px solid rgba(0,255,200,0.2);padding:8px 10px} @@ -122,6 +139,23 @@ tbody tr:hover{background:rgba(0,255,200,0.05);box-shadow:inset 0 0 0 1px rgba(0 tr:hover .score-bar > i{box-shadow:0 0 12px currentColor} .caveat{color:var(--amber);font-size:.72rem} .cloud-tag{color:var(--blue);font-size:.7rem;border:1px solid rgba(91,139,255,.4);padding:1px 6px;border-radius:3px;margin-left:6px} +.quant-cell{color:var(--ink);font-size:.82rem;white-space:nowrap} +.fmt-chip{display:inline-block;font-size:.66rem;letter-spacing:.08em;padding:2px 7px;border-radius:3px;border:1px solid currentColor;font-family:'Fira Code',monospace} +.fmt-mlx{color:var(--cyan);background:rgba(0,255,200,0.08)} +.fmt-gguf{color:var(--mag);background:rgba(255,43,214,0.08)} +.fmt-cloud{color:var(--blue);background:rgba(91,139,255,0.08)} +/* format/quant showdown cards */ +.fcard{background:var(--panel2);border:1px solid rgba(255,255,255,0.06);border-radius:6px;padding:12px 14px;margin-bottom:10px} +.fcard-h{display:flex;justify-content:space-between;align-items:baseline;gap:10px;flex-wrap:wrap;margin-bottom:9px} +.fcard-n{color:var(--ink);font-weight:600;font-family:'Fira Code',monospace;font-size:.92rem} +.fcard-meta{color:var(--dim);font-size:.72rem} +.fcard-v{display:flex;flex-wrap:wrap;gap:8px} +a.fv{display:grid;grid-template-columns:auto auto auto auto;align-items:center;gap:10px;padding:7px 11px;background:rgba(0,0,0,0.25);border:1px solid rgba(0,255,200,0.15);border-radius:5px;transition:all .18s;cursor:pointer;text-decoration:none} +a.fv:hover{border-color:var(--cyan);box-shadow:0 0 12px rgba(0,255,200,0.3);background:rgba(0,255,200,0.06)} +.fv-q{color:var(--ink);font-size:.82rem;min-width:64px} +.fv-f{font-size:.7rem;letter-spacing:.08em;font-family:'Fira Code',monospace} +.fv-s{color:var(--dim);font-size:.78rem} +.fv-sc{font-size:1rem;font-weight:600;min-width:28px;text-align:right} footer{color:var(--dim);font-size:.74rem;margin-top:40px;border-top:1px solid rgba(255,255,255,0.06);padding-top:14px;text-align:center} @media (prefers-reduced-motion: reduce){*{animation:none!important;transition:none!important}} """ @@ -165,21 +199,38 @@ def render_dashboard(data): caveat = "" if m.get("speed_caveat"): caveat = '
⚠ speed suspect
' - cloud = 'CLOUD' if m.get("format") == "cloud" else "" + fmt = (m.get("format") or "").lower() + if fmt == "cloud": + fmt_chip = 'CLOUD' + elif fmt == "gguf": + fmt_chip = 'GGUF' + elif fmt == "mlx": + fmt_chip = 'MLX' + else: + fmt_chip = f'{esc(m.get("format") or "—")}' bar_color = col rows.append(f""" #{i} -
{esc(m['model_name'])}{cloud}
{esc(m['quant'])}
{caveat} +
{esc(m['model_name'])}
{caveat} + {esc(m['quant'])} + {fmt_chip} {speed_str(m)} t/s
{m['total_score']}
{chip} - {esc(m['best_for'])[:70]}… + {esc(m['best_for'])[:60]}… DECODE ▸ """) # JSON for charts - bar_labels = json.dumps([m["model_name"].split("(")[0].strip()[:18] for m in local_sorted]) + def short_label(m): + # family + quant so duplicates (same model, different quant) are distinguishable + fam = m["model_name"].split("(")[0].strip() + fam = _re.sub(r"(?i)\s+(uncensored|heretic|aggressive|hauhaucs).*$", "", fam) + q = (m.get("quant") or "").strip() + name = f"{fam} · {q}" if q else fam + return name[:28] + bar_labels = json.dumps([short_label(m) for m in local_sorted]) bar_speed = json.dumps([m["tok_sec"] for m in local_sorted]) bar_score = json.dumps([m["total_score"] for m in local_sorted]) @@ -195,14 +246,87 @@ def render_dashboard(data): }) radar_json = json.dumps(radar_sets) - stats = f""" + stats_tiles = f"""
Models Tested
{n}
Top Score
{top['total_score'] if top else '—'}
Average
{avg:.1f}
Prod-Ready
{prod}/{n}
""" + stats = f'
{stats_tiles}
' top_name = esc(top["model_name"]) if top else "—" + # compact top-5 list for the right-of-radar panel + trows = [] + for i, m in enumerate(models[:5], 1): + col, _ = verdict_meta(m["verdict"]) + speed = f"{m['tok_sec']:.1f}" if m.get("tok_sec") is not None else "—" + trows.append( + f'' + f'#{i}' + f'{esc(m["model_name"][:20])}' + f'{esc(m["quant"])}' + f'{speed}' + f'{m["total_score"]}' + ) + toplist = ''.join(trows) + + # ---- Format / Quant showdown: group local models by base family ---- + def family_of(m): + # strip parentheticals, uncensored/merge tags, quant words, format, params + name = m["model_name"].split("(")[0].strip() + name = _re.sub(r"(?i)\b(uncensored|heretic|aggressive|hauhaucs|coder|composer|fable5|v\d+\.\d+|merge)\b", "", name) + name = _re.sub(r"\b\d+(\.\d+)?[bB](-[aA]\d+[bB])?\b", "", name) # sizes: 35B, 26B-A4B + name = _re.sub(r"\s+", " ", name).strip(" -") + # collapse known families + for fam in ["Qwen 3.6", "Qwen3", "Gemma 4", "KAT-Coder", "DeepSeek"]: + if _re.sub(r"\s+", "", name).lower().startswith(_re.sub(r"\s+", "", fam).lower()): + return fam + return name or m["model_name"] + + families = {} + for m in models: + if m.get("format") == "cloud": + continue + f = family_of(m) + families.setdefault(f, []).append(m) + # only show families with >=2 variants (the interesting comparisons) + multi = {f: ms for f, ms in families.items() if len(ms) >= 2} + + if multi: + fcards = [] + for fam, ms in sorted(multi.items(), key=lambda kv: -max(x["total_score"] for x in kv[1])): + ms_sorted = sorted(ms, key=lambda x: -x["total_score"]) + best = ms_sorted[0] + spread = max(x["total_score"] for x in ms) - min(x["total_score"] for x in ms) + variants = [] + for x in ms_sorted: + fmt = (x.get("format") or "").upper() + fcol = "var(--cyan)" if x.get("format") == "mlx" else ("var(--mag)" if x.get("format") == "gguf" else "var(--dim)") + col, _ = verdict_meta(x["verdict"]) + speed = f"{x['tok_sec']:.0f} t/s" if x.get("tok_sec") is not None else "—" + variants.append( + f'' + f'{esc(x["quant"])}' + f'{fmt}' + f'{speed}' + f'{x["total_score"]}' + f'' + ) + fcards.append(f""" +
+
{esc(fam)} + {len(ms)} variants · score spread {spread} · best {best["total_score"]} ({esc(best["quant"])})
+
{''.join(variants)}
+
""") + family_panel = f""" +
+

▮ FORMAT & QUANT SHOWDOWN — same family, different quants/formats

+
Families with 2+ variants. Click a row for the full audit. Compare how quant depth and MLX-vs-GGUF change the score.
+ {''.join(fcards)} +
""" + else: + family_panel = "" + body = f""" {head_html("LLM Benchmark Suite")}
@@ -211,27 +335,27 @@ def render_dashboard(data):
{n} models graded on a strict 5-pillar / 100-pt rubric · O(1) LFU + ACID transactions · M3 Max · LM Studio  ·  TOP: {top_name}
-{stats} -
-
-

▮ Score vs Throughput (tok/sec)

-
-
Local models only — cloud baseline (DeepSeek) excluded from speed axis. Gemma 4 bars flagged ⚠ (GPU-offload suspect).
-
-
-

▮ 5-Pillar Radar — Top 3

-
-
Each pillar scored 0–20. Outer = stronger.
+
+
{stats_tiles}
+
+

▮ 5-PILLAR RADAR — TOP 3

+
+
+

▮ SCORE vs THROUGHPUT (tok/sec)

+
+
Local models only — cloud baseline (DeepSeek) excluded from the speed axis. Bars flagged ⚠ have suspected GPU-offload / inference issues (not representative of the model).
+

▮ LEADERBOARD

- + {''.join(rows)}
#ModelSpeedScoreVerdictBest For
#ModelQuantFormatSpeedScoreVerdictBest For
+{family_panel}
Generated from data/benchmark_history.json · re-run generate_dashboard.py to refresh · cyberpunk-terminal UI