Benchmark Evaluation Matrix
Data & TablesMulti-test LLM evals table testing accuracy, latency, and hallucination rate.
Interactive Live Preview
components/ui/eval-benchmark-table.tsxTypeScript · React 19
1"use client";23import * as React from "react";45interface EvalBenchmarkTableProps {6 variant?: string;7}89export default function EvalBenchmarkTable({ variant = "Matrix" }: EvalBenchmarkTableProps) {10 const [activeBenchmark, setActiveBenchmark] = React.useState<"SWE-Bench" | "HumanEval" | "Agent-Ops">("SWE-Bench");1112 const benchmarks = {13 "SWE-Bench": [14 { model: "Claude 3.5 Sonnet", passRate: "49.2%", latency: "240ms", hallucination: "1.2%", cost1k: "$0.003", status: "Leader" },15 { model: "GPT-4o (2024-11)", passRate: "43.8%", latency: "290ms", hallucination: "2.4%", cost1k: "$0.0025", status: "Optimal" },16 { model: "DeepSeek R1", passRate: "48.6%", latency: "480ms", hallucination: "1.6%", cost1k: "$0.0005", status: "Best Value" },17 { model: "Gemini 1.5 Pro", passRate: "41.4%", latency: "210ms", hallucination: "3.1%", cost1k: "$0.0012", status: "Fast" },18 ],19 HumanEval: [20 { model: "Claude 3.5 Sonnet", passRate: "93.7%", latency: "210ms", hallucination: "0.4%", cost1k: "$0.003", status: "Leader" },21 { model: "DeepSeek R1", passRate: "92.8%", latency: "420ms", hallucination: "0.6%", cost1k: "$0.0005", status: "Leader" },22 { model: "GPT-4o (2024-11)", passRate: "90.2%", latency: "260ms", hallucination: "1.1%", cost1k: "$0.0025", status: "Optimal" },23 { model: "Gemini 1.5 Pro", passRate: "88.9%", latency: "190ms", hallucination: "1.4%", cost1k: "$0.0012", status: "Fast" },24 ],25 "Agent-Ops": [26 { model: "Claude 3.5 Sonnet", passRate: "88.4%", latency: "260ms", hallucination: "0.8%", cost1k: "$0.003", status: "Leader" },27 { model: "GPT-4o (2024-11)", passRate: "86.1%", latency: "300ms", hallucination: "1.8%", cost1k: "$0.0025", status: "Optimal" },28 { model: "DeepSeek R1", passRate: "84.9%", latency: "510ms", hallucination: "1.1%", cost1k: "$0.0005", status: "Best Value" },29 { model: "Gemini 1.5 Pro", passRate: "82.0%", latency: "220ms", hallucination: "2.5%", cost1k: "$0.0012", status: "Fast" },30 ],31 };3233 const currentRows = benchmarks[activeBenchmark];3435 return (36 <div className="w-full max-w-120 overflow-hidden rounded-card bg-surface shadow-card border border-line">37 {/* Header */}38 <div className="primitive-card-bar flex flex-wrap items-center justify-between gap-2 border-b border-line bg-surface/90 px-4 py-3">39 <div className="flex items-center gap-2">40 <svg width="15" height="15" viewBox="0 0 24 24" fill="none" stroke="var(--accent)" strokeWidth="2.2" strokeLinecap="round" strokeLinejoin="round">41 <line x1="18" y1="20" x2="18" y2="10" />42 <line x1="12" y1="20" x2="12" y2="4" />43 <line x1="6" y1="20" x2="6" y2="14" />44 </svg>45 <span className="font-display font-semibold text-[14px] text-ink">46 LLM Benchmark Evaluation Matrix47 </span>48 </div>4950 {/* Benchmark Switcher */}51 <div className="flex items-center gap-1 font-mono text-[10.5px]">52 {(["SWE-Bench", "HumanEval", "Agent-Ops"] as const).map((b) => (53 <button54 key={b}55 onClick={() => setActiveBenchmark(b)}56 className={`rounded-chip px-2.5 py-0.5 transition-colors border ${57 activeBenchmark === b58 ? "bg-accent text-accent-ink font-bold border-accent"59 : "bg-field text-ink-3 hover:text-ink border-line"60 }`}61 >62 {b}63 </button>64 ))}65 </div>66 </div>6768 {/* Table Grid */}69 <div className="overflow-x-auto">70 <table className="w-full text-left font-mono text-[11.5px]">71 <thead className="border-b border-line bg-inset/60 text-ink-3 text-[10px] uppercase">72 <tr>73 <th className="py-2.5 px-3.5 font-medium">Model</th>74 <th className="py-2.5 px-3 font-medium">Pass Rate</th>75 <th className="py-2.5 px-3 font-medium">Latency</th>76 <th className="py-2.5 px-3 font-medium">Hallucination</th>77 <th className="py-2.5 px-3 font-medium">Tier</th>78 </tr>79 </thead>80 <tbody className="divide-y divide-line/60">81 {currentRows.map((row) => (82 <tr key={row.model} className="hover:bg-hover/30 transition-colors">83 <td className="py-3 px-3.5 font-bold text-ink font-display text-[13px]">84 {row.model}85 </td>86 <td className="py-3 px-3 font-bold text-accent">87 {row.passRate}88 </td>89 <td className="py-3 px-3 text-ink-2">90 {row.latency}91 </td>92 <td className="py-3 px-3 text-emerald-400">93 {row.hallucination}94 </td>95 <td className="py-3 px-3">96 <span className="rounded-full bg-field px-2 py-0.5 text-[10px] text-ink-3 border border-line">97 {row.status}98 </span>99 </td>100 </tr>101 ))}102 </tbody>103 </table>104 </div>105106 {/* Footer */}107 <div className="flex items-center justify-between border-t border-line bg-inset/40 px-4 py-2 font-mono text-[10.5px] text-ink-3">108 <span>Evaluated across 1,200 test cases</span>109 <button className="text-accent hover:underline">Run Live Eval Suite →</button>110 </div>111 </div>112 );113}114
Props & Specification
| Prop | Type | Default | Description |
|---|---|---|---|
| variant | SWE-Bench | HumanEval | Agent-Ops | SWE-Bench | Visual state and layout presentation mode. |
| className | string | "" | Optional Tailwind CSS wrapper class overrides. |