Benchmark Evaluation Matrix

Data & Tables

Multi-test LLM evals table testing accuracy, latency, and hallucination rate.

Interactive Live Preview
components/ui/eval-benchmark-table.tsxTypeScript · React 19
1"use client";
2
3import * as React from "react";
4
5interface EvalBenchmarkTableProps {
6 variant?: string;
7}
8
9export default function EvalBenchmarkTable({ variant = "Matrix" }: EvalBenchmarkTableProps) {
10 const [activeBenchmark, setActiveBenchmark] = React.useState<"SWE-Bench" | "HumanEval" | "Agent-Ops">("SWE-Bench");
11
12 const benchmarks = {
13 "SWE-Bench": [
14 { model: "Claude 3.5 Sonnet", passRate: "49.2%", latency: "240ms", hallucination: "1.2%", cost1k: "$0.003", status: "Leader" },
15 { model: "GPT-4o (2024-11)", passRate: "43.8%", latency: "290ms", hallucination: "2.4%", cost1k: "$0.0025", status: "Optimal" },
16 { model: "DeepSeek R1", passRate: "48.6%", latency: "480ms", hallucination: "1.6%", cost1k: "$0.0005", status: "Best Value" },
17 { model: "Gemini 1.5 Pro", passRate: "41.4%", latency: "210ms", hallucination: "3.1%", cost1k: "$0.0012", status: "Fast" },
18 ],
19 HumanEval: [
20 { model: "Claude 3.5 Sonnet", passRate: "93.7%", latency: "210ms", hallucination: "0.4%", cost1k: "$0.003", status: "Leader" },
21 { model: "DeepSeek R1", passRate: "92.8%", latency: "420ms", hallucination: "0.6%", cost1k: "$0.0005", status: "Leader" },
22 { model: "GPT-4o (2024-11)", passRate: "90.2%", latency: "260ms", hallucination: "1.1%", cost1k: "$0.0025", status: "Optimal" },
23 { model: "Gemini 1.5 Pro", passRate: "88.9%", latency: "190ms", hallucination: "1.4%", cost1k: "$0.0012", status: "Fast" },
24 ],
25 "Agent-Ops": [
26 { model: "Claude 3.5 Sonnet", passRate: "88.4%", latency: "260ms", hallucination: "0.8%", cost1k: "$0.003", status: "Leader" },
27 { model: "GPT-4o (2024-11)", passRate: "86.1%", latency: "300ms", hallucination: "1.8%", cost1k: "$0.0025", status: "Optimal" },
28 { model: "DeepSeek R1", passRate: "84.9%", latency: "510ms", hallucination: "1.1%", cost1k: "$0.0005", status: "Best Value" },
29 { model: "Gemini 1.5 Pro", passRate: "82.0%", latency: "220ms", hallucination: "2.5%", cost1k: "$0.0012", status: "Fast" },
30 ],
31 };
32
33 const currentRows = benchmarks[activeBenchmark];
34
35 return (
36 <div className="w-full max-w-120 overflow-hidden rounded-card bg-surface shadow-card border border-line">
37 {/* Header */}
38 <div className="primitive-card-bar flex flex-wrap items-center justify-between gap-2 border-b border-line bg-surface/90 px-4 py-3">
39 <div className="flex items-center gap-2">
40 <svg width="15" height="15" viewBox="0 0 24 24" fill="none" stroke="var(--accent)" strokeWidth="2.2" strokeLinecap="round" strokeLinejoin="round">
41 <line x1="18" y1="20" x2="18" y2="10" />
42 <line x1="12" y1="20" x2="12" y2="4" />
43 <line x1="6" y1="20" x2="6" y2="14" />
44 </svg>
45 <span className="font-display font-semibold text-[14px] text-ink">
46 LLM Benchmark Evaluation Matrix
47 </span>
48 </div>
49
50 {/* Benchmark Switcher */}
51 <div className="flex items-center gap-1 font-mono text-[10.5px]">
52 {(["SWE-Bench", "HumanEval", "Agent-Ops"] as const).map((b) => (
53 <button
54 key={b}
55 onClick={() => setActiveBenchmark(b)}
56 className={`rounded-chip px-2.5 py-0.5 transition-colors border ${
57 activeBenchmark === b
58 ? "bg-accent text-accent-ink font-bold border-accent"
59 : "bg-field text-ink-3 hover:text-ink border-line"
60 }`}
61 >
62 {b}
63 </button>
64 ))}
65 </div>
66 </div>
67
68 {/* Table Grid */}
69 <div className="overflow-x-auto">
70 <table className="w-full text-left font-mono text-[11.5px]">
71 <thead className="border-b border-line bg-inset/60 text-ink-3 text-[10px] uppercase">
72 <tr>
73 <th className="py-2.5 px-3.5 font-medium">Model</th>
74 <th className="py-2.5 px-3 font-medium">Pass Rate</th>
75 <th className="py-2.5 px-3 font-medium">Latency</th>
76 <th className="py-2.5 px-3 font-medium">Hallucination</th>
77 <th className="py-2.5 px-3 font-medium">Tier</th>
78 </tr>
79 </thead>
80 <tbody className="divide-y divide-line/60">
81 {currentRows.map((row) => (
82 <tr key={row.model} className="hover:bg-hover/30 transition-colors">
83 <td className="py-3 px-3.5 font-bold text-ink font-display text-[13px]">
84 {row.model}
85 </td>
86 <td className="py-3 px-3 font-bold text-accent">
87 {row.passRate}
88 </td>
89 <td className="py-3 px-3 text-ink-2">
90 {row.latency}
91 </td>
92 <td className="py-3 px-3 text-emerald-400">
93 {row.hallucination}
94 </td>
95 <td className="py-3 px-3">
96 <span className="rounded-full bg-field px-2 py-0.5 text-[10px] text-ink-3 border border-line">
97 {row.status}
98 </span>
99 </td>
100 </tr>
101 ))}
102 </tbody>
103 </table>
104 </div>
105
106 {/* Footer */}
107 <div className="flex items-center justify-between border-t border-line bg-inset/40 px-4 py-2 font-mono text-[10.5px] text-ink-3">
108 <span>Evaluated across 1,200 test cases</span>
109 <button className="text-accent hover:underline">Run Live Eval Suite →</button>
110 </div>
111 </div>
112 );
113}
114

Props & Specification

PropTypeDefaultDescription
variantSWE-Bench | HumanEval | Agent-OpsSWE-BenchVisual state and layout presentation mode.
classNamestring""Optional Tailwind CSS wrapper class overrides.

Related Components in Data & Tables