PageSourceSearch

https://ethpandaops.io/assets/js/57fd304a.58c928f1.js

js ethpandaops.io collected 2026-10-03 21:46:35 UTC 14,870 bytes, 1 lines download raw bytes

1"use strict";(globalThis.webpackChunkpublic_docs=globalThis.webpackChunkpublic_docs||[]).push([[4265],{14028(e,t,i){i.r(t),i.d(t,{API:()=>h,Agentic:()=>d,assets:()=>l,contentTitle:()=>o,default:()=>u,frontMatter:()=>r,metadata:()=>s,toc:()=>c});var s=i(77595),n=i(74848),a=i(28453);const r={slug:"introducing-ethiq",title:"Introducing EthIQ",authors:["samcm","parithosh","qu0b"],description:"An Ethereum protocol knowledge benchmark for AI models",tags:["ethiq","ai","ethereum","benchmark"],image:"/img/blog/introducing-ethiq/performance-chart.png"},o=void 0,l={authorsImageUrls:[void 0,void 0,void 0]},h=({children:e})=>(0,n.jsx)("span",{style:{color:"#4a7c59",fontWeight:"bold"},children:e||(0,n.jsxs)(n.Fragment,{children:[(0,n.jsx)("span",{style:{display:"inline-block",width:"0.55em",height:"0.55em",borderRadius:"50%",backgroundColor:"#4a7c59",marginRight:"0.25em",verticalAlign:"middle"}}),"API"]})}),d=({children:e})=>(0,n.jsx)("span",{style:{color:"#7c4a6e",fontWeight:"bold"},children:e||(0,n.jsxs)(n.Fragment,{children:[(0,n.jsx)("span",{style:{display:"inline-block",width:"0.55em",height:"0.55em",borderRadius:"50%",border:"2px solid #7c4a6e",marginRight:"0.25em",verticalAlign:"middle"}}),"Agentic"]})}),c=[{value:"Motivation",id:"motivation",level:2},{value:"Questions",id:"questions",level:2},{value:"The Results",id:"the-results",level:2},{value:"Frontier",id:"frontier",level:3},{value:"Open Weights",id:"open-weights",level:3},{value:"The Canary \ud83d\udc26",id:"the-canary-bird",level:3},{value:"Try it",id:"try-it",level:2}];function g(e){const t={a:"a",code:"code",h2:"h2",h3:"h3",p:"p",strong:"strong",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",...(0,a.R)(),...e.components},{ClickableImage:i}=t;return i||function(e,t){throw new Error("Expected "+(t?"component":"object")+" `"+e+"` to be defined: you likely forgot to import, pass, or provide it.")}("ClickableImage",!0),(0,n.jsxs)(n.Fragment,{children:[(0,n.jsxs)(t.p,{children:["If you ask an AI model about consensus constants, EVM execution, or state transitions, does it actually know what it's talking about? We built ",(0,n.jsx)(t.a,{href:"https://ethiq.ethpandaops.io",children:"EthIQ"}),", a benchmark that measures how well AI models understand Ethereum protocol internals."]}),"\n",(0,n.jsxs)(t.p,{children:["Models are tested in two modes: ",(0,n.jsx)(h,{})," (direct API calls with a system prompt, no tools) and ",(0,n.jsx)(d,{})," (CLI tools like Claude Code and Codex running in sandboxed Docker containers with bash, file I/O, and Node.js)."]}),"\n",(0,n.jsx)(t.h2,{id:"motivation",children:"Motivation"}),"\n",(0,n.jsx)(t.p,{children:"LLMs are now a daily tool at ethPandaOps, and we needed a way to evaluate them in our specific context. EthIQ gives us a quick signal when a new model releases, and a meaningful eval suite to drive improvements as we invest more in agentic tooling (prompt optimization, fine-tuning, workflows, etc.)"}),"\n",(0,n.jsx)(t.h2,{id:"questions",children:"Questions"}),"\n",(0,n.jsx)(t.p,{children:"The question set is designed to stress two distinct model capabilities: world knowledge (tested by categories like Constants) and raw reasoning (tested by auto-generated categories like EVM Execution). World knowledge questions are intentional. We want to explicitly probe what Ethereum protocol knowledge made it into a model's training data, since recall of these values is genuinely useful."}),"\n",(0,n.jsx)(t.p,{children:"To prevent memorization in the raw reasoning tasks, auto-generated questions start from official Ethereum spec test fixtures but mutate the inputs with a randomized seed (balances, storage values, calldata, deposit amounts, etc.) Ground-truth answers are then re-derived from the mutated inputs using the Python reference implementations. If a model saw the original fixtures during training, its answers won't match. New forks get a fresh seed, keeping the benchmark honest over time."}),"\n",(0,n.jsxs)(t.p,{children:["Questions are organized into datasets tied to Ethereum forks. The first dataset is ",(0,n.jsx)(t.strong,{children:"fusaka"}),", with 325 questions across these categories:"]}),"\n",(0,n.jsxs)(t.table,{children:[(0,n.jsx)(t.thead,{children:(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.th,{style:{textAlign:"left"},children:"Category"}),(0,n.jsx)(t.th,{style:{textAlign:"left"},children:"What it asks"}),(0,n.jsx)(t.th,{style:{textAlign:"left"},children:"Example"})]})}),(0,n.jsxs)(t.tbody,{children:[(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Constants"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Exact protocol constant values"}),(0,n.jsxs)(t.td,{style:{textAlign:"left"},children:["What is the value of ",(0,n.jsx)(t.code,{children:"SLOTS_PER_EPOCH"})," on Mainnet?"]})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"EVM execution"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Trace through bytecode, report the outcome"}),(0,n.jsxs)(t.td,{style:{textAlign:"left"},children:["Given ",(0,n.jsx)(t.code,{children:"0x..."}),", what's in storage slot 0x1?"]})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Consensus state transitions"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Apply slashings, deposits, etc. and compute resulting state"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"After this attester slashing, what is validator 6's balance?"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Consensus epoch processing"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Calculate rewards, penalties, and balance deltas"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"What are the reward/penalty deltas after processing this epoch?"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Consensus fork choice"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Replay block trees and determine the canonical head"}),(0,n.jsx)(t.td,{style:{textAlign:"left"}
1,children:"After these attestations, which block is the head?"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Consensus shuffling"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Compute validator committee assignments"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Which validators are in committee index 2 at slot 4?"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Calculations"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Multi-step arithmetic using protocol constants"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"How many seconds are in one Ethereum epoch?"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Conceptual"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Open-ended explanations graded by LLM rubric"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Explain how RANDAO bias works in validator shuffling"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Cross-fork"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"What changed between forks"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"What changed about max effective balance in Electra?"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"EIP interactions"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"How specific EIPs interact with each other"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"How do EIP-4844 blob commitments appear in beacon blocks?"})]}),(0,n.jsxs)(t.tr,{children:[(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Trick"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:"Questions with wrong premises"}),(0,n.jsx)(t.td,{style:{textAlign:"left"},children:'"When did Verkle Trees ship on mainnet?" (They didn\'t.)'})]})]})]}),"\n",(0,n.jsx)(t.h2,{id:"the-results",children:"The Results"}),"\n",(0,n.jsx)(i,{src:"/img/blog/introducing-ethiq/performance-chart.png",alt:"EthIQ performance chart showing model pass rates over time",caption:"Figure 1: Model performance by release date on the fusaka dataset (325 questions). Whiskers show 95% confidence intervals."}),"\n",(0,n.jsx)(t.h3,{id:"frontier",children:"Frontier"}),"\n",(0,n.jsxs)(t.p,{children:["In ",(0,n.jsx)(h,{})," mode, Anthropic's ",(0,n.jsx)(t.code,{children:"claude-opus-4.6-high"})," leads at 77.5%, followed by OpenAI's ",(0,n.jsx)(t.code,{children:"gpt-5.3-codex-xhigh"})," at 76.3% and Google's ",(0,n.jsx)(t.code,{children:"gemini-3.1-pro-preview-high"})," at 72.3%. ",(0,n.jsx)(t.code,{children:"claude-sonnet-4.6-high"})," is still in progress."]}),"\n",(0,n.jsxs)(t.p,{children:["When ",(0,n.jsx)(d,{})," runs are included, ",(0,n.jsx)(t.code,{children:"gpt-5.3-codex-xhigh"})," takes the lead at 83.7% via the Codex CLI."]}),"\n",(0,n.jsxs)(t.p,{children:["We observed some models (",(0,n.jsx)(t.a,{href:"https://ethiq.ethpandaops.io/model/minimax-m2.5-high?dataset=fusaka",children:(0,n.jsx)(t.code,{children:"minimax-m2.5-high"})}),") suffering degraded performance as their ",(0,n.jsx)(t.code,{children:"reasoning_effort"})," increased. After investigation, we found that these models were exhausting their maximum token output allocation, effectively thinking themselves to death."]}),"\n",(0,n.jsxs)(t.p,{children:[(0,n.jsxs)(t.a,{href:"https://ethiq.ethpandaops.io/performance?dataset=fusaka&prompt=bare&categories=evm-execution",children:["EVM execution questions in ",(0,n.jsx)(h,{})," mode"]})," are a particularly good vibe check for model capability. Watching ",(0,n.jsx)(t.code,{children:"Kimi K2.5"})," step through executing the EVM in it's thinking traces is quite an experience (read: concerning!) We felt bad asking ",(0,n.jsx)(t.code,{children:"llama-3.2-1b"})," to do the same. In general, we weren't expecting models to be so capable at executing the EVM in-context. Fortunately we capped the difficulty of the EVM questions at generation time (based on a few heuristics), so once this dataset set begins to saturate we can raise these limits and unleash a new ",(0,n.jsx)(t.code,{children:"very hard"})," class of questions."]}),"\n",(0,n.jsx)(t.h3,{id:"open-weights",children:"Open Weights"}),"\n",(0,n.jsx)(i,{src:"/img/blog/introducing-ethiq/openweights.png",alt:"EthIQ performance chart showing model pass rates over time for only open weight models",caption:"Figure 2: Open Weights model performance by release date on the fusaka dataset (325 questions). Whiskers show 95% confidence intervals."}),"\n",(0,n.jsxs)(t.p,{children:["Ethereum and open weights models go hand-in-hand, so we added the ability to ",(0,n.jsx)(t.a,{href:"https://ethiq.ethpandaops.io/performance?dataset=fusaka&prompt=bare&open_weights=true",children:"show just open weight models"}),". Kimi's ",(0,n.jsx)(t.code,{children:"k2.5-high"})," is the stand out amongst the open weight models, scoring 60.6%."]}),"\n",(0,n.jsxs)(t.p,{children:[(0,n.jsx)(t.code,{children:"minimax-m2.5-high"})," was disappointing, landing at 37.4%. This bulk of this disparity is in ",(0,n.jsx)(t.code,{children:"Consensus Constants"}),", with ",(0,n.jsx)(t.code,{children:"k2.5-high"})," at 94.7% compared to ",(0,n.jsx)(t.code,{children:"minimax-m2.5-high"})," at 58.9%. ",(0,n.jsx)(t.code,{children:"k2.5"})," is a much larger model at 1 trillion parameters versus ",(0,n.jsx)(t.code,{children:"minimax-m2.5"})," at 230 billion. World knowledge is an important factor when using an LLM for Ethereum!"]}),"\n",(0,n.jsxs)(t.h3,{id:"the-canary-bird",children:["The Canary ","\ud83d\udc26"]}),"\n",(0,n.jsxs)(t.p,{children:[(0,n.jsxs)(t.a,{href:"https://ethiq.ethpandaops.io/performance?dataset=fusaka&prompt=bare&categories=consensus-shuffling",children:["Thankfully all ",(0,n.jsx)(h,{})," evaluations returned failures for ",(0,n.jsx)(t.code,{children:"Consensus shuffling"})]}),". This would require the models to compute SHA256 in-context. If this canary die
1s you'll find the ethPandaOps team in a remote location far away from any electricity."]}),"\n",(0,n.jsx)(t.h2,{id:"try-it",children:"Try it"}),"\n",(0,n.jsxs)(t.p,{children:["Browse the full results at ",(0,n.jsx)(t.a,{href:"https://ethiq.ethpandaops.io",children:"ethiq.ethpandaops.io"}),". You can filter by question category, difficulty, and evaluation mode."]}),"\n",(0,n.jsx)(t.p,{children:"Keep an eye out as we'll be updating this as new forks ship and models are released!"})]})}function u(e={}){const{wrapper:t}={...(0,a.R)(),...e.components};return t?(0,n.jsx)(t,{...e,children:(0,n.jsx)(g,{...e})}):g(e)}},28453(e,t,i){i.d(t,{R:()=>r,x:()=>o});var s=i(96540);const n={},a=s.createContext(n);function r(e){const t=s.useContext(a);return s.useMemo(function(){return"function"==typeof e?e(t):{...t,...e}},[t,e])}function o(e){let t;return t=e.disableParentContext?"function"==typeof e.components?e.components(n):e.components||n:r(e.components),s.createElement(a.Provider,{value:t},e.children)}},77595(e){e.exports=JSON.parse('{"permalink":"/posts/introducing-ethiq","source":"@site/blog/2026-03-05-introducing-ethiq/index.md","title":"Introducing EthIQ","description":"An Ethereum protocol knowledge benchmark for AI models","date":"2026-03-05T00:00:00.000Z","tags":[{"inline":true,"label":"ethiq","permalink":"/posts/tags/ethiq"},{"inline":true,"label":"ai","permalink":"/posts/tags/ai"},{"inline":true,"label":"ethereum","permalink":"/posts/tags/ethereum"},{"inline":true,"label":"benchmark","permalink":"/posts/tags/benchmark"}],"readingTime":5.04,"hasTruncateMarker":false,"authors":[{"name":"samcm","title":"DevOps Engineer","bio":"DevOps Engineer","url":"https://github.com/samcm","github":"https://github.com/samcm","twitter":"https://x.com/samcmAU","imageURL":"/img/team/samcm.jpeg","key":"samcm","page":null},{"name":"parithosh","title":"DevOps Engineer","bio":"DevOps Engineer","url":"https://github.com/parithosh","github":"https://github.com/parithosh","twitter":"https://x.com/parithosh_j","website":"https://parithosh.com/","imageURL":"/img/team/parithosh.jpeg","key":"parithosh","page":null},{"name":"qu0b","title":"DevOps Engineer","bio":"DevOps Engineer","url":"https://github.com/qu0b","github":"https://github.com/qu0b","twitter":"https://x.com/stefan_star","imageURL":"/img/team/qu0b.jpeg","key":"qu0b","page":null}],"frontMatter":{"slug":"introducing-ethiq","title":"Introducing EthIQ","authors":["samcm","parithosh","qu0b"],"description":"An Ethereum protocol knowledge benchmark for AI models","tags":["ethiq","ai","ethereum","benchmark"],"image":"/img/blog/introducing-ethiq/performance-chart.png"},"unlisted":false,"prevItem":{"title":"Validator Report: Investigate Validator Performance","permalink":"/posts/validator-report"},"nextItem":{"title":"EVM Gas Profiling: New Execution Trace Data","permalink":"/posts/evm-gas-profiling"}}')}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.