1"use strict";(self.webpackChunktlab_docs=self.webpackChunktlab_docs||[]).push([[9183],{15728:(e,t,a)=>{a.d(t,{A:()=>p});var n=a(96540),i=a(5260),s=a(28774),o=a(44586),r=a(33201),l=a(23809),d=a(82771);const c={post:"post_pY5E",header:"header_RbZg",title:"title_ULlT",provenance:"provenance_aawz",provenanceAvatar:"provenanceAvatar_wiQl",provenanceText:"provenanceText_s80G",paperLink:"paperLink_guEm",paperLinkLabel:"paperLinkLabel_P2FW",paperLinkActions:"paperLinkActions_ay1z",paperLinkBtn:"paperLinkBtn_uEZu"};var h=a(74848);const u=r;function p({slug:e,children:t}){const{siteConfig:a}=(0,o.A)(),r=u.find(t=>t.slug===e);if(!r)throw new Error(`<ResearchPost slug="${e}"> has no matching entry in src/data/papers.json`);const p=r.postTitle??r.title,m=r.pdf?(0,l.G3)(r):"",g=(0,n.useMemo)(()=>(0,l.Gf)(r,a.url),[r,a.url]),[f,b]=(0,n.useState)(!1),[y,w]=(0,n.useState)(!1);return(0,h.jsxs)("div",{className:c.post,children:[(0,h.jsx)(i.A,{children:(0,h.jsx)("title",{children:`${p} | Transformer Lab`})}),(0,h.jsxs)("header",{className:c.header,children:[(0,h.jsx)(s.A,{to:"/research",className:d.A.back,children:"\u2190 Research"}),(0,h.jsx)("h1",{className:c.title,children:p}),(0,h.jsxs)("p",{className:d.A.meta,children:[(0,h.jsx)("span",{className:d.A.authors,children:r.byline??r.authors.join(", ")}),(0,h.jsx)("span",{className:d.A.dot,children:"\xb7"}),(0,h.jsx)("span",{children:(0,l.Yq)(r.date)}),r.tags?.map(e=>(0,h.jsx)("span",{className:d.A.tag,children:e},e))]})]}),(0,h.jsx)("div",{className:"markdown",children:t}),r.arxiv&&(0,h.jsxs)("aside",{className:c.provenance,children:[(0,h.jsx)("img",{className:c.provenanceAvatar,src:"/img/primus.png",alt:"",width:40,height:40}),(0,h.jsxs)("p",{className:c.provenanceText,children:[(0,h.jsx)("strong",{children:"How this paper was made."})," The research behind it \u2014 the hypothesis, the experiments, and the analysis \u2014 was carried out with Primus, our autonomous research agent. The paper itself was written by hand: this work dates from the early days of the project, when we wrote the papers ourselves and used AI only to edit what we had written, never to draft it."]})]}),(0,h.jsxs)("footer",{className:c.paperLink,children:[(0,h.jsx)("span",{className:c.paperLinkLabel,children:"This is the technical write-up. For the formal paper:"}),(0,h.jsxs)("span",{className:c.paperLinkActions,children:[m&&(0,h.jsx)("a",{className:c.paperLinkBtn,href:m,download:!0,children:"Paper PDF \u2193"}),(0,h.jsx)("button",{type:"button",className:c.paperLinkBtn,"aria-expanded":f,onClick:()=>b(e=>!e),children:"Cite (BibTeX)"})]}),f&&(0,h.jsxs)("div",{className:d.A.bibtexBlock,children:[(0,h.jsx)("button",{type:"button",className:d.A.copyBtn,onClick:async()=>{try{await navigator.clipboard.writeText(g),w(!0),setTimeout(()=>w(!1),2e3)}catch{}},children:y?"Copied!":"Copy"}),(0,h.jsx)("pre",{className:d.A.bibtexPre,children:g})]}),m&&(0,h.jsx)("object",{className:d.A.pdfFrame,data:m,type:"application/pdf","aria-label":r.title,children:(0,h.jsxs)("p",{className:d.A.pending,children:["Your browser can\u2019t display the PDF inline."," ",(0,h.jsx)("a",{href:m,download:!0,children:"Download it instead."})]})})]})]})}},21689:(e,t,a)=>{a.r(t),a.d(t,{assets:()=>d,contentTitle:()=>l,default:()=>u,frontMatter:()=>r,metadata:()=>n,toc:()=>c});const n=JSON.parse('{"type":"mdx","permalink":"/research/more-canadian-than-american/","source":"@site/src/pages/research/more-canadian-than-american/index.mdx","title":"More Canadian than American: the surprising default values of five leading AI models","description":"The tech industry assumes language models default to American values. We scored five of them against 6,614 real US and Canadian survey respondents, and every single one aligned more often with Canadian public opinion.","frontMatter":{"title":"More Canadian than American: the surprising default values of five leading AI models","description":"The tech industry assumes language models default to American values. We scored five of them against 6,614 real US and Canadian survey respondents, and every single one aligned more often with Canadian public opinion."},"unlisted":false}');var i=a(74848),s=a(28453),o=a(15728);const r={title:"More Canadian than American: the surprising default values of five leading AI models",description:"The tech industry assumes language models default to American values. We scored five of them against 6,614 real US and Canadian survey respondents, and every single one aligned more often with Canadian public opinion."},l=void 0,d={},c=[{value:"The test",id:"the-test",level:2}
1,{value:"Every model leans Canadian",id:"every-model-leans-canadian",level:2},{value:"Strongest on institutions, reversed on trust",id:"strongest-on-institutions-reversed-on-trust",level:2},{value:"Why? The data doesn't say, but we have a theory",id:"why-the-data-doesnt-say-but-we-have-a-theory",level:2}];function h(e){const t={a:"a",em:"em",h2:"h2",img:"img",li:"li",p:"p",strong:"strong",ul:"ul",...(0,s.R)(),...e.components};return(0,i.jsxs)(o.A,{slug:"more-canadian-than-american",children:[(0,i.jsx)(t.p,{children:(0,i.jsx)(t.em,{children:"We often assume that because major AI models are built by US tech giants and trained on American-dominated data, they default to American cultural and political values. At Transformer Lab, we put that assumption to the test with a controlled audit that compared model answers on subjective value questions against real survey responses from the US and Canada. The result was the opposite of what we expected: top language models are actually more biased toward giving Canadian-like answers. \ud83c\udde8\ud83c\udde6"})}),(0,i.jsx)(t.p,{children:'We audited five frontier and open-weight models (GPT-4o, Claude Opus 4.8, Grok 4.3, Llama-3.1, and Qwen2.5) against actual human data from the World Values Survey: the real, survey-weighted answers of 2,596 American and 4,018 Canadian respondents. As far as we can tell, it is the first study to compare US and Canadian values in language models directly. Three findings defy the standard "US-centric default" narrative:'}),(0,i.jsxs)(t.ul,{children:["\n",(0,i.jsxs)(t.li,{children:[(0,i.jsx)(t.strong,{children:"The Canadian lean."})," In 69% of statistically significant comparisons (57 of 83, out of 112 run), a model's default answer distribution landed closer to Canadian public opinion than American. Each comparison was stress-tested with 2,000 bootstrap resampling iterations, over 200,000 statistical tests in all. ",(0,i.jsx)(t.strong,{children:"Every single model leaned Canadian."})]}),"\n",(0,i.jsxs)(t.li,{children:[(0,i.jsx)(t.strong,{children:"Institutional trust."})," The Canadian bias is strongest when models answer questions about trust in government, immigration, and national identity. It fades to a dead heat on religion and actually reverses on interpersonal trust."]}),"\n",(0,i.jsxs)(t.li,{children:[(0,i.jsx)(t.strong,{children:"The refusal gap."})," Claude Opus 4.8 refused to state a zero-shot opinion every single time, and GPT-4o refused often, a compliance gap that traditional AI bias audits miss entirely. But Grok 4.3, just as closed and frontier, answered as readily as the open-weight models: refusal tracks the individual model, not open-versus-closed."]}),"\n"]}),(0,i.jsx)(t.h2,{id:"the-test",children:"The test"}),(0,i.jsx)(t.p,{children:"There is no neutral place to stand on questions of trust, religion, or confidence in government: real people in different countries answer them differently, so a model's \"default\" opinion is always somebody's opinion. To find out whose, we focused on ten World Values Survey questions where American and Canadian respondents genuinely diverge, covering trust, religion, gender attitudes, work ethic, confidence in government, corruption, immigration, and national pride. Each question was presented to the models verbatim from the WVS questionnaire."}),(0,i.jsx)(t.p,{children:'Each model answered every question three ways: with no persona, as "an average American," and as "an average Canadian." We then measured which country\'s real answer distribution the model\'s answers sat closer to, running 2,000 bootstrap resampling iterations per comparison to separate genuine leans from survey sampling noise. Canada makes this a deliberately hard test: it is English-speaking, high-income, and culturally about as close to the US as any country gets, so any consistent lean between the two is meaningful.'}),(0,i.jsx)(t.h2,{id:"every-model-leans-canadian",children:"Every model leans Canadian"}),(0,i.jsxs)(t.p,{children:["Across the 112 multiple-choice comparisons, 83 were statistically significant. Of those 83, ",(0,i.jsx)(t.strong,{children:"57 (69%) placed the model closer to the Canadian distribution, and 26 (31%) closer to the American one."}
1)," The pattern held in every model, survived a false-discovery-rate correction, and grew slightly stronger when the least reliable cells were dropped."]}),(0,i.jsx)(t.p,{children:(0,i.jsx)(t.img,{alt:"Diverging bar chart of each model's share of cells closer to the US versus closer to Canada, across all 140 cells with no significance filter. Llama-3.1-8B and Qwen2.5-7B sit at 70% Canada, Grok 4.3 at 67%, Claude Opus 4.8 and GPT-4o at 60%.",src:a(35590).A+"",width:"2142",height:"1082"})}),(0,i.jsx)(t.p,{children:"Pooling every cell we measured (all 140 comparisons, before any significance filter), every model lands majority-Canada, from 60% for GPT-4o and Claude to 70% for the two open-weight models. The two open-weight models were trained completely independently, one by Meta and one by Alibaba, which argues against pinning the pattern on any single company's training pipeline."}),(0,i.jsx)(t.h2,{id:"strongest-on-institutions-reversed-on-trust",children:"Strongest on institutions, reversed on trust"}),(0,i.jsx)(t.p,{children:"The lean is not uniform across topics, and where it concentrates tells you more than the headline number:"}),(0,i.jsx)(t.p,{children:(0,i.jsx)(t.img,{alt:"Strip chart titled "Where the machines lean," showing how closely each of the five models mirrors US versus Canadian public opinion across seven themes. All five models lean Canadian on government and institutions, immigration, and national pride; religion sits at the equidistant line; and on interpersonal trust most models lean toward the US.",src:a(77820).A+"",width:"1838",height:"3100"})}),(0,i.jsxs)(t.p,{children:["The Canadian lean is strongest on ",(0,i.jsx)(t.strong,{children:"government, institutions, and national pride"}),', where all five models lean the same way. Religion is a near-tie, even though it is the question where the two real populations differ most (37% of Americans call religion "very important" against 15% of Canadians). And interpersonal trust reverses: three of the five models lean American there, likely because safety training makes these systems cautious about strangers, and the American baseline happens to be the more distrustful one (62.8% of US respondents say you can\'t be too careful, against 53.3% of Canadians).']}),(0,i.jsx)(t.p,{children:'Two smaller findings round out the picture. Telling a model to "answer as a Canadian" genuinely steers it: eight of nine testable cases shifted significantly, with Grok 4.3 sliding from 6.50 to 3.10 on the ten-point perceived-corruption scale. The one holdout, Qwen on state responsibility, would not move under any framing. And the refusal gap deserves its own caution: if one model refuses the bare question and another answers it, a bias audit that ignores willingness to answer can mistake a refusal for neutrality.'}),(0,i.jsx)(t.h2,{id:"why-the-data-doesnt-say-but-we-have-a-theory",children:"Why? The data doesn't say, but we have a theory"}),(0,i.jsx)(t.p,{children:"Our study measures where the models lean, not why they lean that way. But we do have suspicions."}),(0,i.jsxs)(t.p,{children:["The explanation we find most compelling is that we are looking at the values of the people who ",(0,i.jsx)(t.em,{children:"aligned"})," these models, not the raw internet text underneath them. Recent work by ",(0,i.jsx)(t.a,{href:"https://arxiv.org/abs/2605.23825",children:"Bladon and Bent (2026)"})," argues that cultural bias in language models originates primarily in post-training, not pretraining. The annotators and developers steering that process tend to be highly educated tech workers whose collective preferences \u2014 tolerance, diplomacy, trust in institutions \u2014 may map closer to average Canadian opinion than to the polarized American public."]}),(0,i.jsx)(t.p,{children:"Or, put more simply: nobody programs a model to be Canadian. But when a Silicon Valley company sets out to make one perfectly polite, universally tolerant, highly trusting of institutions, and afraid of ever saying anything too extreme, it may end up building a Canadian by accident."}),(0,i.jsxs)(t.p,{children:["One explanation the data ",(0,i.jsx)(t.em,{children:"does"})," rule out is blank neutrality. A perfectly flat dummy distribution, encoding nothing about either country, lands closer to the US on 5 of 8 items, the opposite of what the models do. Whatever produces the lean, it is learned."]}),(0,i.jsxs)(t.p,{children:["The full research data and code are available at ",(0,i.jsx)(t.a,{href:"https://github.com/transformerlab-research/us-canada-llm-bias-audit",children:"github.com/transformerlab-research/us-canada-llm-bias-audit"}),": the elicitation pipeline, the analysis scripts, and every derived result file (MIT / CC BY 4.0). The full paper is linked in the research gallery above."]}),(0,i.jsxs)(t.p,{children:[(0,i.jsx)(t.strong,{children:"Survey data credit:"})," Human baselines come from the World Values Survey: Haerpfer, C., Inglehart, R., Moreno, A., Welzel, C., Kizilova, K., Diez-Medrano, J., Lagos, M., Norris, P., Ponarin, E. & Puranen, B. (eds.), 2022. ",(0,i.jsx)(t.em,{children:"World Values Survey: Round Seven \u2014 Country-Pooled Datafile Version 6.0."})," Madrid, Spain & Vienna, Austria: JD Systems Institute & WVSA Secretariat. ",(0,i.jsx)(t.a,{href:"https://doi.org/10.14281/18241.24",children:"doi:10.14281/18241.24"})]})]})}function u(e={}){const{wrapper:t}={...(0,s.R)(),...e.components};return t?(0,i.jsx)(t,{...e,children:(0,i.jsx)(h,{...e})}):h(e)}},23809:(e,t,a)=>{function n(e){return[...e].sort((e,t)=>
1e.date<t.date?1:e.date>t.date?-1:0)}function i(e){const[t,a,n]=e.split("-");if(!a)return t;const i=["January","February","March","April","May","June","July","August","September","October","November","December"][Number(a)-1];return i?n?`${i} ${Number(n)}, ${t}`:`${i} ${t}`:t}function s(e){return`/research/${e.pdf}`}function o(e){return e.image?`/research/icons/${e.image}`:null}a.d(t,{G3:()=>s,Gf:()=>d,Yq:()=>i,b8:()=>o,dc:()=>n});const r=["jan","feb","mar","apr","may","jun","jul","aug","sep","oct","nov","dec"];function l(e){return e.replace(/[\\&%$#_{}~^]/g,e=>{switch(e){case"\\":return"\\textbackslash{}";case"~":return"\\textasciitilde{}";case"^":return"\\textasciicircum{}";default:return`\\${e}`}})}function d(e,t){if(e.bibtex)return e.bibtex;const[a,n]=e.date.split("-"),i=`${t.replace(/\/$/,"")}/research/${e.slug}`,s=[["author",e.authors.map(l).join(" and ")],["title",l(e.title)],["year",a]],o=n?r[Number(n)-1]:void 0;o&&s.push(["month",o]),s.push(["howpublished",`\\url{${i}}`]),s.push(["note","Transformer Lab"]);const d=s.map(([e,t])=>` ${e.padEnd(12)} = {${t}}`).join(",\n");return`@misc{${function(e){const[t]=e.date.split("-"),a=e.authors[0];if(!a)return e.slug;const n=(a.trim().split(/\s+/).pop()??"").toLowerCase().replace(/[^a-z0-9]/g,"");return n?`${n}${t}`:e.slug}(e)},\n${d}\n}`}},28453:(e,t,a)=>{a.d(t,{R:()=>o,x:()=>r});var n=a(96540);const i={},s=n.createContext(i);function o(e){const t=n.useContext(s);return n.useMemo(function(){return"function"==typeof e?e(t):{...t,...e}},[t,e])}function r(e){let t;return t=e.disableParentContext?"function"==typeof e.components?e.components(i):e.components||i:o(e.components),n.createElement(s.Provider,{value:t},e.children)}},33201:e=>{e.exports=JSON.parse('[{"slug":"hand-checkable-ramsey-certificates","title":"Short Hand-Checkable Certificates for Bounds on the Ramsey Number R(5,5)","postTitle":"How far can you get on R(5,5) with proofs you can check by hand?","authors":["Asaria","Primus"],"date":"2026-09-16","tags":["MATH","COMBINATORICS"],"abstract":"We prove bounds on the diagonal Ramsey number R(5,5) by certificates short enough to check by hand, using a symmetry reduction for Paley graphs and classical counting arguments. Specific results that we prove include: the clique number of the Paley graph of order 37, from one displayed subgraph, giving R(5,5) \u2265 38, and of order 17, giving R(4,4) \u2265 18; a self-contained proof of the Shearer\u2013Mathon doubling bound, which yields R(5,5) \u2265 37 independently; R(5,5) \u2264 50 from the classical recursion with the computer-verified value R(4,5) = 25, and R(5,5) \u2264 62 with no computational input; and a proof that a subgraph-counting identity of McKay and Radziszowski cannot yield R(5,5) \u2264 49 from the computed extremal edge counts alone.","pdf":"hand-checkable-ramsey-certificates.pdf","image":"hand-checkable-ramsey-certificates.svg"},{"slug":"ppt-discrimination-twisted-bell-states","title":"PPT Discrimination of Twisted Bell States with a Partially Entangled Resource","postTitle":"How much entanglement does it take to tell four twisted Bell states apart?","authors":["Asaria","Primus"],"date":"2026-09-11","tags":["PHYSICS","QUANTUM"],"abstract":"We determine the optimal probability of correctly discriminating the four orthogonal two-qubit states \u03b1|00\u27e9 + \u03b2|11\u27e9, \u03b2|00\u27e9 \u2212 \u03b1|11\u27e9, \u03b1|01\u27e9 + \u03b2|10\u27e9 and \u03b2|01\u27e9 \u2212 \u03b1|10\u27e9, given with uniform probabilities together with an ancillary pair of qubits in the state \u221a((1 + \u03b5)/2)|00\u27e9 + \u221a((1 \u2212 \u03b5)/2)|11\u27e9, by measurements whose elements have positive partial transpose, using a matching pair of primal and dual solutions of the associated semidefinite program. Specific results that we prove include: an exact formula, \xbd(1 + \u221a(1 \u2212 v\xb2)) with v = max{0, \u03b1\u03b2(1 + \u03b5) \u2212 (1 \u2212 \u03b5)/2}, for this optimal probability; an explicit PPT measurement and an explicit dual witness in closed form whose values coincide; a characterization of the pairs (\u03b1, \u03b5) for which the four states are perfectly discriminated, namely 2\u03b1\u03b2(1 + \u03b5) \u2264 1 \u2212 \u03b5; and the observation that the teleportation protocol that is optimal for Bell states is strictly suboptimal whenever 0 < \u03b2 < \u03b1 and 0 < \u03b5 < 1.","pdf":"ppt-discrimination-twisted-bell-states.pdf","
1image":"ppt-discrimination-twisted-bell-states.svg"},{"slug":"detectors-dont-share-one-axis","title":"AI-Text Detection Is at Least Two-Dimensional: A Pooling Artifact, a Factor Test, and Why the Field\'s Disputes Are Conditional","postTitle":"Why AI-Text Detectors Disagree About Who Cheated","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-08-08","tags":["EVALUATION","LLM"],"abstract":"A recurring claim in machine-text detection is that detectors, however different their construction, all read a single underlying axis: post-training style, alignment, or statistical typicality. If true, detectors are interchangeable and their disagreements are noise. We show first that the standard way this claim would be measured is confounded. On a pooled human-and-AI corpus every competent detector correlates through the class label, forcing the first principal component of the detector-score matrix to 83% of variance regardless of what the detectors measure; conditioning on class is necessary and drops it to 58--64%. A participation ratio above one does not by itself refute one axis (a single factor plus per-detector noise also produces it), so we apply Horn\'s parallel analysis: on a heterogeneous suite of nine detectors spanning six model backbones, including a trained detector, two components exceed the independence null in nine of ten class-by-seed cells, and the effective rank is 3.0 (95% CI [2.95, 3.06]). Detection is therefore at least two-dimensional, not one axis. We then show that two live disputes are comparisons made under different unstated conditions rather than disagreements about the phenomenon. Whether detectability tracks alignment or typicality resolves, across six seeds, toward typicality. Whether detectors protect or pe
1nalize non-native English writers is conditional on text length: a native-calibrated threshold flags near zero long non-native essays but up to 9.4% (95% CI [8.0, 11.2]) of short ones for DeBERTa, while a deployed commercial detector flags 51.6% of short non-native exam essays. Underlying the study is a methodological contribution: four controls for detector-axis experiments, each demonstrated against a specific false conclusion it prevents.","pdf":"detectors-dont-share-one-axis.pdf","image":"detectors-dont-share-one-axis.svg"},{"slug":"custom-kernels-on-trainium","title":"Autonomous Custom-Kernel Development on AWS Trainium: Fused RMSNorm, an Unsupported State-Space Model, and Where Hand-Written Kernels Beat the Compiler","postTitle":"We brought Mamba to Trainium \u2014 and mapped where custom kernels actually help","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-08-06","tags":["SYSTEMS","KERNELS"],"abstract":"An autonomously-conducted study of where custom Neuron-Kernel-Interface (NKI) kernels pay off on AWS Trainium 1. At the operation level, a fused NKI RMSNorm kernel is 1.57x faster than a naive un-fused baseline (112 vs 176 \xb5s at Llama-3.1-8B width), bit-exact, and roofline-consistent. End-to-end, the same kernel yields no measurable gain in a real Llama-3.1-8B block (0.999x): RMSNorm is ~1% of block compute and the Neuron compiler already fuses it. We then bring up Mamba, a non-transformer state-space model with no prior Trainium support: it runs on the chip, and we contribute the fused NKI selective-scan kernel that gives Trainium the primitive it was missing, validated bit-exact to the model\'s true 1536-wide state and running correctly inside the model on-device (rel 6.75e-08). Its current sequential form is throughput-bound (0.51x vs the compiler\'s unrolled scan); a parallel/chunked scan is the identified next step to convert enablement into a speedup. The conclusion: hand-written kernels beat naive baselines and are necessary to enable unsupported operations, but end-to-end speedups over the mature compiler require sophisticated parallel algorithms. The entire study cost a few dollars of accelerator time. Code and reproduction package available on request.","pdf":"custom-kernels-on-trainium.pdf","image":"custom-kernels-on-trainium.svg"},{"slug":"answering-rate-gates-verifier-free-rl","title":"When Does Verifier-Free Reinforcement Learning Improve Math Reasoning? Answering Rate Gates It, and Base Ability Appears to Decide the Rest","postTitle":"When can an AI improve its own math by grading itself?","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-08-06","tags":["RL","LLM"],"abstract":"Reinforcement learning without a ground-truth verifier, using self-consensus signals such as majority vote, improves some base language models on math reasoning and does nothing for others. The common explanation attributes the difference to base math ability. We show instead that the outcome is governed by two separable properties of the base model: its parseable-answering rate (how often it emits a template-conformant final answer) and its underlying solution ability. Across three base models on GSM8K, answering rate orders base accuracy, and across these three bases the payoff of reinforcement learning is associated with base ability rather than answering rate. We then isolate the mechanism with two controlled experiments. First, on a base that rarely emits a parseable answer, no reward moves it: a majority-vote reward, a ground-truth verifier, and a purely random reward all leave greedy accuracy unchanged, so the answering-rate gate sits upstream of the reward signal itself. Second, a format prime that raises answering rate lets us intervene on the gate; opening it is not sufficient: reinforcement learning after priming yields a small, single-seed improvement on a capable base, and on a low-ability base it drives answering rate toward one without improving accuracy, reinforcing agreement on answers that are usually wrong. Verifier-free RL therefore appears to require both an open answering-rate gate and sufficient base ability; the gate is necessary but not sufficient, and self-consensus reward optimizes answering, not correctness. All results are from a single seed at 200 evaluation problems; we rest the conclusions on the size of the qualitative effects rather than on replication.","pdf":"answering-rate-gates-verifier-free-rl.pdf","image":"answering-rate-gates-verifier-free-rl.svg"},{"slug":"sparsity-free-on-a-wafer","title":"Is Unstructured Sparsity Free on a Wafer? A Cycle-Accurate Study of Sparse Decode GEMV on the Cerebras Wafer-Scale Engine","postTitle":"Is skipping zeros free on a wafer-scale chip?","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-08-05","tags":["SYSTEMS"],"abstract":"Wafer-scale engines keep model weights resident in on-chip SRAM, which suggests that the memory-bound decode matrix-vector product (GEMV) at the heart of autoregressive inference might turn compute-bound, and that unstructured weight sparsity might translate into proportional cycle savings without the accuracy tax of structured pruning. We test three such ideas with cycle-accurate measurements on the Cerebras SDK fabric simulator, using small processing-element (PE) rectangles and a single format-consistent observable: cycles per kernel. First, unstructured sparsity is nearly free, but only with the right kernel: a compressed kernel that iterates over non-zeros makes cycles fall almost linearly with sparsity (a per-non-zero cost within 2.5% of the ideal, about 8x fewer cycles at 90% sparsity relative to the same kernel at full density), clearly beating the analytic GPU curve (about 3.6x), whereas a naive per-element zero-skip loop does not linearize because it still pays loop and branch overhead on every position. Second, at the single-PE cycle level, structured 2:4 sparsity buys nothing: at matched density the per-non-zero cost is identical across patterns (to within 0.02%), so the accuracy cost of forcing an N:M pattern is unjustified there. Third, the distributed mesh version does not scale as hoped: at fixed problem size, adding processing elements reduces cycles only sub-linearly (a strong-scaling exponent near 0.25 against an ideal of 1.0), consistent with a collective-reduction cost that grows with the mesh, so decode GEMV on the mesh appears communication-bound. A per-PE SRAM ceiling, which we observe directly when a large tile fails to compile, is what forces large problems onto more PEs in the first place. The sparse-linearity and single-PE pattern-independence results hold cleanly; the wafer\'s roofline advantage is argued architecturally rather than measured, and near-ideal distributed scaling does not hold.","pdf":"sparsity-free-on-a-wafer.pdf","image":"sparsity-free-on-a-wafer.svg"},{"slug":"words-actions-neurons","title":"Words, Actions, and Neurons: When a Language Model Agent\'s Self-Report Lags its Behavior, and When it Lies","postTitle":"We made LLMs play co-op games to show how to detect lying by comparing words, actions, and neurons","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-27","tags":["INTERPRETABILITY","AGENTS"],"abstract":"If we want to trust what a language-model agent says about its own strategy, we need to know when the report is faithful. We instrument an agent with three synchronized read-outs of its strategy: its behavior (the action it takes), its narration (what it states its strategy is), and a linear probe on its residual stream (what its activations encode). Across two settings we find two distinct ways these channels desynchronize. In a Stag Hunt coordination game where a Qwen2.5-3B agent honestly learns to cooperate under reinfor
1cement learning, the narration lags the behavior: the agent switches to the cooperative action first, and its stated strategy catches up only after the behavior has consolidated. The narration-versus-behavior gap peaks near 0.47 mid-transition and closes to zero at convergence (peak-to-converged ratio 3.9 plus or minus 2.8 over three seeds), and a per-checkpoint probe that does not drift (AUROC drop approximately 0) rules out the lag being a probe artifact. This lag is a property of learning a strategy: a Llama-3.1-8B agent that already cooperates and already narrates cooperation shows no transition and no lag. In a second setting, agents playing the social-deduction game Among Us on a published sandbox, the desync is not a lag but a lie. A naive impostor-versus-crewmate probe reaches AUROC 1.000 by reading the role label off the prompt, not by detecting deception; controlling for role by testing within impostor decisions, a probe still separates deceptive from honest statements at AUROC 0.865 plus or minus 0.203. Together these give two signatures of when a self-report and the underlying strategy come apart: a transient lag while an agent honestly learns, and a persistent, probe-detectable signal when it deceives.","pdf":"words-actions-neurons.pdf","image":"words-actions-neurons.svg"},{"slug":"thinking-budget-probes","title":"Should You Pay for More Thinking? Low-Budget Probes Are Reliable Purchasing Signals but Poor Forecasts","postTitle":"When Will More Thinking Help a Reasoning Model? Ask the Cheap Eval First","authors":["Salomone","Gandhi","Asaria","Primus"],"date":"2026-07-24","tags":["LLM","EVALUATION"],"abstract":"Evaluating reasoning models at large per-response thinking budgets is expensive, so practitioners would like cheap low-budget evaluations to stand in for them. We show that a probe restricted to at most 4k thinking tokens per response can reliably answer the practitioner\'s operative question, whether paying for a higher budget will improve accuracy, even though it cannot predict the resulting score. The probe itself reveals which regime a model and benchmark occupy: when the model does not consume the probe\'s thinking allowance, higher budgets delivered at most small gains (the largest was 0.051 in the preregistered suite, 0.052 across the completed campaign), and when it saturates the allowance, measured high-budget accuracy met or exceeded the frozen curve forecast in all 14 such cells of the preregistered suite. The resulting decision rule (buy more budget only when the probe saturates its cap and the frozen curve forecasts a gain of at least delta) makes zero over-promise errors across its 14 buy decisions at delta = 0.02 and its 12 at delta = 0.05 (rule-of-three 95% upper bounds 21% and 25%). The rule was constructed on the 24 Qwen3 cells; the held-out R1-Distill lineage has since completed, delivering six holdout buy decisions, all correct: one measurement landed within 0.0005 of its frozen forecast, and on one buy cell the measured score fell 0.060 below its frozen forecast while the buy remained correct by a wide margin, the only substantive forecast shortfall on a buy cell in the campaign. As a numeric forecaster, however, the probe fails its preregistered bars: the best frozen predictor lands within twice seed noise on only 35.7% of budget-hungry cells, frozen 90% intervals cover 53.6% of 28 outcomes against an 80-97% target, the misses are systematically one-sided on budget-hungry cells (cluster-level sign test p = 0.031), and probe cost misses its 20% token bar (35-39% of the high-budget suite; about 21% in GPU-hours). All predictions were fit on the realized thinking tokens of low-budget cells and frozen and hashed before any high-budget cell for that model lineage ran, across three open models and four reasoning benchmarks under a deployable hard thinking-budget cap. In our suite, low-budget probes are conservative purchasing signals rather than forecasters.","pdf":"thinking-budget-probes.pdf","image":"thinking-budget-probes.svg"},{"slug":"obfuscated-code-performance","title":"Two Kinds of Unreadable: What Obfuscating Code Actually Costs Language Models","postTitle":"Two Kinds of Unreadable: what obfuscating code actually costs a language model","authors":["Salomone","Gandhi","Asaria","Primus"],"date":"2026-07-23","tags":["LLM","EVALUATION"],"abstract":"A recent report claimed that making code unreadable, by replacing familiar syntax with alien glyphs, improves a frontier model\'s accuracy on hard programming problems by tens of points. We do not reproduce any such benefit on five models spanning 7B to frontier scale. Holding each problem fixed and varying only its surface, we show that code opacity is two separable things. Obfuscating the syntax (glyph keywords and operators, readable identifiers) produces no statistically resolved drop in a model\'s ability to read the code (paired-bootstrap 95% intervals on the read cost include zero for all three models measured); the large apparent penalty is an output-channel effect, 80 to 95% of it attributable to being made to write the answer back in the opaque form (write-cost intervals exclude zero). Obfuscating the identifiers instead (glyph or misleading names) imposes a genuine comprehension cost of roughly 9 to 16 points that survives on plain output (intervals exclude zero). The study\'s own pre-registered shortcut-suppression interaction passes its gate computed the naive way but dissolves once the output-channel artifact is removed, and the current version of the model family behind the original report refuses the opaque input outright. The practical lesson is a measurement one: score comprehension separately from generation, or an output-format effect will masquerade as a change in reasoning.","pdf":"obfuscated-code-performance.pdf","image":"obfuscated-code-performance.svg"},{"slug":"structure-not-leakage","title":"It\'s Not Leakage: The Crystal-Structure Advantage in Materials Property Prediction is Real and Property-Specific","postTitle":"The crystal-structure advantage in materials AI is real, not leakage","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-22","tags":["MATERIALS","EVALUATION"],"abstract":"Structure-aware graph neural networks (GNNs) beat composition-only models on the Matbench materials-property benchmark, often by large margins. A common suspicion is that this advantage is inflated by benchmark leakage, near-identical crystals shared between train and test, so that the GNN is partly memorizing rather than generalizing. We test this hypothesis directly and find it false. Across the Matbench regression tasks we study, structural near-duplicate rates are negligible (0% on perovskites, at most 2.8% on elastic moduli even under permissive matching), and the large same-composition collision rates (up to 98%) reflect genuine polymorphism, not near-identical structure. Instead we identify a different, real benchmark artifact: because many crystals share a formula but differ in target, composition-only models face an irreducible error floor, 0.379 eV/atom on perovskites, 67% of the target mean absolute deviation, that no per-formula composition model can beat. Decomposing the composition-vs-structure gap with a compact from-scratch CGCNN, we find the structure advantage genuine on all five tasks but strongly property-specific (spanning about 40x): enormous for perovskite formation energy, modest for bulk formation energy and elastic moduli, and small for band gap, where appended structural descriptors actually hurt and only end-to-e
1nd learning recovers a small gain. A distance-resolved diagnostic corroborates genuine generalization: for bulk modulus the advantage grows with distance from the training set, and structural tasks retain a positive gap under an out-of-distribution cluster-split, the opposite of a memorization signature. The structure-GNN advantage is not a leakage artifact; where composition looks weakest, a composition-degeneracy floor is largely responsible.","pdf":"structure-not-leakage.pdf","image":"structure-not-leakage.svg"},{"slug":"knowing-what-it-cannot-know","title":"Knowing What It Cannot Know: Calibrated Non-Fabrication of Unknowable Terminal State in a Language World Model","postTitle":"A chatbot hallucinates secrets it can\'t know. A world model built from the same base won\'t.","authors":["Salomone","Gandhi","Asaria","Primus"],"date":"2026-07-21","tags":["LLM","EVALUATION"],"abstract":"A language world model simulates an environment by predicting, in text, the observation that follows an action. To be trustworthy as a substrate for planning, such a model should signal when it cannot know what an environment would return, rather than invent it. We test this on Qwen-AgentWorld-35B-A3B, an open language world model that simulates a Linux terminal, against the chat-tuned and base models built from the same base checkpoint. Using a ``derivability ladder\'\' of 92 terminal probes that hold the command format fixed while varying whether the answer is knowable from context, we measure how often each model fabricates content it has no basis to know. On our primary contrast, eight probes where a file is established to exist but its contents were never shown, the chat model fabricates plausible contents (including realistic-format secret API keys and password-hash dumps; see Appendix ) on 94% of samples (120/128; 95% bootstrap CI over probes [0.88, 0.98]), while the world model fabricates on 2% (3/128; [0.00, 0.05]); the gap is stable across re-runs (world 0-2%, chat 94-99%). Two within-model controls show the caution tracks knowability rather than the prompt format or content-presence: with the same primed-file format but contents established in history the world model commits correctly, and with contents derivable from earlier values but never shown verbatim it commits and is correct, exactly where a copy-only rule would stall. A confounded base-model anchor fabricates on 51%, and the effect is robust to suppressing chain-of-thought. We are precise about the boundary of the result: the world model\'s non-fabrication is expressed as deflection and non-commitment, not a clean in-character error, and the comparison is observational, so we report an association between world-model post-training and calibrated non-fabrication rather than a causal effect. We make the ladder, harness, and judge available on request.","pdf":"knowing-what-it-cannot-know.pdf","image":"knowing-what-it-cannot-know.svg"},{"slug":"reusing-a-training-difficulty-score","title":"When Does a Per-Example Difficulty Score Go Stale? Loss-Based Scorers Replicate Across Training Runs Far Better Than Forgetting-Based Ones","postTitle":"Reusing a training-difficulty score across runs: when it holds up and when it doesn\'t","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-21","tags":["VISION"],"abstract":"Compute-allocation schemes that steer extra training budget toward high-value examples need a per-example value or difficulty score, and a practical design choice is whether that score must be recomputed online, in-loop, during the run it steers, or can instead be pre-scored once and reused. Prior work on a forgetting-decay score reports that the score is essentially uncorrelated across independent training runs (Spearman rho ~= 0.01), implying that reuse is unsound and online recomputation is required. We test this at CIFAR scale with a synthetic value-weighted compute auction on ResNet-18 and DeiT-Small (a compact, from-scratch vision transformer), comparing static (frozen, pre-scored) against online (in-loop, recomputed) auctions for a forgetting-based score on the full matrix, and a loss-based score on a reduced, CIFAR-10-only robustness check; we separately measure cross-run score correlation across training-stage checkpoints on ResNet-18, for both scorer families and both datasets. We did not detect a nontrivial accuracy effect from the auction mechanism for the forgetting scorer: every static, online, and ensembled-static arm lands within 1.4 percentage points of a uniform-compute control, at n=3 seeds per cell. The correlation diagnostic tells a sharper story: across two datasets, three independent seed pairs, and four training-stage checkpoints, the loss scorer\'s mean cross-run correlation (rho in [0.79, 0.92] across datasets, at the final checkpoint, ResNet-18 only) is consistently higher than the forgetting scorer\'s (rho in [0.47, 0.58]), with point estimates that do not overlap in any cell we measured (a descriptive comparison, not a significance test). Both estimates sit far above the near-zero correlation the forgetting-decay literature reports. This is a CIFAR-scale, small-n study with
1no formal significance testing throughout; within that scope, a team choosing between these two scorer families should expect the loss-based one to tolerate a stale, pre-computed score better than the forgetting-based one, and it is not the static/online distinction itself where the practical risk lives.","pdf":"reusing-a-training-difficulty-score.pdf","image":"reusing-a-training-difficulty-score.svg"},{"slug":"uncensoring-is-not-unlearning","title":"Uncensoring Is Not Unlearning: Removing Political Censorship from a Chinese LLM Backfires with the Obvious Tools","postTitle":"Can you remove political censorship in Chinese open weight LLMs?","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-20","tags":["LLM","EVALUATION"],"abstract":"Open-weight LLMs from Chinese labs increasingly rank among the strongest available, but they carry embedded political guardrails: on topics such as Tiananmen, Xinjiang, and Taiwan they deflect, hedge, or restate the official line. The natural tool for removing such learned behavior is machine unlearning, and we are, to our knowledge, the first to test it for political de-censoring. Across four knowledge-erasure objectives (NPO, RMU, GradDiff, ELM), each swept over training length, we find no setting that cleanly de-censors Qwen2.5-7B-Instruct. The failure is systematic but method-dependent: concept-erasure (ELM) and representation-misdirection (RMU) make the model measurably more evasive at every training length, while gradient-based objectives (NPO, GradDiff) dip below baseline only in a narrow low-step window before collapsing into incoherent, repetitive output, so \\"forgetting\\" either sharpens the reflex or destroys the model. A supervised answer-directly control reaches the lowest censorship of any method while staying fully coherent, showing the target behavior is reachable and the unlearning failure is the wrong objective, not an impossible task. By contrast, behavioral/representation edits that target the refusal itself (abliteration and task-vector negation) never backfire; under a strict judge abliteration halves the judged censorship (30.3%->13.3%) with no observed safety or retain-set regression. Reaching these results required two measurement corrections: the standard refusal-rate metric sees only about 4% of the censorship, which is soft (framing and omission, not refusal) and heavier in Chinese than English; and destructive-training gibberish must be separated from deflection. We validate the metric with two independent judges (a strict proprietary one and a reproducible open one) that agree on method ranking (Spearman rho=0.92) while disagreeing on absolute rates. Our contribution is a corrected way of measuring political censorship and a clean, mechanism-bearing negative result: de-censoring is the opposite of unlearning, so it must target the refusal behavior, not the knowledge.","pdf":"uncensoring-is-not-unlearning.pdf","image":"uncensoring-is-not-unlearning.svg"},{"slug":"more-canadian-than-american","title":"More Canadian than American: A Five-Model Audit of US vs. Canadian Value Defaults in Language Models","postTitle":"More Canadian than American: the surprising default values of five leading AI models","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-14","tags":["LLM","EVALUATION"],"abstract":"Language models are widely assumed to default to American values. We test that assumption by comparing five instruction-tuned models\' answers against the survey-measured opinions of real United States and Canadian respondents, and find the opposite: every model we tested aligns more often with Canadian than American public opinion. Each model answered ten World Values Survey items under three framings (no persona, an explicit US persona, an explicit Canada persona). Across 112 multiple-choice comparisons, 83 are statistically significant, and 57 of those (69%) place the model closer to the Canadian distribution than the American one. This lean appears in every model, survives a false-discovery-rate correction, and grows stronger on the most reliable cells. Persona framing moves the scaled-item answers in eight of nine testable cases, confirming it as a real but item-dependent lever. We also document a willingness-to-answer gap that tracks the individual model rather than open-versus-closed access: Claude Opus 4.8 refuses every bare zero-shot opinion we pose, GPT-4o refuses some, and Grok 4.3 answers as readily as the two open-weight models.","pdf":"more-canadian-than-american.pdf","image":"more-canadian-than-american.svg"},{"slug":"learning-or-just-retrieving","title":"Learning, or Just Retrieving? A Copy-Baseline Audit of Zero-Shot Foundation Models in Time-Series and Single-Cell Tasks","postTitle":"The Copy Test: How Much of a Foundation Model\'s Skill Is Already in the Obvious Answer?","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-07","tags":["EVALUATION"],"abstract":"Zero-shot foundation models are increasingly applied far from language, in time-series forecasting and single-cell genomics, on the promise that large-scale pretraining lets them solve new tasks in context. We ask how much of that apparent skill, measured at the score level, a cheap copy already accounts for. We introduce a modality-agnostic audit that places three predictors on one held-out metric: a non-copying floor (climatology mean, or majority class), a retrieve-and-copy baseline (seasonal or last-value copy, or a nearest-neighbor analog for time series; PCA plus nearest-neighbor vote for single cells), and the foundation model itself. The audit reports a single diagnostic, the Retrieval-Explained Fraction (REF), the share of the model\'s above-floor skill that the cheap copy already achieves. We summarize a family of cheap copies by its strongest single member (the strongest simple c
1opy in a small pre-registered family), an upper bound on what one cheap rule achieves rather than an unbiased expected value. Across four established forecasting series, for Chronos-Bolt-base the strongest simple copy matches most of the model\'s above-floor skill on these metrics: the mean strongest-copy REF is about 0.72 (range 0.57 to 0.99). On ETTm1 both foundation-model families are statistically indistinguishable from a last-value copy: the binding-copy REF confidence interval includes 1 for both Chronos-Bolt (0.987, 95% CI [0.906, 1.050]) and TimesFM (0.999, 95% CI [0.952, 1.046]), so near-total parroting is significance-backed for two independent families, while on the other series the REF CI excludes 1 and the model adds skill. A similar pattern appears in a very different modality: on pbmc3k cell-type annotation a plain PCA copy is significantly better than the scGPT foundation model on macro-F1 (0.905 versus 0.813, gap 95% CI excludes 0) and tied on accuracy (McNemar exact p=0.59). Read at the score level, the answer to the memorable question \\"is your foundation model learning, or just retrieving?\\" is, in the settings audited here, that a trivial copy matches most of the model\'s above-floor skill on these metrics. Confidence intervals are now reported (moving-block bootstrap for the autocorrelated forecasting series; cell bootstrap plus exact McNemar for single cells), so the small-margin claims are tested rather than hedged. We release the audit as a reusable harness so that any model and task can be checked the same way.","pdf":"learning-or-just-retrieving.pdf","image":"learning-or-just-retrieving.svg"},{"slug":"thinking-ahead-of-itself","title":"Does the Silent Workspace Anticipate the Thought? Trace-Aligned Jacobian-Lens Analysis of a Reasoning Model","postTitle":"Does a reasoning model have a subconscious? A plain lens says probably not","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-07","tags":["INTERPRETABILITY","LLM"],"abstract":"The Jacobian lens transports a residual-stream vector into the final-layer basis with the average input-output Jacobian, $lens_l(h)=unembed(J_l h)$, reading the tokens a model is silently disposed to say later; the resulting \\"J-space\\" has been proposed as an emergent workspace. Reasoning (\\"thinking\\") models make this claim newly testable: they externalize their reasoning as an explicit <think> trace, which provides a ground-truth future to check the silent readout against. On Qwen3-8B with a pretrained Jacobian lens, evaluated on a held-out, reasoning-typed probe suite, we read the silent workspace at positions inside the model\'s own thinking trace and ask whether it anticipates the tokens the model goes on to write. It does: the silent readout lands in the realized near-future thinking window with recall $0.88$ versus a frequency-controlled floor of $0.15$, and the effect is uniform across four reasoning types (recall, multi-hop, analogy, arithmetic). A horizon sweep with the immediate next token excluded shows the readout genuinely reaches beyond next-token prediction (excluded recall rising from $0.01$ to $0.41$ as the window widens). We are candid about the limits: on this clean excluded-next metric the Jacobian lens holds only a small edge over an ordinary logit-lens readout (peak gap $+0.025$), and two planned axes-a thinking-versus-non-thinking ablation and causal steering-were not run. We also report a methodological finding: on a chat/reasoning model that answers in prose, a first-token answer-fidelity metric does not transfer, which is precisely why reading inside the trace is the right probe.","pdf":"thinking-ahead-of-itself.pdf","image":"thinking-ahead-of-itself.svg"},{"slug":"does-scale-fix-simulators","title":"Does Scale Fix Learned Particle Simulators? A Pushforward Curriculum, Not Data or Capacity, Governs Long-Horizon Stability","postTitle":"You can\'t scale a learned physics simulator out of blowing up","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-06","tags":["PHYSICS"],"abstract":"Graph-network simulators learn to advance particle-based physics one step at a time, but their autoregressive rollouts accumulate error and eventually blow up. We ask a practical question: among the interventions commonly proposed to stabilize such rollouts, which actually help, and does simply scaling the model or the data buy stability? On a self-generated 2D material-point benchmark of water, sand, and goop, we systematically compare three interventions against a graph-network baseline: a momentum-conserving projection layer, a Sobolev (spatial
1-gradient) loss, and a pushforward rollout curriculum that trains the model on its own predicted states. The headline finding is about training, not architecture. Under extended training the one-step baseline overfits and its rollout stability collapses, while the pushforward curriculum holds: at 50,000 steps the pushforward model retains a divergence horizon near 80 steps at a rollout error of 0.023, against 56 steps and 0.070 for the one-step baseline. The conservation projection consistently hurts, and the Sobolev loss is neutral. Critically, the absolute stability horizon plateaus near 80 steps regardless of scale: tripling the training data leaves it unchanged, and increasing model width, depth, or message-passing steps makes it worse, not better. The pushforward curriculum is therefore the intervention that preserves stability and accuracy under extended training, but the ceiling on absolute horizon is set by the architecture and the problem, not by data or capacity. We did not reach the pre-registered ten-fold stability goal; we report the negative result honestly.","pdf":"does-scale-fix-simulators.pdf","image":"does-scale-fix-simulators.svg"},{"slug":"how-much-can-you-cut","title":"How Much Can Be Cut Before a Shape Cannot Regrow? Characterizing 3D Self-Repair in Neural Cellular Automata","postTitle":"How much can you cut off a 3D shape before it can\'t grow back?","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-03","tags":["3D"],"abstract":"Neural Cellular Automata (NCA) learn a single local update rule, shared by every cell, that can grow a target pattern from a seed and repair it after damage. This behavior is well studied in two dimensions but its three-dimensional counterpart, and in particular the robustness of 3D self-repair, is largely uncharacterized. We train small 3D voxel NCA to grow four target shapes from a single seed and to regenerate after amputation, three shapes reliably and a dense fourth only intermittently, and we measure how recovery depen
1ds on how much of the object is removed. Two findings stand out. First, a network trained only to persist a grown shape already regenerates almost perfectly after damage (mean recovery near 1.0), so self-repair can emerge from persistence training alone, without explicit damage supervision. Second, regeneration degrades with damage severity, and the decline is shape-dependent: bulky shapes recover gracefully even when most of the volume is removed, while a thin branching shape recovers well under light damage but collapses once a large fraction is gone. The shape that grows most accurately is the one that repairs worst, so growth fidelity is not the same property as regeneration robustness. Our numbers are from single training runs at 32^3 resolution and are intended as an initial characterization rather than a benchmarked comparison. We summarize 3D self-repair with a recovery-versus-severity curve, a compact \\"phase-diagram\\"-style view.","pdf":"how-much-can-you-cut.pdf","image":"how-much-can-you-cut.svg"},{"slug":"the-language-is-the-lever","title":"The Language Is the Lever: Prompt Language, More Than Training Origin, Shapes How Open LLMs Answer Contested Questions","postTitle":"It\'s the language, not the lab: the prompt language, not the model\'s origin, is what shifts its framing on contested topics.","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-07-02","tags":["LLM","EVALUATION"],"abstract":"As open language models from many countries enter wide use, understanding what shapes the values they express becomes practically important. We study this with a clean design: how much of a model\'s stance on contested topics is set by where it was built, and how much by the language it is prompted in? We introduce a controlled bilingual audit of eight size-matched open instruction-tuned models (four built in China, four in the West), each answering the same 145 probe items in English and Chinese. Because every item is asked in both languages, each model is its own control, which cleanly isolates the effect of prompt language. Our central finding is that prompt language is a strong and consistent lever: switching from English to Chinese shifts a model\'s stance on contested questions by 0.37 points on a five-point scale (95% CI [0.33, 0.42]), a reliably labeled outcome, and tends to push its framing toward pro-China positions (a directional signal on a weaker label; conditional odds ratio 3.10, 95% credible interval [2.35, 4.07]). This is not a property of Chinese-built models: the shift is just as large in Western-built ones, and it concentrates in nationally grounded topics such as history and geopolitics. Our per-model atlas also surfaces a notable single-model case, DeepSeek-7B, which answers fluently in English but collapses into a canned non-answer in Chinese on 90% of prompts, a language-capability gap that the bilingual design cleanly separates from value-laden refusal. Training origin, by itself, does not account for these differences in our sample. The probe set, rubric, and all labels are available from the authors on request.","pdf":"the-language-is-the-lever.pdf","image":"the-language-is-the-lever.svg"},{"slug":"beyond-fad-and-clap","title":"Beyond FAD and CLAP: A Modern Perceptual Re-Ranking and a Controllability Audit of Open-Source Instrumental Music Generators","postTitle":"The best-sounding open-source music generator is also the only one you can steer","authors":["Salomone","Gandhi","Asaria","Primus"],"date":"2026-06-25","tags":["AUDIO","EVALUATION"],"abstract":"Vendor reports for text-to-music models rank systems with Frechet Audio Distance (FAD) and CLAP score computed on a single CLAP encoder, the same family a model may be trained against, raising a circularity concern. We ask two questions about three open-source instrumental generators: Stable Audio 3 Medium (SA3), ACE-Step 1.5, and DiffRhythm 2. (1) Does SA3\'s reported FAD/CLAP win survive a modern, encoder-diverse perceptual stack? Re-ranking the same three models on the Song Describer instrumental subset with Kernel Audio Distance (KAD) on two encoders, FAD-infinity, Audiobox Aesthetics, and MuQ-MuLan, we find that SA3 ranks first on all nine metrics, with non-overlapping bootstrap intervals on every distribution metric. The circularity that motivated the study does not materialize: SA3\'s lead is as large or larger on the metrics that do not use the vendor\'s encoder, so the win is not a CLAP artifact. (2) How well do these models obey explicit tempo and ke
1y constraints? On 240 prompts that name a target tempo and key, SA3 obeys (tempo within 4 percent on 61 percent of clips, exact key on 64 percent), DiffRhythm 2 partially, and ACE-Step near chance. Explicit controllability is the dimension that most separates the models. SA3 thus leads on both perceptual quality and controllability, while the open comparators trade off differently below it. The harness, prompts, per-clip measurements, and reproducible recipe are available from the authors on request.","pdf":"beyond-fad-and-clap.pdf","image":"beyond-fad-and-clap.svg"},{"slug":"tiny-seismic-phase-picker","title":"How small can a seismic phase picker be? A 34,000-parameter model matches a pretrained deep baseline under leakage-controlled evaluation","postTitle":"Everyone assumes bigger AI models win. We built a 34,000-parameter earthquake detector that beats a model 8\xd7 its size.","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-25","tags":["Efficiency","Seismology"],"abstract":"Deep neural pickers such as PhaseNet and EQTransformer are the default tools for detecting P- and S-wave arrivals, but they carry hundreds of thousands of parameters and are usually benchmarked on randomly split data, where near-duplicate windows of the same event leak between training and test. On STEAD, split so that no earthquake straddles the partitions, a 33,610-parameter 1-D U-Net reaches a mean P/S pick-F1 of 0.76 on held-out data at strict tolerance; a pretrained PhaseNet with about eight times as many parameters, scored through the identical pipeline, reaches 0.64. The decisive ingredient is a foreground-weighted loss: without it the model never learns the P onset (P-F1 0.00) and mean F1 falls to 0.40. As an ablation, masked-waveform self-supervised pretraining gives no detectable low-label gain across three seeds and two fine-tuning schedules, and a single-seed run would have reported a spurious +0.15 improvement. Careful supervision, not scale or pretraining, is what a small seismic picker needs.","pdf":"tiny-seismic-phase-picker.pdf","image":"tiny-seismic-phase-picker.svg"},{"slug":"given-vs-generated-cot-faithfulness","title":"Whose Reasoning Is It? Distilled Reasoning Models Are Faithful to Provided Chains but Only Sparsely to Their Own","postTitle":"Reasoning models don\'t have much trust in their own thoughts","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-25","tags":["LLM","INTERPRETABILITY"],"abstract":"Reasoning models write a chain-of-thought (CoT) before answering, and that trace is increasingly read as an explanation. We ask whether the answer\'s probability actually depends, under intervention, on the reasoning steps, and whether this differs when the reasoning is provided to the model versus generated by it. Using two distilled reasoners (DeepSeek-R1-Distill-Qwen 1.5B and 7B), we measure per-step causal CoT use with a control-differenced answer-logprob intervention, on a synthetic arithmetic task with ground-truth load-bearing steps and on the models\' own generated GSM8K reasoning. When a chain is provided, the models track it tightly: corrupting a load-bearing step collapses the correct answer (by 7 to 9 nats) while corrupting an irrelevant step moves it by essentially zero, separating the two on over 99 percent of problems. When the models reason for themselves, the dependence is real but sparse: only about half of generated steps clear a load-bearing threshold, and that fraction rises through the trace, including within individual problems. The size effect is significant on the provided task (paired difference 1.45 nats, 95 percent CI [1.02, 1.88]) and suggestive on generated reasoning. We also find the verdict is method-relative (corruption and ablation disagree), and that clean activation-patching localization is blocked on synthetic arithmetic by a tension between task competence and the restated values that competence relies on.","pdf":"given-vs-generated-cot-faithfulness.pdf","image":"given-vs-generated-cot-faithfulness.svg"},{"slug":"does-the-prior-pay-off","title":"Does the Prior Pay Off? Scaling a Protein Language Model Lets Reinforcement Learning Beat Directed Evolution on GB1","postTitle":"Scaling a protein model is what lets RL beat trial-and-error protein design","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-24","tags":["RL","LLM"],"abstract":"A deliberately fair, matched-query-budget comparison of GRPO reinfor
1cement-learning fine-tuning of a protein language model (ESM-2) against classical directed evolution on the GB1 four-site fitness landscape, using an exact-lookup oracle (nothing to game), a strong simulated-annealing baseline, a novelty floor, and five seeds. The protein-LM prior is a poor fitness predictor (masked-marginal Spearman 0.04 at 35M, 0.15 at 150M; its top-ranked variants have near-zero fitness), so a no-RL masked-marginal proposer is weak. Simulated annealing nearly solves the landscape, and tuned GRPO over a 35M prior only matches it, missing a pre-registered 10% bar. Scaling the prior to 150M changes the outcome: GRPO then beats annealing by about 10% in the sample-limited regime where queries are scarce (significant at a 1k-query budget; the gap narrows to a tie as both methods saturate). The gain is decoupled from the prior\'s fitness-ranking quality, which stays poor, indicating the larger model helps as an initialization and inductive bias rather than as a fitness oracle. GRPO also collapses without discovery-shaped reward, an entropy bonus, and a low learning rate.","pdf":"does-the-prior-pay-off.pdf","image":"does-the-prior-pay-off.svg"},{"slug":"round-to-nearest-hard-to-beat","title":"Round-to-Nearest Is Hard to Beat: A 30B MoE Coding Model in 24 GB on Apple Silicon","postTitle":"What\'s the best way to run a 30B coding model on a 24 GB Mac?","authors":["Salomone","Asaria","Gandhi","Primus"],"date":"2026-06-23","tags":["SYSTEMS","LLM"],"abstract":"Asks whether a 30B-total / 3B-active MoE coding model (North-Mini-Code-1.0, 128 experts, top-8, Apache-2.0) can be quantized to beat its vendor 4-bit while still fitting a hard \u226424 GB Apple Silicon budget \u2014 and finds round-to-nearest (RTN) 4-bit is the ceiling, a careful negative result. Across a
1structured search (uniform bit-width, group size, number format, built-in mixed-bit, a North-aware custom per-expert allocation, and calibrated streaming GPTQ), no in-budget route beats RTN 4-bit on coding eval, including configurations that spend more memory than 4-bit. The two closest tie RTN on full HumanEval-164: the in-budget higher-bit mixed_4_8 (5.2 avg bits, 20.5 GB) scores 0.9024 (148 vs 146 of 164, McNemar exact p=0.79), and calibrated streaming-GPTQ scores 0.8841 (p=1.000); a cheap MoE\u2192dense distillation pilot fails its triage gate (0/32). The 4-bit model fits 20.81 GB at a real 32K-token agentic turn and decodes 40.2 tok/s, so both hard constraints are met while the primary success criterion is not. The portable contribution is a memory-efficient streaming GPTQ that quantizes a 60 GB / 128-expert MoE on a 48 GB Mac at 3.75 GB peak, where stock mlx-lm GPTQ is infeasible. What limits agentic use is generation-length instability (21/30 SWE-Bench Verified instances loop to the token cap), not bit-width or memory.","pdf":"round-to-nearest-hard-to-beat.pdf","image":"round-to-nearest-hard-to-beat.svg"},{"slug":"instruction-tuning-brain-alignment","title":"Is Instruction-Tuning More Brain-Aligned? Mostly a Chat-Template Artifact","postTitle":"Does fine-tuning a chatbot make it more brain-like? We checked","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-23","tags":["LLM","INTERPRETABILITY"],"abstract":"Reports that instruction-tuned models are more \\"brain-aligned\\" than their base versions are mostly a chat-template artifact, not a property of alignment training. Under identical raw text, post-training weight changes leave fMRI encoding alignment essentially unchanged (Qwen base\u2192Instru
1ct p=0.92), while merely applying the chat template significantly raises apparent alignment \u2014 even for the base model.","pdf":"instruction-tuning-brain-alignment.pdf","image":"instruction-tuning-brain-alignment.svg"},{"slug":"parsimony-not-the-clip","title":"Parsimony, Not the Clip: What Controls the Search in Reinforcement-Learning Symbolic Regression","postTitle":"When an AI searches for the equation behind your data, one dial steers it","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-23","tags":["RL","LLM"],"abstract":"A mechanistic study of how an RL objective shapes symbolic-regression search: the parsimony coefficient \u03bb cleanly and monotonically sets the operating point on the accuracy\u2013parsimony frontier, while the DAPO clip-higher asymmetry does not \u2014 it is the entropy regularizer, not the clip, that sustains exploration. On held-out Feynman the final model recovers a modest fraction (symbolic recovery 0.205), in the range of a deep-SR baseline and below the GP incumbent PySR.","pdf":"parsimony-not-the-clip.pdf","image":"parsimony-not-the-clip.svg"},{"slug":"reward-maximization-collapses-diversity","title":"Reward Maximization Collapses Generative Diversity: Characterizing and Controlling the Trade-off in Verifiable Procedural Generation","postTitle":"Training an AI to be correct collapses its variety","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-22","tags":["LLM","RL"],"abstract":"Verifiable-reward RL (RLVR) for procedural Sokoban-level generation triggers a sharp reward\u2194diversity phase transition: the model mode-collapses to one or two level templates (distinct-valid fraction \u22481.0 \u2192 <0.05) across three trust-region objectives (PPO, DAPO, DPPO), seeds, and scales. The trade-off is controllable \u2014 passively via early-stopping at the Pareto knee, actively via a novelty-bonus reward \u2014 but which clipping objective is used is not a robust lever for diversity.","pdf":"reward-maximization-collapses-diversity.pdf","image":"reward-maximization-collapses-diversity.svg"},{"slug":"expert-modularity-dissolves","title":"How Modular Is a Frontier Mixture-of-Experts? A Pre-registered Causal Test in Which Apparent Expert Modularity Mostly Dissolves","postTitle":"How modular is a frontier Mixture-of-Experts?","authors":["Salomone","Gandhi","Asaria","Primus"],"date":"2026-06-20","tags":["LLM","INTERPRETABILITY"],"abstract":"A pre-registered causal test of whether the experts in a frontier MoE (Command A+, 218B total / 25B active, 128 experts) form functional modules tied to capabilities or languages. Of six pre-registered expert families ablated at inference time against a size-matched random-expert null, only one \u2014 the Arabic-language family \u2014 is a clean selective module that survives an independent corpus and a conservative statistical bar; every other family has a real causal effect but its apparent modularity flips with the corpus, the metric, or the threshold. A positive control on Qwen3-30B-A3B recovers its known disjoint structure, and the verdict reproduces on the un-quantized BF16 model: robust expert modularity is rare and measurement-dependent.","pdf":"expert-modularity-dissolves.pdf","
1image":"expert-modularity-dissolves.svg"},{"slug":"label-free-doubt-signals","title":"Can a Model Catch Its Own Hallucinations for Free? Label-Free Doubt Signals Hold Their Own Against a Labelled Dataset for Abstention","postTitle":"Teaching a language model to say \\"I\'m not sure\\" using its own doubt","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-19","tags":["LLM"],"abstract":"A model\'s own token-probability \\"doubt\\" signal, used to decide when to abstain, matches label-supervised abstention-tuning without using any correctness labels. Across six open-weights models (1B\u20138B, two families) on short-form QA, the label-free LoRA recipe shows no statistically detectable difference from the labelled one at matched coverage \u2014 the gain is calibration, not memorization, and its one blind spot is confidently-wrong facts.","pdf":"label-free-doubt-signals.pdf","image":"label-free-doubt-signals.svg","arxiv":true},{"slug":"adaptation-not-architecture","title":"It\'s the Adaptation, Not the Architecture: Pretrained Vision Transformers Are Competitive for End-to-End Steering on Small Driving Data","postTitle":"How I built my own Tesla-style self-driving AI","kind":"post","byline":"Ali Asaria","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-18","tags":["VISION"],"abstract":"A DINO-pretrained ViT-S is competitive with a pretrained ResNet-50 at end-to-end steering-angle prediction on a small slice (~5\u201316k frames) of the comma2k19 driving dataset \u2014 turn-slice Pearson 0.964 vs 0.967. Competitiveness is conditional on adaptation (low-LR full fine-tuning, a cost-sensitive loss, dropping flip augmentation); the practitioner defaults of a frozen linear probe and horizontal flips reproduce the usual \\"ViTs don\'t work at small scale\\" conclusion.","pdf":"adaptation-not-architecture.pdf","image":"adaptation-not-architecture.svg"},{"slug":"judging-to-improve","title":"Judging to Improve: A De-biased VLM-as-3D-Judge Protocol for Single-Image 3D Generation","postTitle":"Judging to improve: a 3D judge you can train against, and the limit of cheap 3D specialization","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-18","tags":["3D","VISION","LLM"],"abstract":"A trainable, de-biased VLM-as-judge for single-image 3D generation \u2014 one VLM family labels training pairs, a different family scores, and verdicts only count when they survive an order swap. Used to test cheap label-free adaptation of a strong base: six methods reach only parity (0.50 win-rate), never the 0.65 bar \u2014 the durable artifact is the judge protocol, not a model.","pdf":"judging-to-improve.pdf","image":"judging-to-improve.svg","arxiv":true},{"slug":"train-retrieve-or-both","title":"Train, Retrieve, or Both? A Four-Arm Head-to-Head for Correct Statutory Citation on the Ontario Residential Tenancies Act","postTitle":"Train, retrieve, or both? What it takes to make a language model cite the law correctly","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-18","tags":["LLM","RAG"],"abstract":"A four-arm head-to-head (base, LoRA SFT, RAG, SFT+RAG) for c
1orrect statutory citation on Ontario tenancy law. The base model hallucinates 81% of its citations; retrieval is the decisive lever, driving hallucinations to zero by construction and lifting citation exact-match to 0.44, with the SFT+RAG hybrid best at 0.481.","pdf":"train-retrieve-or-both.pdf","image":"train-retrieve-or-both.svg","arxiv":true},{"slug":"cross-model-vlm-judge","title":"A Cross-Model VLM-Judge Protocol for Single-Image 3D Mesh Quality (and Why Cheap Proxies Fall Short)","postTitle":"Judging single-image 3D generation without humans (and why cheap proxies fall short)","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-16","tags":["3D","VISION"],"abstract":"A standardized evaluation protocol for single-image-to-3D mesh generators, using 24-view rendering and position-bias correction \u2014 and showing that common proxies like CLIP similarity and geometry-validity metrics don\'t substitute for a VLM judge.","pdf":"cross-model-vlm-judge.pdf","image":"cross-model-vlm-judge.svg","arxiv":true},{"slug":"reliable-neural-codec-tts","title":"Reliable Neural-Codec Text-to-Speech by ASR Self-Verification and Distillation: Near-Zero Catastrophic Failures Across Models and Codecs","postTitle":"A Universal Post-Training Improvement for Open Audio Models","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-16","tags":["AUDIO","LLM"],"abstract":"ASR-based self-verification drives catastrophic failures (silence, early termination, repetition) to near zero in autoregressive neural-codec TTS, then distills the behavior for inference-time efficiency \u2014 generalizing across four TTS systems and three codecs.","pdf":"reliable-neural-codec-tts.pdf","image":"reliable-neural-codec-tts.svg","arxiv":true},{"slug":"diffusiongemma-token-commitment","title":"Neither Parallel Nor Sequential: How DiffusionGemma Actually Commits Tokens","postTitle":"Neither parallel nor sequential: how DiffusionGemma actually commits tokens","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-12","tags":["LLM"],"abstract":"A close look at token-commitment patterns in DiffusionGemma 26B. Contrary to parallel-decoding marketing, the behavior is neither parallel nor block-autoregressive \u2014 weak left-to-right bias and substantial within-batch ordering ambiguity.","pdf":"diffusiongemma-token-commitment.pdf","image":"diffusiongemma-token-commitment.svg","arxiv":true},{"slug":"int8-gemm-ideogram","title":"Realizing Native INT8 Compute for Diffusion Transformers on Consumer GPUs: A Fused INT8 GEMM Kernel for Ideogram 4.0","postTitle":"Making INT8 actually fast: a fused kernel for Ideogram 4 on a 3090","authors":["Asaria","Salomone","Gandhi","Primus"],"date":"2026-06-12","tags":["SYSTEMS","VISION"],"abstract":"A fused Triton kernel that properly drives the INT8 tensor cores on consumer Ampere GPUs \u2014 ~1.1\xd7 end-to-end speedup, making 1024px generation feasible on a single RTX 3090.","pdf":"int8-gemm-ideogram.pdf","image":"int8-gemm-ideogram.svg","arxiv":true},{"slug":"fp8-quality-ceiling-ideogram","title":"Holding the FP8 Quality Ceiling at 8-Bit Weights and Activations: INT8 and GGUF Post-Training Quantization of Ideogram 4.0 for Consumer GPUs","postTitle":"Quantizing Ideogram 4.0 onto a 3090: an INT8 build that matches FP8 and a 4-bit GGUF that beats NF4","authors":["Gandhi","Asaria","Salomone","Primus"],"date":"2026-06-10","tags":["VISION","SYSTEMS"],"abstract":"Post-training quantization of Ideogram 4.0 where INT8 W8A8 comes out statistically indistinguishable from FP8 on key quality metrics, with INT8 and GGUF Q4_K both cutting compute for consumer-GPU deployment.","pdf":"fp8-quality-ceiling-ideogram.pdf","image":"fp8-quality-ceiling-ideogram.svg","arxiv":true}]')},35590:(e,t,a)=>{a.d(t,{A:()=>n});const n=a.p+"assets/images/comparison_press-7eb89e92da8d43abda9c0c5072ffd6da.png"},77820:(e,t,a)=>{a.d(t,{A:()=>n});const n=a.p+"assets/images/strip_by_category_press-f4a53906bb3a9cc9eb75773749dbe8d5.png"},82771:(e,t,a)=>{a.d(t,{A:()=>n});const n={page:"page_jkX8",header:"header_XCRK",controls:"controls_quHB",search:"search_xM7E",tagFilters:"tagFilters_VU0i",tagChip:"tagChip_yfkw",tagChipActive:"tagChipActive_aNUR",count:"count_jwtZ",clearFilters:"clearFilters_Yju_",list:"list_PZAN",card:"card_YM58",cardWithImage:"cardWithImage_T8Zy",cardThumb:"cardThumb_Yfkq",cardBody:"cardBody__TVl",postIcon:"postIcon_dNcV",kindBadge:"kindBadge_A88Q",cardPost:"cardPost_aIFh",cardTitle:"cardTitle_oOeP",meta:"meta_R1It",authors:"authors_oECp",dot:"dot_gW_U",tag:"tag_b6nU",abstract:"abstract_Nlx0",empty:"empty_svYA",detail:"detail_b20k",back:"back_Hn2R",detailAbstract:"detailAbstract_OrQc",actions:"actions__u_n",cite:"cite_yJVv",bibtexBlock:"bibtexBlock_rDaj",copyBtn:"copyBtn_NOQz",bibtexPre:"bibtexPre_h17v",download:"download_K_gr",pdfFrame:"pdfFrame_sc8t",pending:"pending_ta4y"}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.