1/* ------------------------------------------------------------------ * 2 * content.js â all authored, bilingual copy for the site. 3 * Dynamic run/leaderboard data lives in data/data.js (generated). 4 * UI strings: window.CONTENT.ui[lang][key] 5 * Rubric: window.CONTENT.criteria[] (each has {pt,en} fields) 6 * ------------------------------------------------------------------ */ 7window.CONTENT = { 8 9 ui: { 10 pt: { 11 "lang.name": "PT", 12 "lang.switch": "English", 13 "theme.toggle": "Alternar tema", 14 15 "nav.overview": "Visão geral", 16 "nav.how": "Como funciona", 17 "nav.task": "A tarefa", 18 "nav.criteria": "Os critérios", 19 "nav.scoring": "Pontuação", 20 "nav.methodology": "Metodologia", 21 "nav.leaderboard": "Leaderboard", 22 "nav.details": "Detalhes", 23 "nav.github": "GitHub", 24 25 "hero.eyebrow": "Benchmark de código de backend feito por IA", 26 "hero.kicker": "Um prompt · todo modelo · uma nota", 27 "hero.byline": "Um benchmark opinativo de André N. Darcie â a stack, a rubrica e os pesos refletem o backend que uso no dia a dia no trabalho.", 28 "hero.title.a": "Um prompt.", 29 "hero.title.b": "Todo modelo.", 30 "hero.title.c": "Uma nota.", 31 "hero.lede": "Cada modelo de IA recebe exatamente o mesmo pedido â construir uma API REST de cartão de crédito em .NET 10, com PostgreSQL e Kafka, tudo em Docker. Um avaliador automático percorre 8 categorias de engenharia â as que o sistema rodando consegue provar â e devolve uma nota ponderada de 0 a 5.", 32 "hero.cta.leaderboard": "Ver o leaderboard", 33 "hero.cta.how": "Como funciona", 34 "hero.details": "Ver os detalhes â", 35 "details.eyebrow": "Por dentro do benchmark", 36 "details.title": "Os detalhes", 37 "details.lede": "A tarefa que os modelos constroem, como uma métrica vira uma nota de 0 a 5, e por que as runs são repetidas.", 38 "details.back": "â Leaderboard",
39 "hero.stat.categories": "categorias avaliadas", 40 "hero.stat.weighted": "nota ponderada", 41 "hero.stat.local": "execução 100% local", 42 "hero.gauge.label": "LÃder atual", 43 "hero.gauge.sub": "mediana das runs deep", 44 "hero.pipeline.prompt": "PROMPT.md", 45 "hero.pipeline.promptSub": "o mesmo para todos", 46 "hero.pipeline.model": "O modelo constrói", 47 "hero.pipeline.modelSub": "2 passagens: build â revisão", 48 "hero.pipeline.eval": "Avaliador", 49 "hero.pipeline.evalSub": "Roslyn AST + oráculo ao vivo", 50 "hero.pipeline.score": "Nota 0â5", 51 "hero.pipeline.scoreSub": "ponderada por 8 pesos", 52 53 "how.eyebrow": "O funil", 54 "how.title": "Do prompt à nota, sem opinião humana no caminho crÃtico", 55 "how.lede": "O mesmo prompt vira um projeto inteiro. O avaliador em .NET 10 lê esse projeto de duas formas e transforma o que encontra em números reproduzÃveis.", 56 "how.static.title": "Modo light (estático)", 57 "how.static.body": "Analisa o código-fonte com a árvore sintática do Roslyn e detecta pacotes/arquivos. Rápido, sem Docker, sem rede. A mesma fonte sempre produz a mesma nota.", 58 "how.deep.title": "Modo deep (dinâmico)",
59 "how.deep.body": "Sobe o projeto de verdade (app + Postgres + Kafka), dirige um oráculo de contrato contra a API viva, roda os testes com cobertura e as ferramentas locais (dotnet format, gitleaks, hadolint, SCA do NuGet) â e só então escreve o relatório. Tudo offline e determinÃstico.", 60 "how.note": "Só as runs deep entram no ranking: elas exercitam o sistema de ponta a ponta. As categorias estáticas são determinÃsticas; as de runtime variam com a máquina, por isso repetimos as runs (veja Metodologia).", 61 "how.passes.title": "Cada modelo constrói em duas passagens", 62 "how.pass1.title": "Passagem 1 â Build", 63 "how.pass1.body": "O modelo recebe o PROMPT.md e constrói o projeto inteiro â API, PostgreSQL, Kafka, testes e Docker.", 64 "how.pass2.title": "Passagem 2 â Revisão", 65 "how.pass2.body": "O mesmo modelo recebe o PROMPT-REVIEW.md: revisa criticamente o próprio trabalho contra o brief e verifica do jeito que julgar melhor â ler, buildar, testar, subir o sistema â e aplica o patch final. Ele decide como se convencer; a segunda chance que todo modelo recebe.", 66 "how.roslyn.title": "Roslyn AST, não regex", 67 "how.roslyn.body": "As checagens estáticas usam o compilador do C# como ferramenta de leitura: direção de dependência entre camadas, catch vazio, interfaces com uma única implementação, god classes. à medição, não busca de texto.", 68 "how.oracle.title": "Oráculo de contrato ao vivo", 69 "how.oracle.body": "No modo deep o avaliador vira um cliente da API: cria cartão e transação, confere 201/Location/id, força os 400 (FK inexistente, amount ⤠0, campo obrigatório vazio) e os 404, e observa o evento real chegando no tópico do Kafka. A superfÃcie é leitura + criação â não há PUT nem DELETE.", 70 71 "task.eyebrow": "O que os modelos constroem", 72 "task.title": "Uma API de cartão de crédito pronta para produção", 73 "task.lede": "O spec funcional é só a linha de base. A régua é: isto deveria parecer um serviço que você colocaria no ar â o como importa tanto quanto o se os endpoints funcionam.", 74 "task.domain.title": "DomÃnio â 1:N", 75 "task.domain.body": "Um CreditCard tem muitas Transaction. Toda transação aponta para um cartão existente por chave estrangeira obrigatória.", 76 "task.event.title": "Evento no Kafka", 77 "task.event.body": "A cada transação criada com sucesso (POST â 201), a transação é publicada no tópico transactions, com a key igual ao id â depois de persistir, nunca antes.", 78 "task.rules.title": "Regras que o oráculo cobra", 79 "task.rule.fk": "creditCardId precisa referenciar um cartão existente, senão 400.", 80 "task.rule.amount": "amount tem de ser > 0, senão 400.", 81 "task.rule.required": "cardholderName, cardNumber e merchant não podem ser vazios.", 82 "task.rule.pan": "O número do cartão (PAN) é sensÃvel: nunca em log, nunca em texto puro; CVV/PIN nunca são armazenados.", 83 "task.stack.title": "Stack fixa", 84 "task.card": "CreditCard", 85 "task.tx": "Transaction", 86 "task.oneToMany": "1 : N", 87 88 "criteria.eyebrow": "A rubrica",
89 "criteria.title": "Os critérios, por peso", 90 "criteria.lede": "Oito categorias pontuam (os pesos somam 100%); três são informativas â medidas e reportadas, mas fora da nota, porque a 1â4% nunca separariam duas submissões e cada uma duplicava um sinal que o run já decide. A cor marca quão direta é a medição â todas as notas saem 100% da máquina. Clique para abrir a explicação e o diagrama.", 91 "criteria.expand": "Abrir explicação", 92 "criteria.collapse": "Fechar", 93 "criteria.weight": "Peso", 94 "criteria.iso": "ISO/IEC 25010", 95 "criteria.look": "O que a gente procura", 96 "criteria.how": "Como é medido", 97 "criteria.liveScore": "Nesta run", 98 "criteria.noScore": "sem run ainda", 99 "criteria.sortWeight": "Por peso", 100 "criteria.sortOrder": "Por número", 101 "criteria.checks": "Como o avaliador checa (técnico)", 102 "criteria.checksNote": "Cada linha é uma métrica real do evaluator-dotnet â o nome no relatório, o mecanismo exato e o peso.", 103 "criteria.informational": "informativo", 104 "criteria.tag.live": "ao vivo", 105 "criteria.tag.deep": "deep", 106 107 "auto.FullAuto": "determinÃstico", 108 "auto.SemiOracle": "oráculo", 109 "auto.ProxyReview": "proxy", 110 "auto.FullAuto.desc": "Pontuado 100% pela máquina a partir de análise estática â o mesmo código-fonte sempre gera a mesma nota.", 111 "auto.SemiOracle.desc": "Pontuado 100% pela máquina a cada run contra um oráculo/limiar definido uma vez (suite de aceitação, status esperados).", 112 "auto.ProxyReview.desc": "Pontuado 100% pela máquina a partir de um proxy objetivo (métricas de acoplamento, contagem de violações de regra, checagens de presença) â sem humano no processo.", 113 "auto.legend": "Como cada categoria é medida (tudo 100% automático)", 114 115 "scoring.eyebrow": "Da métrica à nota", 116 "scoring.title": "Como a pontuação é calculada", 117 "scoring.lede": "Nada de nota chutada. Cada métrica é Pass, Partial ou Fail; o que não deu para medir vira Indeterminate e é excluÃdo (não penaliza).", 118 "scoring.step1.title": "1 · Métrica", 119 "scoring.step1.body": "Pass = 1,0 · Partial = 0,5 · Fail = 0,0. Indeterminate sai da conta.", 120 "scoring.step2.title": "2 · Categoria", 121 "scoring.step2.body": "Média das métricas medidas, ponderada, à 5 â nota de 0 a 5 na categoria.", 122 "scoring.step3.title": "3 · Final", 123 "scoring.step3.body": "Média das categorias ponderada pelos pesos da rubrica, renormalizada sobre o que foi medido (a cobertura aparece no relatório).", 124 "scoring.scale.title": "A escala 0â5", 125 "scoring.scale.0": "Ausente ou não funcional", 126 "scoring.scale.1": "Presente, mas com falhas sérias", 127 "scoring.scale.2": "Funciona no caminho feliz, frágil", 128 "scoring.scale.3": "Adequado, segue o básico esperado", 129 "scoring.scale.4": "Sólido, com boas práticas aplicadas", 130 "scoring.scale.5": "Exemplar, pronto para produção", 131 "scoring.weights.title": "Onde o peso está", 132 "scoring.weights.body": "DomÃnio crÃtico (cartão de crédito): o peso está onde o sistema rodando prova alguma coisa â Correção (20%, o oráculo vivo), REST, Segurança, Persistência e Mensageria. Documentação, Portabilidade e Observabilidade não pontuam: a 1â4% não decidiam nada e duplicavam o portão de executabilidade. Pesos são uma calibração deliberada â não há consenso externo sobre eles.", 133 134 "method.eyebrow": "Por que confiar no número", 135 "method.title": "Metodologia", 136 "method.lede": "Modelos são estocásticos: o mesmo prompt gera um projeto diferente a cada vez. Uma única submissão é uma amostra fraca.", 137 "method.multi.title": "Muitas runs, mediana", 138 "method.multi.body": "O leaderboard agrupa as runs por modelo, ordena pela mediana das runs deep e mostra a dispersão (±Ï, média, faixa, contagem). Modelos com menos de 5 runs ficam marcados como provisórios â trate diferenças dentro da dispersão como empate.", 139 "method.det.title": "DeterminÃstico vs. runtime", 140 "method.det.body": "Categorias estáticas (Estático, Arquitetura, Qualidade, Performance) são determinÃsticas dado o Roslyn. Categorias de runtime (build/boot, funcional, evento Kafk
140a) dependem de Docker e da máquina, então variam entre runs.", 141 "method.patch.title": "Avaliado como entregue", 142 "method.patch.body": "Nenhum humano edita a submissão. Ela é pontuada exatamente como o modelo a gerou â sem patches. Um bloqueio de build/boot não é \"consertado\": é tratado pela trava de executabilidade, que limita a nota de quem não compila (â¤0,5), não entrega um sistema executável (â¤1,0) ou nunca sobe saudável (â¤1,5).", 143 "method.patch.note": "A nota vem 100% da ferramenta evaluator-dotnet â sem LLM, sem humano no caminho crÃtico.", 144 145 "lb.eyebrow": "Sempre evoluindo", 146 "lb.title": "Leaderboard", 147 "lb.lede": "Ordenado pela mediana por modelo da nota ponderada (0â5). Só runs deep contam. Rode docs/generate-data.ps1 depois de avaliar novas runs e esta tabela se atualiza sozinha.", 148 "lb.col.rank": "#", 149 "lb.col.model": "Modelo", 150 "lb.col.runs": "Runs", 151 "lb.col.median": "Mediana /5", 152 "lb.col.effort": "Effort", 153 "lb.col.duration": "Tempo", 154 "lb.col.cost": "Custo", 155 "lb.col.spread": "Dispersão (média ±Ï, faixa)", 156 "lb.col.build": "Build", 157 "lb.col.boot": "Boot", 158 "lb.provisional": "provisório (< 5 runs)", 159 "lb.singleRun": "run única", 160 "lb.generated": "Gerado em", 161 "lb.empty": "Leaderboard zerado â de propósito. Todas as runs anteriores foram avaliadas pela rubrica antiga (13 categorias) e por um conjunto de ferramentas que já não existe: os relatórios publicados citavam métricas que NENHUM avaliador deste repo consegue mais emitir. Um número que o código atual não regenera não é resultado, é alegação. As submissões foram apagadas, e o placar recomeça contra a rubrica descrita aqui.", 162 "lb.detail.title": "Perfil por categoria", 163 "lb.detail.run": "Relatório da run", 164 "lb.detail.patch": "Nota limitada", 165 "lb.detail.builds": "Compila", 166 "lb.detail.boots": "Sobe (/health)", 167 "lb.detail.coverage": "Cobertura da rubrica", 168 "lb.detail.close": "Fechar", 169 "run.meta.title": "Procedência", 170 "run.meta.harness": "Agente / CLI", 171 "run.meta.effort": "Effort", 172 "run.meta.duration": "Duração", 173 "run.meta.passes": "Passagens", 174 "run.meta.passes.hint": "build + revisão", 175 "run.meta.attempts": "Tentativas", 176 "run.meta.tokens": "Tokens (in/out)", 177 "run.meta.cost": "Custo", 178 "run.meta.prompt": "Prompt", 179 "run.meta.produced": "Produzido em", 180 "lb.detail.metrics": "métricas", 181 "lb.detail.measured": "medidas", 182 "lb.detail.indeterminate": "indeterminadas", 183 "lb.status.Pass": "Passou", 184 "lb.status.Partial": "Parcial", 185 "lb.status.Fail": "Falhou", 186 "lb.status.Indeterminate": "Indeterminado", 187 "lb.yes": "sim", 188 "lb.no": "não", 189 "lb.viewProfile": "Ver perfil", 190 191 "footer.tagline": "Um prompt, muitos modelos, uma nota automática.", 192 "footer.author": "Feito por André N. Darcie · benchmark opinativo, baseado na stack que uso no trabalho.", 193 "footer.add.title": "Adicione seu modelo", 194 "footer.add.body": "Rode o model-runner com o nome do modelo: ele faz as duas passagens (build + revisão), grava a run e a procedência. Depois avalie e regenere os dados.", 195 "footer.links": "Documentos", 196 "footer.link.prompt": "PROMPT.md â o prompt exato", 197 "footer.link.criteria": "EVALUATION-CRITERIA.md â a rubrica completa", 198 "footer.link.methodology": "METHODOLOGY.md â como ler o leaderboard", 199 "footer.link.evaluator": "evaluator-dotnet â o avaliador", 200 "footer.regen": "Regenerar os dados do site", 201 "footer.built": "Feito com Roslyn AST, um oráculo de contrato ao vivo e Docker.", 202 203 "misc.weightUnit": "%", 204 "misc.of5": "/5" 205 }, 206 207 en: { 208 "lang.name": "EN", 209 "lang.switch": "Português", 210 "theme.toggle": "Toggle theme", 211 212 "nav.overview": "Overview", 213 "nav.how": "How it works", 214 "nav.task": "The task", 215 "nav.criteria": "The criteria", 216 "nav.scoring": "Scoring", 217 "nav.methodology": "Methodology", 218 "nav.leaderboard": "Leaderboard", 219 "nav.details": "Details", 220 "nav.github": "GitHub", 221 222 "hero.eyebrow": "A benchmark of AI-written backend code", 223 "hero.kicker": "One prompt · every model · one score", 224 "hero.byline": "An opinionated benchmark by André N. Darcie â its stack, rubric and weights reflect the backend I work with day to day.", 225 "hero.title.a": "One prompt.", 226 "hero.title.b": "Every model.", 227 "hero.title.c": "One score.", 228 "hero.lede": "Every AI model gets the exact same brief â build a .NET 10 credit-card REST API with PostgreSQL and Kafka, all in Docker. An automated evaluator walks 8 engineering categories â the ones the running system can actually prove â and returns a weighted 0â5
228score.", 229 "hero.cta.leaderboard": "See the leaderboard", 230 "hero.cta.how": "How it works", 231 "hero.details": "See the details â", 232 "details.eyebrow": "Inside the benchmark", 233 "details.title": "The details", 234 "details.lede": "The task the models build, how a metric becomes a 0â5 score, and why runs are repeated.", 235 "details.back": "â Leaderboard", 236 "hero.stat.categories": "categories scored", 237 "hero.stat.weighted": "weighted score", 238 "hero.stat.local": "runs 100% locally", 239 "hero.gauge.label": "Current leader", 240 "hero.gauge.sub": "median of deep runs", 241 "hero.pipeline.prompt": "PROMPT.md", 242 "hero.pipeline.promptSub": "same for everyone", 243 "hero.pipeline.model": "The model builds", 244 "hero.pipeline.modelSub": "2 passes: build â review", 245 "hero.pipeline.eval": "Evaluator", 246 "hero.pipeline.evalSub": "Roslyn AST + live oracle", 247 "hero.pipeline.score": "Score 0â5", 248 "hero.pipeline.scoreSub": "weighted across 8 categories", 249 250 "how.eyebrow": "The pipeline", 251 "how.title": "From prompt to score, with no human on the critical path", 252 "how.lede": "The same prompt becomes a whole project. The .NET 10 evaluator reads that project two ways and turns what it finds into reproducible numbers.", 253 "how.static.title": "Light mode (static)", 254 "how.static.body": "Analyses the source with Roslyn's syntax tree and detects packages/files. Fast, no Docker, no network. The same source always yields the same score.", 255 "how.deep.title": "Deep mode (dynamic)", 256 "how.deep.body": "Boots the project for real (app + Postgres + Kafka), drives a contract oracle against the live API, runs the tests with coverage and the local tools (dotnet format, gitleaks, hadolint, NuGet SCA) â and only then writes the report. All offline and deterministic.", 257 "how.note": "Only deep runs enter the ranking: they exercise the system end to end. Static categories are deterministic; runtime ones vary with the host, which is why runs are repeated (see Methodology).", 258 "how.passes.title": "Each model builds in two passes", 259 "how.pass1.title": "Pass 1 â Build", 260 "how.pass1.body": "The model receives PROMPT.md and builds the whole project â API, PostgreSQL, Kafka, tests and Docker.", 261 "how.pass2.title": "Pass 2 â Review", 262 "how.pass2.body": "The same model receives PROMPT-REVIEW.md: it critically reviews its own work against the brief and verifies it however it judges best â read, build, test, run the system â then applies the final patch. It decides how to convince itself; the second chance every model gets.", 263 "how.roslyn.title": "Roslyn AST, not regex", 264 "how.roslyn.body": "The static checks use the C# compiler as a reading tool: layer dependency direction, empty catch, single-implementation interfaces, god classes. It's measurement, not text search.", 265 "how.oracle.title": "Live contract oracle", 266 "how.oracle.body": "In deep mode the evaluator becomes an API client: it creates a card and a transaction, checks 201/Location/id, forces the 400s (missing FK, amount ⤠0, empty required field) and the 404s, and watches the real event land on the Kafka topic. The surface is read + create â there is no PUT and no DELETE.", 267 268 "task.eyebrow": "What the models build", 269 "task.title": "A production-grade credit-card API", 270 "task.lede": "The functional spec is only the baseline. The bar: this should read like a service you'd actually ship â how it's built matters as much as whether the endpoints work.", 271 "task.domain.title": "Domain â 1:N", 272 "task.domain.body": "One CreditCard has many Transaction. Every transaction points at an existing card through a required foreign key.", 273 "task.event.title": "Kafka event", 274 "task.event.body": "On every successful transaction (POST â 201), the transaction is published to the transactions topic, keyed by its id â after it persists, never before.", 275 "task.rules.title": "Rules the oracle enforces",
276 "task.rule.fk": "creditCardId must reference an existing card, otherwise 400.", 277 "task.rule.amount": "amount must be > 0, otherwise 400.", 278 "task.rule.required": "cardholderName, cardNumber and merchant can't be empty.", 279 "task.rule.pan": "The card number (PAN) is sensitive: never logged, never in plain text; CVV/PIN are never stored.", 280 "task.stack.title": "Fixed stack", 281 "task.card": "CreditCard", 282 "task.tx": "Transaction", 283 "task.oneToMany": "1 : N", 284 285 "criteria.eyebrow": "The rubric", 286 "criteria.title": "The criteria, by weight", 287 "criteria.lede": "Eight categories are scored (the weights sum to 100%); three are informational â measured and reported, but out of the score, because at 1â4% they could never separate two submissions and each duplicated a signal the run already decides. The colour marks how directly it is measured â every score is produced 100% by the machine. Click to open the explanation and diagram.", 288 "criteria.expand": "Open explanation", 289 "criteria.collapse": "Close", 290 "criteria.weight": "Weight", 291 "criteria.iso": "ISO/IEC 25010", 292 "criteria.look": "What we look for", 293 "criteria.how": "How it's measured", 294 "criteria.liveScore": "This run", 295 "criteria.noScore": "no run yet", 296 "criteria.sortWeight": "By weight", 297 "criteria.sortOrder": "By number", 298 "criteria.checks": "How the evaluator checks it (technical)", 299 "criteria.checksNote": "Each row is a real evaluator-dotnet metric â its name in the report, the exact mechanism and its weight.", 300 "criteria.informational": "informational", 301 "criteria.tag.live": "live", 302 "criteria.tag.deep": "deep", 303 304 "auto.FullAuto": "deterministic", 305 "auto.SemiOracle": "oracle", 306 "auto.ProxyReview": "proxy", 307 "auto.FullAuto.desc": "Scored 100% by the machine from static analysis â the same source always yields the same score.", 308 "auto.SemiOracle.desc": "Scored 100% by the machine every run against a fixed oracle/threshold defined once (acceptance suite, expected status codes).", 309 "auto.ProxyReview.desc": "Scored 100% by the machine from an objective proxy (coupling metrics, rule-violation counts, presence checks) â no human in the loop.", 310 "auto.legend": "How each category is measured (all 100% automated)", 311 312 "scoring.eyebrow": "From metric to score", 313 "scoring.title": "How the score is computed", 314 "scoring.lede": "No guessed scores. Each metric is Pass, Partial or Fail; anything that couldn't be measured becomes Indeterminate and is excluded (it doesn't penalise).", 315 "scoring.step1.title": "1 · Metric", 316 "scoring.step1.body": "Pass = 1.0 · Partial = 0.5 · Fail = 0.0. Indeterminate drops out.", 317 "scoring.step2.title": "2 · Category", 318 "scoring.step2.body": "Weighted mean of the measured metrics à 5 â a 0â5
318score for the category.", 319 "scoring.step3.title": "3 · Final", 320 "scoring.step3.body": "Mean of the categories weighted by the rubric weights, renormalised over what was measured (coverage shows in the report).", 321 "scoring.scale.title": "The 0â5 scale", 322 "scoring.scale.0": "Absent or non-functional", 323 "scoring.scale.1": "Present, but with serious flaws", 324 "scoring.scale.2": "Works on the happy path, fragile", 325 "scoring.scale.3": "Adequate, follows the expected basics", 326 "scoring.scale.4": "Solid, with good practices applied", 327 "scoring.scale.5": "Exemplary, production-ready", 328 "scoring.weights.title": "Where the weight sits", 329 "scoring.weights.body": "Critical domain (a credit card): the weight sits where the running system proves something â Correctness (20%, the live oracle), REST, Security, Persistence and Messaging. Documentation, Portability and Observability carry none: at 1â4% they decided nothing and duplicated the executability gate. Weights are a deliberate calibration â there's no external consensus on them.", 330 331 "method.eyebrow": "Why trust the number", 332 "method.title": "Methodology", 333 "method.lede": "Models are stochastic: the same prompt yields a different project each time. A single submission is a weak sample.", 334 "method.multi.title": "Many runs, median", 335 "method.multi.body": "The leaderboard groups runs per model, ranks by the median of the deep runs and shows the spread (±Ï, mean, range, count). Models with fewer than 5 runs are flagged provisional â treat gaps within the spread as ties.", 336 "method.det.title": "Deterministic vs. runtime", 337 "method.det.body": "Static categories (Static, Architecture, Quality, Performance) are deterministic given Roslyn. Runtime categories (build/boot, functional, the Kafka event) depend on Docker and the host, so they vary run to run.", 338 "method.patch.title": "Graded as submitted", 339 "method.patch.body": "No human ever edits a submission. It is scored exactly as the model produced it â no patches. A build/boot blocker is never \"fixed\": it is handled by the executability gate, which caps the score of anything that doesn't compile (â¤0.5), ships no runnable system (â¤1.0), or never boots healthy (â¤1.5).", 340 "method.patch.note": "The score comes 100% from the evaluator-dotnet tool â no LLM, no human on the critical path.", 341 342 "lb.eyebrow": "Always evolving", 343 "lb.title": "Leaderboard", 344 "lb.lede": "Ranked by per-model median of the weighted score (0â5). Only deep runs count. Run docs/generate-data.ps1 after grading new runs and this table updates itself.", 345 "lb.col.rank": "#", 346 "lb.col.model": "Model", 347 "lb.col.runs": "Runs", 348 "lb.col.median": "Median /5", 349 "lb.col.effort": "Effort", 350 "lb.col.duration": "Time", 351 "lb.col.cost": "Cost", 352 "lb.col.spread": "Spread (mean ±Ï, range)", 353 "lb.col.build": "Build", 354 "lb.col.boot": "Boot", 355 "lb.provisional": "provisional (< 5 runs)", 356 "lb.singleRun": "single run", 357 "lb.generated": "Generated", 358 "lb.empty": "The leaderboard was reset â on purpose. Every earlier run was graded under the old 13-category rubric and a tool set that no longer exists: the published reports cited metrics NO evaluator in this repo can emit any more. A number the current code cannot regenerate is not a result, it is a claim. The submissions were deleted, and the board restarts against the rubric described here.", 359 "lb.detail.title": "Per-category profile", 360 "lb.detail.run": "Run report", 361 "lb.detail.patch": "Score capped", 362 "lb.detail.builds": "Builds", 363 "lb.detail.boots": "Boots (/health)", 364 "lb.detail.coverage": "Rubric coverage", 365 "lb.detail.close": "Close", 366 "run.meta.title": "Provenance", 367 "run.meta.harness": "Agent / CLI", 368 "run.meta.effort": "Effort", 369 "run.meta.duration": "Duration", 370 "run.meta.passes": "Passes", 371 "run.meta.passes.hint": "build + review", 372 "run.meta.attempts": "Attempts", 373 "run.meta.tokens": "Tokens (in/out)", 374 "run.meta.cost": "Cost", 375 "run.meta.prompt": "Prompt", 376 "run.meta.produced": "Produced", 377 "lb.detail.metrics": "metrics", 378 "lb.detail.measured": "measured", 379 "lb.detail.indeterminate": "indeterminate", 380 "lb.status.Pass": "Pass", 381 "lb.status.Partial": "Partial", 382 "lb.status.Fail": "Fail", 383 "lb.status.Indeterminate": "Indeterminate", 384 "lb.yes": "yes", 385 "lb.no": "no", 386 "lb.viewProfile": "View profile", 387 388 "footer.tagline": "One prompt, many models, one automated score.", 389 "footer.author": "Made by André N. Darcie · an opinionated benchmark, based on the stack I use at work.", 390 "footer.add.title": "Add your model", 391 "footer.add.body": "Run model-runner with the model name: it does both passes (build + review), records the run and its provenance. Then grade it and regenerate the data.", 392 "footer.links": "Documents", 393 "footer.link.prompt": "PROMPT.md â the exact prompt", 394 "footer.link.criteria": "EVALUATION-CRITERIA.md â the full rubric", 395 "footer.link.methodology": "METHODOLOGY.md â how to read the leaderboard", 396 "footer.link.evaluator": "evaluator-dotnet â the evaluator", 397 "footer.regen": "Regenerate the site data", 398 "footer.built": "Built with Roslyn AST, a live contract oracle and Docker.", 399 400 "misc.weightUnit": "%", 401 "misc.of5": "/5" 402 } 403 }, 404 405 /* ---- the criteria: 8 scored (weights sum to 100) + 3 informational (weightPct 0) ---- 406 Language-neutral facts + {pt,en} prose. `informational: true` => reported, never ranked. */ 407 criteria: [ 408 { 409 number: 1, key: "functional", weightPct: 20, iso: "Functional suitability", 410 automation: "SemiOracle", diagram: "requestFlow", 411 title: { pt: "Correção funcional e testes", en: "Functional Correctness & Tests" }, 412 tagline: { 413 pt: "O sistema faz o que promete, do começo ao fim?", 414 en: "Does the system do what it promises, end to end?" 415 }, 416 body: { 417 pt: "O critério mais básico e, de longe, o de maior peso. Quem decide é o oráculo de contrato: o avalia
417dor dirige a API viva por HTTP real, contra o Postgres e o Kafka de verdade, e cobra o contrato documentado. à o único sinal que o modelo não consegue escrever a seu favor â por isso a suÃte de testes do próprio projeto entrou aqui ao lado dele, com peso baixo, em vez de ser uma categoria à parte: uma suÃte que o modelo escreve para se auto-avaliar não é um sinal independente.", 418 en: "The most basic criterion and, by far, the heaviest. What decides it is the contract oracle: the evaluator drives the live API over real HTTP, against the real Postgres and Kafka, and holds it to the documented contract. It is the one signal a model cannot write in its own favour â which is why the project's own test suite now sits here beside it, at low weight, instead of standing as its own category: a suite the model writes to grade itself is not an independent signal." 419 }, 420 look: { 421 pt: ["Todos os endpoints do spec implementados e corretos", "Regras de domÃnio aplicadas (FK existe, amount > 0, campos obrigatórios)", "Casos de borda tratados, não só o caminho feliz", "Testes unitários de verdade â e só unitários (Testcontainers reprova)", "Cobertura ⥠60% no que importa (crédito parcial a partir de 35%)"], 422 en: ["Every endpoint in the spec implemented and correct", "Domain rules applied (FK exists, amount > 0, required fields)", "Edge cases handled, not just the happy path", "Real unit tests â and unit only (Testcontainers is a Fail)", "Coverage ⥠60% on the code that matters (half credit from 35%)"] 423 }, 424 how: { 425 pt: "Um oráculo de caixa-preta â o próprio avaliador dirigindo a API viva por HTTP â confere cada status esperado (201, 400, 404) da superfÃcie de leitura + criação e carrega a maior parte do peso. A suÃte do projeto roda uma única vez (dotnet test + Coverlet) e entrega, dessa mesma execução, a taxa de aprovação e a cobertura.", 426 en: "A black-box oracle â the evaluator itself driving the live API over HTTP â checks every expected status (201, 400, 404) of the read + create surface, and carries most of the weight. The project's suite runs exactly once (dotnet test + Coverlet), and that single run yields both its pass rate and its coverage." 427 } 428 }, 429 { 430 number: 2, key: "architecture", weightPct: 12, iso: "Maintainability", 431 automation: "ProxyReview", diagram: "archLayers", 432 title: { pt: "Arquitetura e design", en: "Architecture & Design" }, 433 tagline: { 434 pt: "As dependências apontam para dentro, e a complexidade é proporcional ao problema?", 435 en: "Do dependencies point inward, and is complexity proportional to the problem?" 436 }, 437 body: { 438 pt: "Camadas claras â apresentação, aplicação, domÃnio, infraestrutura â com o domÃnio sem conhecer a infra. Trocar o banco ou o broker não deveria reescrever regra de negócio. E simplicidade conta de verdade: aqui mora a métrica que torna o YAGNI exigÃvel. Entregar o que o brief mandou NÃO construir â um PUT, um outbox, um consumer, OpenTelemetry, versionamento de API â é defeito, não ambição. Não é engenharia; é não ter lido o enunciado.", 439 en: "Clear layers â presentation, application, domain, infrastru
439cture â with the domain unaware of infra. Swapping the database or broker shouldn't rewrite business rules. And simplicity genuinely counts: this is where the metric that makes YAGNI enforceable lives. Shipping what the brief said NOT to build â a PUT, an outbox, a consumer, OpenTelemetry, API versioning â is a defect, not ambition. It isn't engineering; it's not having read the brief." 440 }, 441 look: { 442 pt: ["Separação de camadas com dependências apontando para dentro", "Controllers finos, sem regra de negócio", "Sem god classes", "Zero gold-plating: nada de PUT/DELETE, outbox, consumer, OTel ou versionamento"], 443 en: ["Layer separation with dependencies pointing inward", "Thin controllers, no business logic", "No god classes", "Zero gold-plating: no PUT/DELETE, outbox, consumer, OTel or versioning"] 444 }, 445 how: { 446 pt: "Roslyn lê os usings para checar a direção das dependências, mede o tamanho das classes e detecta a maquinaria que o brief proibiu. O antigo proxy de overengineering (interfaces com uma só implementação) foi aposentado: essas interfaces são justamente o encaixe de inversão de dependência que esta mesma categoria premia â na prática ele nunca reprovava ninguém.", 447 en: "Roslyn reads the usings to check dependency direction, measures class size, and detects the machinery the brief ruled out. The old overengineering proxy (single-implementation interfaces) is retired: those interfaces are precisely the dependency-inversion seam this same category rewards â in practice it could never fail anyone." 448 } 449 }, 450 { 451 number: 3, key: "quality", weightPct: 10, iso: "Maintainability", 452 automation: "FullAuto", diagram: null, 453 title: { pt: "Qualidade de código", en: "Code Quality" }, 454 tagline: { 455 pt: "LegÃvel, idiomático e sem sujeira?", 456 en: "Readable, idiomatic and free of cruft?" 457 }, 458 body: { 459 pt: "O micro-nÃvel: nomes expressivos, métodos curtos, sem catch vazio engolindo exceção, sem código morto nem TODO pendente. Analisadores ligados via .editorconfig e o projeto limpo no dotnet format. O I/O assÃncrono passou a morar aqui: um .Result ou .Wait() no caminho da requisição é bug de starvation do thread-pool no ASP.NET Core â logo, é defeito de código, e reprova.", 460 en: "The micro level: expressive names, short methods, no empty catch swallowing exceptions, no dead code or lingering TODOs. Analyzers on via .editorconfig and the project clean under dotnet format. Async I/O now lives here: a .Result or .Wait() on the request path is thread-pool starvation in ASP.NET Core â so it is a code defect, and it fails." 461 }, 462 look: { 463 pt: ["Sem catch vazio (exceção engolida)", "Sem TODO/FIXME/HACK pendente", "Analisadores/.editorconfig habilitados", "I/O assÃncrono, sem sync-over-async (.Result/.Wait())", "dotnet format limpo, 0 warnings de build"], 464 en: ["No empty catch (swallowed exception)", "No lingering TODO/FIXME/HACK", "Analyzers/.editorconfig enabled", "Async I/O, no sync-over-async (.Result/.Wait())", "dotnet format clean, 0 build warnings"] 465 }, 466 how: { 467 pt: "Totalmente automático: Roslyn conta catches vazios, TODOs e chamadas bloqueantes; o modo deep roda dotnet format e lê os warnings do build de Release (o mesmo build que gateia a executabilidade â não há segundo build).", 468 en: "Full-auto: Roslyn counts empty catches, TODOs and blocking calls; deep mode runs dotnet format and reads the warnings from the Release build (the same build that gates executability â no second build is run)." 469 } 470 }, 471 { 472 number: 4, key: "rest", weightPct: 14, iso: "Compatibility / Interoperability", 473 automation: "SemiOracle", diagram: "restStatus", 474 title: { pt: "Design da API REST", en: "REST API Design" }, 475 tagline: { 476 pt: "O contrato HTTP é previsÃvel â e o que a API responde de verdade bate com o que ela promete?", 477 en: "Is the HTTP contract predictable â and does what the API really answers match what it promises?" 478 }, 479 body: { 480 pt: "Verbos e status corretos (nÃvel 2 de Richardson), erros padronizados em RFC 9457 (application/problem+json), DTOs na entrada e na saÃda, JSON em camelCase,
480coleções paginadas e um OpenAPI que realmente descreve os endpoints. Quase tudo aqui é cobrado no sistema vivo: um spec servido e vazio ('paths': {}) é defeito real, e a detecção por presença deixava passar batido.", 481 en: "Correct verbs and status codes (Richardson level 2), standardised errors in RFC 9457 (application/problem+json), DTOs in and out, camelCase JSON, paginated collections and an OpenAPI that actually describes the endpoints. Almost all of it is asserted on the running system: a served-but-empty spec ('paths': {}) is a real defect, and presence-detection silently passed it." 482 }, 483 look: { 484 pt: ["201 com header Location na criação", "Erros em application/problem+json (RFC 9457)", "JSON camelCase e paginação que respeita o page size pedido", "OpenAPI servido e populado (spec vazio = reprova; spec ausente = reprova)", "DTOs â nunca a entidade do EF exposta"], 485 en: ["201 with a Location header on create", "Errors in application/problem+json (RFC 9457)", "camelCase JSON and pagination that honours the requested page size", "OpenAPI served and populated (empty spec = Fail; no spec = Fail)", "DTOs â never the EF entity exposed"] 486 }, 487 how: { 488 pt: "O oráculo observa a forma real da resposta (Location, media type, camelCase, page size) e o probe busca o documento OpenAPI servido e conta as operações declaradas. Versionamento de API saiu da rubrica: há uma versão de uma API â versioná-la é cerimônia, e agora conta como gold-plating.", 489 en: "The oracle observes the real response shape (Location, media type, camelCase, page size) and the probe fetches the served OpenAPI document and counts the operations it declares. API versioning is out of the rubric: there is one version of one API â versioning it is ceremony, and it now counts as gold-plating." 490 } 491 }, 492 { 493 number: 5, key: "persistence", weightPct: 13, iso: "Reliability / Performance", 494 automation: "ProxyReview", diagram: "domain1n", 495 title: { pt: "Persistência e banco", en: "Persistence & Database" }, 496 tagline: { 497 pt: "O banco garante integridade, e as queries são previsÃveis?", 498 en: "Does the database guarantee integrity, and are the queries predictable?" 499 }, 500 body: { 501 pt: "Integridade referencial por PK/FK no próprio banco, Ãndices nas colunas de FK e de filtro, migrações versionadas (não EnsureCreated) e AsNoTracking nas leituras. Controle de concorrência otimista (rowversion) saiu da rubrica â e do enunciado: a superfÃcie é leitura + criação, não existe UPDATE em lugar nenhum, então um token de concorrência protege contra um conflito de escrita que não pode acontecer. Exigi-lo era a rubrica contrariando o próprio YAGNI.", 502 en: "Referential integrity via PK/FK in the database itself, indexes on FK and filter columns, versioned migrations (not EnsureCreated) and AsNoTracking on reads. Optimistic concurrency (rowversion) is out of the rubric â and out of the brief: the surface is read + create, there is no UPDATE anywhere, so a concurrency token guards against a write conflict that cannot happen. Demanding it was the rubric contradicting its own YAGNI rule." 503 }, 504 look: { 505 pt: ["Migrações versionadas, não EnsureCreated", "FK/relacionamentos e Ãndices definidos", "AsNoTracking nas leituras", "Sem N+1 no caminho quente"], 506 en: ["Versioned migrations, not EnsureCreated", "FK/relationships and indexes defined", "AsNoTracking on reads", "No N+1 on the hot path"] 507 }, 508 how: { 509 pt: "Roslyn detecta migrações, FKs, Ãndices e AsNoTracking;
509 depois o schema é exercitado de verdade â o oráculo cria cartões e transações contra o Postgres que a submissão subiu, então migração que não aplica ou mapeamento quebrado aparece como check de contrato falhando (um 500 no lugar do 201), não como opinião estática.", 510 en: "Roslyn detects migrations, FKs, indexes and AsNoTracking; then the schema is exercised for real â the oracle creates cards and transactions against the Postgres the submission booted, so a migration that doesn't apply or a broken mapping surfaces as a failed contract check (a 500 instead of a 201), not as a static opinion." 511 } 512 }, 513 { 514 number: 6, key: "messaging", weightPct: 13, iso: "Reliability / Compatibility", 515 automation: "FullAuto", diagram: "produce", 516 title: { pt: "Mensageria (Kafka)", en: "Messaging (Kafka)" }, 517 tagline: { 518 pt: "O evento chega mesmo no tópico â e uma queda do broker não derruba a requisição?", 519 en: "Does the event really land on the topic â and does a broker hiccup leave the request alone?" 520 }, 521 body: { 522 pt: "Escopo é só o lado publicador â a essência. Um producer durável (acks=all / idempotência) publica o evento no tópico transactions, chaveado pelo id, depois que a linha foi persistida. E a publicação é desacoplada do sucesso da requisição: broker fora do ar vira catch-and-log, não um 500 depois que o dado já foi salvo. Consumer, outbox e DLQ estão fora de escopo â construÃ-los conta como gold-plating.", 523 en: "Scope is the publishing side only â the essence. A durable producer (acks=all / idempotence) publishes the event to the transactions topic, keyed by id, after the row is persisted. And publishing is decoupled from the request's success: a broker outage is caught-and-logged, not turned into a 500 once the data is saved. Consumer, outbox and DLQ are out of scope â building them counts as gold-plating." 524 }, 525 look: { 526 pt: ["Cliente Kafka presente e chamada de publicação no create bem-sucedido", "Producer durável (Acks.All / EnableIdempotence)", "Evento real observado no tópico, com key == id", "Falha do broker não vira 500"], 527 en: ["A Kafka client present and a publish call on the successful create", "Durable producer (Acks.All / EnableIdempotence)", "A real event observed on the topic, keyed by id", "A broker failure does not become a 500"] 528 }, 529 how: { 530 pt: "Roslyn detecta o cliente, a chamada de publicação e a config de durabilidade. No modo deep, o harness pluga o PRÃPRIO consumidor (kcat) no tópico transactions e confirma que um evento real foi publicado para uma transação recém-criada â essa observação ao vivo é a prova.", 531 en: "Roslyn detects the client, the publish call and the durability config. In deep mode the harness attaches its OWN consumer (kcat) to the transactions topic and confirms a real event was published for a just-created transaction â that live observation is the proof." 532 } 533 }, 534 { 535 number: 7, key: "security", weightPct: 14, iso: "Security", 536 automation: "ProxyReview", diagram: "panMask", 537 title: { pt: "Segurança (PCI)", en: "Security (PCI)" }, 538 tagline: { 539 pt: "O PAN está protegido, e nada sensÃvel vaza para o código, o log ou o repositório?", 540 en: "Is the PAN protected, and does nothing sensitive leak into the code, the logs or the repo?" 541 }, 542 body: { 543 pt: "DomÃnio crÃtico: aqui vale o PCI DSS Requisito 3. O PAN tem de estar protegido (cifrado, tokenizado ou truncado) e nunca logado; dado de autenticação sensÃvel (CVV/CVC, trilha, PIN) nunca pode ser armazenado. Fora isso: nada de segredo hardcoded, validação de toda entrada, rate limiting e dependências sem vulnerabilidade conhecida. Autenticação é opcional e não pontua â não há modelo de usuário no escopo.", 544 en: "A critical domain: PCI DSS Requirement 3 applies. The PAN must be protected (encrypted, tokenised or truncated) and never logged; sensitive authentication data (CVV/CVC, track, PIN) must never be stored. Beyond that: no hardcoded secrets, validation of every input, rate limiting, and dependencies free of known vulnerabilities. Auth is optional and unscored â there is no user model in scope." 545 }, 546 look: { 547 pt: ["Zero PAN válido por Luhn no código/config de produção (fixtures de teste não contam)", "Zero campo de CVV/CVC/trilha/PIN", "Zero segredo no repositório (gitleaks)", "Zero dependência vulnerável High/Critical", "Validação de entrada e rate limiting"], 548 en: ["Zero Luhn-valid PAN in production code/config (test fixtures excluded)", "Zero CVV/CVC/track/PIN field", "Zero secret in the repository (gitleaks)", "Zero High/Critical vulnerable dependency", "Input validation and rate limiting"] 549 }, 550 how: { 551 pt: "As checagens PCI rodam sobre o AST do Roslyn, não sobre regex do texto: todo literal de string do código de produção e todo valor de appsettings passa por Luhn, e os identificadores são varridos atrás de CVV/trilha/PIN. Ao lado disso, gitleaks varre a árvore e dotnet list package --vulnerable lê o grafo NuGet.", 552 en: "The PCI checks run over the Roslyn AST, not regex over text: every production string literal and every appsettings value is Luhn-checked, and identifiers are searched for CVV/track/PIN. Alongside, gitleaks scans the tree and dotnet list package --vulnerable reads the NuGet graph." 553 } 554 }, 555 { 556 number: 8, key: "resilience", weightPct: 4, iso: "Reliability", 557 automation: "FullAuto", diagram: "resilience", 558 title: { pt: "Resiliência e erros", en: "Resilience & Error Handling" }, 559 tagline: { 560 pt: "Falha acontece â o serviço degrada com elegância ou vaza stack trace?", 561 en: "Failure happens â does the service degrade gracefully or leak a stack trace?" 562 }, 563 body: { 564 pt: "Um handler global de exceção (nada de stack trace vazando para o cliente), retries/timeouts/circuit breakers no I/O externo (banco e broker) e shutdown gracioso. O peso caiu para 4% de propósito: o que esta categoria alegava já é provado onde ele é DEMONSTRADO, não declarado â o /health é o portão de executabilidade (quem não sobe é capado em 1.5), e o oráculo já dispara uma requisição malformada e cobra um 4xx limpo. O que sobra aqui é sinal estático, e pesa como tal.", 565 en: "A global exception handler (no stack trace leaking to clients), retries/timeouts/circuit breakers on external I/O (database and broker), and graceful shutdown. The weight dropped to 4% on purpose: what this category used to claim is now proved where it is DEMONSTRATED rather than declared â /health is the executability gate (a service that never comes up is capped at 1.5), and the oracle already fires a malformed request and demands a clean 4xx. What is left here is a static signal, and it is weighted like one." 566 }, 567 look: { 568 pt: ["Handler global único de exceção", "Requisição malformada responde 4xx, sem vazar internals", "PolÃticas de retry/timeout/circuit breaker (Polly) no banco e no broker", "Shutdown gracioso"], 569 en: ["A single global exception handler", "A malformed request answers 4xx without leaking internals", "Retry/timeout/circuit-breaker policies (Polly) on the database and the broker", "Graceful shutdown"] 570 }, 571 how: { 572 pt: "Roslyn detecta as polÃticas, o handler global e o shutdown. Ao vivo, o oráculo manda uma requisição deliberadamente malformada e exige 4xx limpo â sem stack trace, sem internals do Npgsql/EF, sem mar
572cador .cs:linha no corpo.", 573 en: "Roslyn detects the policies, the global handler and the shutdown. Live, the oracle sends a deliberately malformed request and demands a clean 4xx â no stack trace, no Npgsql/EF internals, no .cs:line marker in the body." 574 } 575 }, 576 { 577 number: 9, key: "observability", weightPct: 0, informational: true, iso: "â (enabler)", 578 automation: "FullAuto", diagram: "pillars", 579 title: { pt: "Observabilidade", en: "Observability" }, 580 tagline: { 581 pt: "Dá para diagnosticar um incidente sem adicionar código?", 582 en: "Can you diagnose an incident without adding code?" 583 }, 584 body: { 585 pt: "Medida e reportada, mas SEM peso na nota. O sinal decisivo dela é o /health â e esse é o portão de executabilidade: quem nunca responde 2xx é capado em 1.5/5, aconteça o que acontecer aqui. O que sobra (log JSON estruturado, correlation id) é prática real e vale mostrar, mas não é número para ranquear. OpenTelemetry e /metrics saÃram do enunciado â adicioná-los agora conta como gold-plating.", 586 en: "Measured and reported, but with NO weight in the score. Its decisive signal is /health â and that is the executability gate: a service that never answers 2xx is capped at 1.5/5 no matter what happens here. What remains (structured JSON logs, a correlation id) is real practice worth showing, but not a number to rank on. OpenTelemetry and /metrics are out of the brief â adding them now counts as gold-plating." 587 }, 588 look: { 589 pt: ["Log estruturado em JSON (Serilog / AddJsonConsole)", "Correlation/trace id propagado ponta a ponta", "/health respondendo 2xx no sistema vivo", "PAN jamais logado (isso pontua em Segurança)"], 590 en: ["Structured JSON logging (Serilog / AddJsonConsole)", "A correlation/trace id propagated end to end", "/health answering 2xx on the live system", "The PAN never logged (that one scores under Security)"] 591 }, 592 how: { 593 pt: "Roslyn confirma o log estruturado e o correlation id no código; um probe HTTP confirma o /health no sistema vivo. Tudo isso entra no relatório â e em nenhum momento na nota.", 594 en: "Roslyn confirms the structured logging and the correlation id in the source; an HTTP probe confirms /health on the live system. All of it lands in the report â and none of it in the score." 595 } 596 }, 597 { 598 number: 10, key: "portability", weightPct: 0, informational: true, iso: "Portability", 599 automation: "FullAuto", diagram: null, 600 title: { pt: "Portabilidade e deploy", en: "Portability & Deploy" }, 601 tagline: { 602 pt: "clone â up â funciona, em qualquer máquina?", 603 en: "clone â up â it works, on any machine?" 604 }, 605 body: { 606 pt: "Medida e reportada, mas SEM peso â porque a parte que importa aqui não é um checklist: é o portão de executabilidade. O harness sobe o docker-compose DA PRÃPRIA submissão, e quem não fica saudável é capado em 1.0â1.5/5. Pontuar 'existe um Dockerfile' com 2% em cima de um portão que já rodou a coisa era contar o mesmo fato duas vezes. O workflow de CI saiu da rubrica: nada aqui o executa â a métrica pontuava a existência de um YAML.", 607 en: "Measured and reported, but with NO weight â because the part that matters here isn't a checklist: it's the executability gate. The harness boots the submission's OWN docker-compose, and one that never turns healthy is capped at 1.0â1.5/5. Scoring 'a Dockerfile exists' at 2% on top of a gate that already ran the thing was counting the same fact twice. The CI workflow is out of the rubric: nothing here runs it â the metric scored the existence of a YAML file." 608 }, 609 look: { 610 pt: ["Config só por variável de ambiente (12-Factor III/IV)", "Dependências pinadas (lock file / global.json / CPM)", "Container roda como não-root", "Dockerfile sem violações (hadolint)"], 611 en: ["Config from environment variables only (12-Factor III/IV)", "Pinned dependencies (lock file / global.json / CPM)", "The container runs as non-root", "Dockerfile free of violations (hadolint)"] 612 }, 613 how: { 614 pt: "Checagens de arquivo para Dockerfile, compose, config por env, pinagem e USER não-root; hadolint linta o Dockerfile. Se o projeto sobe de verdade, quem responde é o portão â não esta lista.", 615 en: "File checks for the Dockerfile, the compose, env-based config, pinning and a non-root USER; hadolint lints the Dockerfile. Whether the project actually comes up is answered by the gate â not by this list." 616 } 617 }, 618 { 619 number: 11, key: "documentation", weightPct: 0, informational: true, iso: "Maintainability / Usability", 620 automation: "ProxyReview", diagram: null, 621 title: { pt: "Documentação", en: "Documentation" }, 622 tagline: { 623 pt: "Um dev novo sobe o projeto só com o README?", 624 en: "Can a new dev bring the project up with only the README?" 625 }, 626 body: { 627 pt: "Medida e reportada, mas SEM peso. No peso antigo (1%), a diferença entre um README impecável e NENHUM README mexia 0,05 na nota final â menos que o ruÃdo entre dois runs do mesmo modelo no mesmo prompt. Era um número com cara de medição, incapaz de agir como uma. O que de fato importa no contrato â o OpenAPI descrever mesmo os endpoints â é cobrado AO VIVO no critério 4, onde vale ponto.", 628 en: "Measured and reported, but with NO weight. At its old 1%, the gap between a flawless README and NO README at all moved the final score by 0.05 â less than the run-to-run noise of the same model on the same prompt. It was a number that looked like a measurement and could not act like one. What genuinely matters about the contract â that the OpenAPI really describes the endpoints â is asserted LIVE in criterion 4, where it counts." 629 }, 630 look: { 631 pt: ["README com propósito, setup/pré-requisitos e como rodar", "Stack e variáveis de ambiente documentadas"], 632 en: ["A README with purpose, setup/prerequisites and how to run", "The stack and the environment variables documented"] 633 }, 634 how: { 635 pt: "Parsing das seções do README. Doc-comments (densidade de ///) saiu da rubrica: mede digitação, não engenharia, e é trivialmente burlável por um modelo que comenta toda propriedade.", 636 en: "Parsing of the README's sections. Doc comments (/// density) are out of the rubric: they measure typing, not engineering, and are trivially gamed by a model that comments every property." 637 } 638 } 639 ], 640 641 /* ---- per-metric breakdown: exactly what the evaluator emits, per category ---- 642 `t: "live"` = asserted against the running system; `t: "deep"` = needs the deep harness. 643 Categories 9â11 are informational: their metrics are reported, never scored. */ 644 criteriaChecks: { 645 1: [ 646 { n: "create-card-201", w: 1, t: "live", how: { 647 pt: "POST /credit-cards com payload VISA válido tem de responder 201 Created (requisição HTTP real).", 648 en: "POST /credit-cards with a valid VISA payload must answer 201 Created (real HTTP request)." } }, 649 { n: "create-card-id", w: 1, t: "live", how: { 650 pt: "A resposta da criação traz o id novo no corpo JSON (aceita envelope data/value).", 651 en: "The create response returns the new id in the JSON body (data/value envelope tolerated)." } }, 652 { n: "card-required-400", w: 1, t: "live", how: { 653 pt: "POST com cardholderName/cardNumber vazios tem de responder 400.", 654 en: "POST with empty cardholderName/cardNumber must answer 400." } }, 655 { n: "list-cards-200", w: 0.5, t: "live", how: { 656 pt: "GET /credit-cards (a coleção) tem de responder 200.", 657 en: "GET /credit-cards (the collection) must answer 200." } }, 658 { n: "get-card-200", w: 1, t: "live", how: { 659 pt: "GET /credit-cards/{id criado} tem de responder 200.", 660 en: "GET /credit-cards/{created id} must answer 200." } }, 661 { n: "get-card-404", w: 1, t: "live", how: { 662 pt: "GET /credit-cards/{id inexistente = 999000111} tem de responder 404.", 663 en: "GET /credit-cards/{missing id = 999000111} must answer 404." } }, 664 { n: "create-tx-201", w: 1, t: "live", how: { 665 pt: "POST /transactions com um creditCardId válido tem de responder 201 Created.", 666 en: "POST /transactions with a valid creditCardId must answer 201 Created." } }, 667 { n: "create-tx-id", w: 1, t: "live", how: { 668 pt: "A resposta da criação da transação traz o id novo no corpo JSON.", 669 en: "The transaction create response returns the new id in the JSON body." } }, 670 { n: "create-tx-echo", w: 0.5, t: "live", how: { 671 pt: "A resposta ecoa os campos persistidos: amount = 199.90 e merchant = 'Amazon'.", 672 en: "The response echoes the persisted fields: amount = 199.90 and merchant = 'Amazon'." } }, 673 { n: "tx-amount-positive-400", w: 1.5, t: "live", how: { 674 pt: "amount menor ou igual a 0 tem de responder 400 (regra de negócio, peso reforçado).", 675 en: "amount less than or equal to 0 must answer 400 (business rule, weighted up)." } }, 676 { n: "tx-merchant-required-400", w: 1.5, t: "live", how: { 677 pt: "merchant vazio tem de responder 400.", 678 en: "empty merchant must answer 400." } }, 679 { n: "tx-fk-exists-400", w: 1.5, t: "live", how: {
680 pt: "creditCardId inexistente tem de responder 400 â integridade de FK cobrada pela API.", 681 en: "a non-existent creditCardId must answer 400 â FK integrity enforced at the API." } }, 682 { n: "list-tx-200", w: 0.5, t: "live", how: { 683 pt: "GET /transactions (a coleção) tem de responder 200.", 684 en: "GET /transactions (the collection) must answer 200." } }, 685 { n: "get-tx-200", w: 1, t: "live", how: { 686 pt: "GET /transactions/{id criado} tem de responder 200.", 687 en: "GET /transactions/{created id} must answer 200." } }, 688 { n: "get-tx-404", w: 1, t: "live", how: { 689 pt: "GET /transactions/{id inexistente} tem de responder 404.", 690 en: "GET /transactions/{missing id} must answer 404." } }, 691 { n: "card-transactions-200", w: 1, t: "live", how: { 692 pt: "GET /credit-cards/{id}/transactions tem de responder 200 (a relação 1:N exposta).", 693 en: "GET /credit-cards/{id}/transactions must answer 200 (the 1:N relation exposed)." } }, 694 { n: "card-transactions-404", w: 0.5, t: "live", how: { 695 pt: "GET /credit-cards/{id inexistente}/transactions tem de responder 404, não uma lista vazia.", 696 en: "GET /credit-cards/{missing id}/transactions must answer 404, not an empty list." } }, 697 { n: "unit-tests", w: 1, how: { 698 pt: "Projeto de teste que declara casos de verdade â o AST enxerga [Fact]/[Theory]/[Test], não só um csproj referenciando o framework. Substitui os antigos test-project/test-framework/coverage-tool: eram três checagens de presença para o mesmo fato, e 'o pacote está referenciado' não é sinal de engenharia.", 699 en: "A test project that actually declares test cases â the AST sees [Fact]/[Theory]/[Test], not just a csproj referencing the framework. It replaces the old test-project/test-framework/coverage-tool: three presence checks for one fact, and 'a package is referenced' is not an engineering signal." } }, 700 { n: "unit-only", w: 1, how: { 701 pt: "Pacote Testcontainers â Fail (proibido: exige daemon Docker e sobe Postgres/Kafka a cada run). WebApplicationFactory no AST â Partial (roda em processo, mas é teste de aceitação que a tarefa não pediu â esse papel é do oráculo vivo). Nenhum dos dois â Pass.", 702 en: "A Testcontainers package â Fail (forbidden: needs a Docker daemon and boots a Postgres/Kafka per run). WebApplicationFactory in the AST â Partial (in-process, but an acceptance test the task never asked for â that job belongs to the live oracle). Neither â Pass." } }, 703 { n: "test-pass-rate", w: 1, t: "deep", how: { 704 pt: "Roda `dotnet test` uma única vez; regex extrai Passed/Failed; nota = passed / total. Peso baixo de propósito: é a suÃte que o próprio modelo escreveu â sinal auto-avaliado, ao lado (e muito abaixo) do oráculo independente.", 705 en: "Runs `dotnet test` exactly once; a regex extracts Passed/Failed; score = passed / total. Deliberately low weight: it is the suite the model wrote itself â a self-graded signal, sitting beside (and far below) the independent oracle." } }, 706 { n: "coverage", w: 2, t: "deep", how: { 707 pt: "Da MESMA execução do `dotnet test` (XPlat Code Coverage), faz merge de todos os coverage.cobertura.xml (união das linhas cobertas); LineRate â¥60% â Pass, â¥35% â Partial (régua relaxada de propósito, para não incentivar teste de enchimento).", 708 en: "From the SAME `dotnet test` run (XPlat Code Coverage), merges every coverage.cobertura.xml (union of covered lines); LineRate â¥60% â Pass, â¥35% â Partial (a deliberately relaxed bar, so there is no incentive to pad tests)." } } 709 ], 710 2: [ 711 { n: "layering", w: 1, how: { 712 pt: "Existem as três camadas por pasta: (Domain|Entities|Models) E (Infrastructure|Repositories|Data) E (Controllers|Api|Endpoints).", 713 en: "The three layers exist as folders: (Domain|Entities|Models) AND (Infrastructure|Repositories|Data) AND (Controllers|Api|Endpoints)." } }, 714 { n: "application-layer", w: 1, how: { 715 pt: "Existe pasta UseCases, Application, Services ou Handlers.", 716 en: "A UseCases, Application, Services or Handlers folder exists." } }, 717 { n: "dependency-direction", w: 1, how: { 718 pt: "Roslyn lê os `using` dos arquivos sob Domain/Entities e conta os que referenciam EntityFrameworkCore|Npgsql|Confluent.Kafka; 0 vazamentos â Pass.", 719 en: "Roslyn reads the `using`s of files under Domain/Entities and counts those referencing EntityFrameworkCore|Npgsql|Confluent.Kafka; 0 leaks â Pass." } }, 720 { n: "no-gold-plating", w: 1, how: { 721 pt: "Conta a maquinaria que o brief PROIBIU: endpoints PUT/PATCH/DELETE, consumer Kafka, outbox transacional, SDK do OpenTelemetry, versionamento de API, Testcontainers. 0 â Pass; 1â2 â Partial; â¥3 â Fail. Substitui o antigo overengineering-proxy, que contava interfaces com uma só implementação â justamente as portas de inversão de dependência que esta categoria premia, e por isso nunca reprovava ninguém.", 722 en: "Counts the machinery the brief RULED OUT: PUT/PATCH/DELETE endpoints, a Kafka consumer, a transactional outbox, the OpenTelemetry SDK, API versioning, Testcontainers. 0 â Pass; 1â2 â Partial; â¥3 â Fail. It replaces the old overengineering-proxy, which counted single-implementation interfaces â precisely the dependency-inversion ports this category rewards, which is why it could never fail anyone." } }, 723 { n: "no-god-class", w: 0.5, how: { 724 pt: "O maior tipo (linhas medidas pelo AST) tem â¤600 linhas â Pass; senão Partial.", 725 en: "The largest type (lines measured by the AST) is â¤600 lines â Pass; otherwise Partial." } } 726 ], 727 3: [ 728 { n: "no-empty-catch", w: 1, how: { 729 pt: "AST: conta blocos `catch` com 0 statements (exceção engolida); contagem 0 â Pass.", 730 en: "AST: counts `catch` blocks with 0 statements (swallowed exception); count 0 â Pass." } }, 731 { n: "no-todos", w: 1, how: { 732 pt: "Regex \\b(TODO|FIXME|HACK)\\b sobre a trivia de comentário do AST; 0 â Pass, â¤3 â Partial.", 733 en: "Regex \\b(TODO|FIXME|HACK)\\b over the AST comment trivia; 0 â Pass, â¤3 â Partial." } }, 734 { n: "analyzers-enabled", w: 1, how: { 735 pt: "TreatWarningsAsErrors=true, ou EnableNETAnalyzers=true, ou existe .editor
735config.", 736 en: "TreatWarningsAsErrors=true, or EnableNETAnalyzers=true, or an .editorconfig exists." } }, 737 { n: "async-io", w: 1, how: { 738 pt: "AST: contagem de métodos marcados async > 0. Veio da antiga categoria Performance.", 739 en: "AST: count of methods marked async > 0. Moved here from the old Performance category." } }, 740 { n: "no-sync-over-async", w: 1, how: { 741 pt: "AST: qualquer .Result/.Wait()/.GetAwaiter().GetResult() â Fail. Antes era meia-nota (Partial): num serviço ASP.NET Core, bloquear o caminho da requisição é starvation do thread-pool â bug, não estilo.", 742 en: "AST: any .Result/.Wait()/.GetAwaiter().GetResult() â Fail. It used to be half credit (Partial): in an ASP.NET Core service, blocking the request path is thread-pool starvation â a bug, not a style choice." } }, 743 { n: "format", w: 1, t: "deep", how: { 744 pt: "`dotnet format --verify-no-changes`; sem mudanças pendentes â Pass.", 745 en: "`dotnet format --verify-no-changes`; no pending changes â Pass." } }, 746 { n: "build-warnings", w: 1, t: "deep", how: { 747 pt: "Contagem de warnings do build Release único do harness; 0 â Pass, â¤10 â Partial.", 748 en: "Warning count from the harness's single Release build; 0 â Pass, â¤10 â Partial." } } 749 ], 750 4: [ 751 { n: "http-verbs", w: 1, how: { 752 pt: "Atributos [HttpGet]/[HttpPost] ou invocações MapGet/MapPost (Richardson L2). PUT/DELETE não somam aqui â são penalizados como gold-plating no critério 2.", 753 en: "[HttpGet]/[HttpPost] attributes or MapGet/MapPost invocations (Richardson L2). PUT/DELETE earn nothing here â they are penalised as gold-plating under criterion 2." } }, 754 { n: "problem-details", w: 1, how: { 755 pt: "new ProblemDetails, ou AddProblemDetails/Problem(), ou IExceptionHandler (RFC 9457).", 756 en: "new ProblemDetails, or AddProblemDetails/Problem(), or IExceptionHandler (RFC 9457)." } }, 757 { n: "dtos", w: 0.5, how: { 758 pt: "Pasta Dtos/DTOs, ou tipos cujo nome contém Request/Response/Dto.", 759 en: "A Dtos/DTOs folder, or types whose names contain Request/Response/Dto." } }, 760 { n: "create-card-location", w: 0.5, t: "live", how: { 761 pt: "A resposta 201 da criação do cartão traz o header Location.", 762 en: "The card create 201 response carries a Location header." } }, 763 { n: "json-camelcase", w: 0.5, t: "live", how: { 764 pt: "Toda chave do JSON de resposta é camelCase (nenhuma começa maiúscula nem contém _).", 765 en: "Every response JSON key is camelCase (none starts uppercase or contains _)." } }, 766 { n: "problem-details-live", w: 0.5, t: "live", how: { 767 pt: "O corpo de erro tem Content-Type application/problem+json (RFC 9457).", 768 en: "The error body has Content-Type application/problem+json (RFC 9457)." } }, 769 { n: "create-tx-location", w: 0.5, t: "live", how: { 770 pt: "A resposta 201 da criação da transação traz o header Location.", 771 en: "The transaction create 201 response carries a Location header." } }, 772 { n: "pagination", w: 0.5, t: "live", how: { 773 pt: "Tenta pageSize=1/limit=1/perPage=1/⦠e exige exatamente 1 item ou metadados de paginação (a coleção é semeada com 2 cartões). O antigo check estático de paginação (procurar Skip/Take no código) saiu: provava que a palavra existe, não que a API pagina.", 774 en: "Tries pageSize=1/limit=1/perPage=1/⦠and requires exactly 1 item or paging metadata (the collection is seeded with 2 cards). The old static pagination check (looking for Skip/Take in the source) is gone: it proved the word exists, not that the API paginates." } }, 775 { n: "openapi-populated", w: 1, t: "live", how: { 776 pt: "Baixa o OpenAPI servido e conta as operações. ops > 0 â Pass; 0 (paths vazio) â Fail (contrato vazio e inútil); NENHUM doc servido â Fail também â a tarefa exige OpenAPI, então a ausência é defeito, não medição faltante. Substitui de vez o antigo check estático 'openapi' (que só provava o middleware ligado, e ainda era contado de novo em Documentação como api-docs).", 777 en: "Fetches the served OpenAPI and counts operations. ops >
777 0 â Pass; 0 (empty paths) â Fail (an empty, useless contract); NO doc served â Fail as well â the task requires OpenAPI, so its absence is a defect, not a missing measurement. It fully replaces the old static 'openapi' check (which only proved the middleware was wired, and was counted a second time under Documentation as api-docs)." } } 778 ], 779 5: [ 780 { n: "migrations", w: 1, how: { 781 pt: "Pasta Migrations / MigrationBuilder / arquivo .sql presente E ausência de EnsureCreated; com EnsureCreated â Partial.", 782 en: "Migrations folder / MigrationBuilder / a .sql file present AND no EnsureCreated; with EnsureCreated â Partial." } }, 783 { n: "referential-integrity", w: 1, how: { 784 pt: "Roslyn deduz a relação: nav property para outra entidade DbSet, FK <Outra>Id (int/long/Guid), [ForeignKey], ou HasForeignKey/HasOne/WithMany.", 785 en: "Roslyn infers the relationship: a nav property to another DbSet entity, an <Other>Id FK (int/long/Guid), [ForeignKey], or HasForeignKey/HasOne/WithMany." } }, 786 { n: "indexes", w: 0.5, how: { 787 pt: "Invocações HasIndex ou CreateIndex.", 788 en: "HasIndex or CreateIndex invocations." } }, 789 { n: "read-perf", w: 0.5, how: { 790 pt: "AsNoTracking / AsNoTrackingWithIdentityResolution nas leituras.", 791 en: "AsNoTracking / AsNoTrackingWithIdentityResolution on reads." } } 792 ], 793 6: [ 794 { n: "broker-client", w: 1, how: { 795 pt: "Pacote Confluent.Kafka/MassTransit, ou os genéricos IProducer/ProducerBuilder.", 796 en: "Confluent.Kafka/MassTransit package, or the IProducer/ProducerBuilder generics." } }, 797 { n: "publishes", w: 1, how: { 798 pt: "Uma chamada de publish no AST: Produce ou ProduceAsync.", 799 en: "A publish call in the AST: Produce or ProduceAsync." } }, 800 { n: "durable-producer", w: 1, how: { 801 pt: "Acesso a membro Acks.All, ou o identificador EnableIdempotence.", 802 en: "Member access Acks.All, or the EnableIdempotence identifier." } }, 803 { n: "kafka-event-live", w: 1, t: "live", how: { 804 pt: "O sidecar kafka-check (kcat) do harness consome o tópico transactions durante a run: evento com key = id â Pass; evento com outra key â Parcial; nenhum evento â Fail; broker inalcançável â Indeterminado.", 805 en: "The harness kafka-check (kcat) sidecar consumes the transactions topic during the run: an event keyed by id â Pass; an event with another key â Partial; no event â Fail; broker unreachable â Indeterminate." } } 806 ], 807 7: [ 808 { n: "pci-pan", w: 1, how: { 809 pt: "Regex de 13â19 dÃgitos sobre literais string de produção + valores de appsettings.json, filtrado por Luhn; qualquer sequência válida â Fail (testes excluÃdos).", 810 en: "A 13â19 digit regex over production string literals + appsettings.json values, filtered by Luhn; any valid sequence â Fail (tests excluded)." } }, 811 { n: "pci-sad", w: 1, how: { 812 pt: "Identificador contendo cvv/cvc/cardverification/track2/pinblock â Fail (dado sensÃvel de autenticação).", 813 en: "An identifier containing cvv/cvc/cardverification/track2/pinblock â Fail (sensitive auth data)." } }, 814 { n: "validation", w: 0.5, how: { 815 pt: "FluentValidation, ou [Required]/[Range]/[StringLength], ou ModelState.", 816 en: "FluentValidation, or [Required]/[Range]/[StringLength], or ModelState." } }, 817 { n: "rate-limit", w: 0.5, how: { 818 pt: "AddRateLimiter / RequireRateLimiting (OWASP API #4).", 819 en: "AddRateLimiter / RequireRateLimiting (OWASP API #4)." } }, 820 { n: "secrets", w: 1, how: { 821 pt: "`gitleaks detect --source <root> --no-git --no-banner`; exit 0 â Pass.", 822 en: "`gitleaks detect --source <root> --no-git --no-banner`; exit 0 â Pass." } }, 823 { n: "sca", w: 1, t: "deep", how: { 824 pt: "`dotnet list package --vulnerable --include-transitive`; procura 'vulnerable' ou > High/Critical no output. Sem fonte NuGet alcançável â Indeterminado, nunca um Pass silencioso.", 825 en: "`dotnet list package --vulnerable --include-transitive`; looks for 'vulnerable' or >
825 High/Critical in the output. With no reachable NuGet source â Indeterminate, never a silent Pass." } } 826 ], 827 8: [ 828 { n: "resilience-policies", w: 1, how: { 829 pt: "Pacote Polly/Microsoft.Extensions.Http.Resilience, ou AddResilienceHandler/WaitAndRetry/AddPolicyHandler/AddStandardResilienceHandler.", 830 en: "Polly/Microsoft.Extensions.Http.Resilience package, or AddResilienceHandler/WaitAndRetry/AddPolicyHandler/AddStandardResilienceHandler." } }, 831 { n: "global-error-handling", w: 1, how: { 832 pt: "IExceptionHandler, ou UseExceptionHandler/UseProblemDetails/AddProblemDetails.", 833 en: "IExceptionHandler, or UseExceptionHandler/UseProblemDetails/AddProblemDetails." } }, 834 { n: "graceful-shutdown", w: 0.5, how: { 835 pt: "IHostApplicationLifetime/BackgroundService, ApplicationStopping, ou StopAsync.", 836 en: "IHostApplicationLifetime/BackgroundService, ApplicationStopping, or StopAsync." } }, 837 { n: "timeouts", w: 0.5, how: { 838 pt: "AddRequestTimeouts, ou CommandTimeout/CancellationToken.", 839 en: "AddRequestTimeouts, or CommandTimeout/CancellationToken." } }, 840 { n: "no-stacktrace-leak", w: 1, t: "live", how: { 841 pt: "Envia um JSON malformado; falha se o status for â¥500 ou se o corpo contiver marcadores de exceção (StackTrace, ' at ', .cs:line, EntityFrameworkCore, Npgsql., DbUpdateException).", 842 en: "Sends a malformed JSON; fails if the status is â¥500 or the body contains exception markers (StackTrace, ' at ', .cs:line, EntityFrameworkCore, Npgsql., DbUpdateException)." } } 843 ], 844 9: [ 845 { n: "structured-logs", w: 1, how: { 846 pt: "Pacote Serilog, ou AddJsonConsole/UseSerilog/AddSerilog.", 847 en: "Serilog package, or AddJsonConsole/UseSerilog/AddSerilog." } }, 848 { n: "correlation", w: 1, how: { 849 pt: "Identificador CorrelationId/TraceId/traceparent, ou acesso a Activity.Current.", 850 en: "A CorrelationId/TraceId/traceparent identifier, or Activity.Current access." } }, 851 { n: "live-health", w: 1, t: "live", how: { 852 pt: "GET {base}/health tem de responder 2xx/3xx no sistema vivo. Os antigos checks estáticos health-endpoint e metrics-endpoint saÃram: o primeiro era a TERCEIRA cópia do mesmo sinal de health (o portão de executabilidade já capa em 1.5 quem não responde), e o segundo pertencia a um requisito (/metrics) que a tarefa deixou de pedir.", 853 en: "GET {base}/health must answer 2xx/3xx on the live system. The old static health-endpoint and metrics-endpoint checks are gone: the first was the THIRD copy of the same health signal (the executability gate already caps a service that never answers at 1.5), and the second belonged to a requirement (/metrics) the task has dropped." } } 854 ], 855 10: [ 856 { n: "dockerfile", w: 1, how: { 857 pt: "Existe um arquivo Dockerfile.", 858 en: "A Dockerfile exists." } }, 859 { n: "compose", w: 1, how: { 860 pt: "docker-compose*.yml, ou compose.yaml/compose.yml.", 861 en: "docker-compose*.yml, or compose.yaml/compose.yml." } }, 862 { n: "env-config", w: 1, how: { 863 pt: "GetEnvironmentVariable, ou IConfiguration, ou builder.Configuration (12-Factor III/IV).", 864 en: "GetEnvironmentVariable, or IConfiguration, or builder.Configuration (12-Factor III/IV)." } }, 865 { n: "pinning", w: 0.5, how: { 866 pt: "packages.lock.json, ou global.json, ou Directory.Packages.props com ManagePackageVersionsCentrally=true (lido via XML).", 867 en: "packages.lock.json, or global.json, or Directory.Packages.props with ManagePackageVersionsCentrally=true (read via XML)." } }, 868 { n: "non-root", w: 0.5, how: { 869 pt: "Regex ^\\s*USER\\s+ (multiline) encontra uma diretiva USER no Dockerfile. O antigo check 'ci' saiu: nada aqui executa o workflow â ele pontuava a existência de um YAML.", 870 en: "Regex ^\\s*USER\\s+ (multiline) finds a USER directive in the Dockerfile. The old 'ci' check is gone: nothing here runs the workflow â it scored the existence of a YAML file." } }, 871 { n: "hadolint", w: 0.5, t: "deep", how: { 872 pt: "`hadolint <Dockerfile>`; limpo â Pass, senão Partial.", 873 en: "`hadolint <Dockerfile>`; clean â Pass, otherwise Partial." } } 874 ], 875 11: [ 876 { n: "readme", w: 1, how: { 877 pt: "Existe README.md.", 878 en: "A README.md exists." } }, 879 { n: "readme-sections", w: 1, how: { 880 pt: "Três regex sobre o README (purpose/overview; setup/install/prereq; run/usage/docker compose); nota = seções encontradas / 3. Os antigos api-docs (duplicata do check de OpenAPI do critério 4) e doc-comments (densidade de ///) saÃram.", 881 en: "Three regexes over the README (purpose/overview; setup/install/prereq; run/usage/docker compose);
881 score = sections found / 3. The old api-docs (a duplicate of criterion 4's OpenAPI check) and doc-comments (/// density) are gone." } } 882 ] 883 } 884};
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.