1<!DOCTYPE html><!--Vv4HRFzSbAJZU99gF7vmh--><html lang="en"><head><meta charSet="utf-8"/><meta name="viewport" content="width=device-width, initial-scale=1"/><link rel="preload" href="/_next/static/media/248e1dc0efc99276-s.p.8a6b2436.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/media/797e433ab948586e-s.p.29207c2f.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/media/caa3a2e1cccd8315-s.p.3b6cae6d.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="stylesheet" href="/_next/static/chunks/4eaf45b13857288d.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/chunks/e55c23589ab78d68.css" data-precedence="next"/><link rel="preload" as="script" fetchPriority="low" href="/_next/static/chunks/15d2e4c42c92d9d3.js"/>
1<script src="/_next/static/chunks/32dd778f0fb2c3aa.js" async=""></script>
1<script src="/_next/static/chunks/9506605a49babbca.js" async=""></script>
1<script src="/_next/static/chunks/4c641512fe57ba9c.js" async=""></script>
1<script src="/_next/static/chunks/45aa5395923444b5.js" async=""></script>
1<script src="/_next/static/chunks/turbopack-5e3ba5206c83fb74.js" async=""></script>
1<script src="/_next/static/chunks/e9aa5be042e3b0a3.js" async=""></script>
1<script src="/_next/static/chunks/8a2c28f82c6d4f4c.js" async=""></script>
1<script src="/_next/static/chunks/3e8b73794b0ba35e.js" async=""></script>
1<script src="/_next/static/chunks/18b6abd70bbeface.js" async=""></script>
1<script src="/_next/static/chunks/8827248f36da4bf2.js" async=""></script>
1<script src="/_next/static/chunks/47f82402146413fb.js" async=""></script>
1<script src="/_next/static/chunks/b7312459423597a8.js" async=""></script>
1<script src="/_next/static/chunks/705056c08acab783.js" async=""></script>
1<script src="/_next/static/chunks/1a514990f47add69.js" async=""></script>
1<link rel="preload" href="https://policy.app.cookieinformation.com/uc.js" as="script"/><link rel="preload" href="https://www.googletagmanager.com/gtag/js?id=G-P67YVZS2LZ&l=gtagLayer" as="script"/><meta name="next-size-adjust" content=""/><title>Workshop Assessment | Visdom Maturity Matrix</title><meta name="description" content="Walk 20 areas, check criteria you have, find the level blocking your agents. Generates a shareable maturity report by perspective."/><link rel="canonical" href="https://visdom-maturity-matrix.virtuslab.com/workshop"/><meta property="og:title" content="Workshop assessment"/><meta property="og:description" content="Walk 16 capabilities, check the criteria you have, and find the level blocking your agents. Generates a shareable maturity report."/><meta property="og:url" content="https://visdom-maturity-matrix.virtuslab.com/workshop"/><meta property="og:image:alt" content="Workshop assessment - find the level blocking your agents"/><meta property="og:image:type" content="image/png"/><meta property="og:image" content="https://visdom-maturity-matrix.virtuslab.com/workshop/opengraph-image?c1d725e4417a00ac"/><meta property="og:image:width" content="1200"/><meta property="og:image:height" content="630"/><meta property="og:type" content="website"/><meta name="twitter:card" content="summary_large_image"/><meta name="twitter:title" content="Workshop assessment"/><meta name="twitter:description" content="Walk 16 capabilities, check the criteria you have, and find the level blocking your agents. Generates a shareable maturity report."/><meta name="twitter:image" content="https://visdom-maturity-matrix.virtuslab.com/workshop/opengraph-image"/><link rel="icon" href="/favicon.ico" sizes="32x32"/><link rel="icon" href="/favicon/favicon-light-32x32.png" type="image/png" sizes="32x32" media="(prefers-color-scheme: light)"/><link rel="icon" href="/favicon/favicon-dark-32x32.png" type="image/png" sizes="32x32" media="(prefers-color-scheme: dark)"/><link rel="apple-touch-icon" href="/favicon/apple-touch-icon-light.png" sizes="180x180"/>
1<script src="/_next/static/chunks/a6dad97d9634a72d.js" noModule=""></script>
1</head><body class="geist_9c6cb61b-module__8NX9hq__variable geist_mono_d6617093-module__z61v7q__variable merriweather_519133d3-module__aYUzJa__variable"><div hidden=""><!--$--><!--/$--></div>
1<script type="application/ld+json">{"@context":"https://schema.org","@graph":[{"@type":"Organization","@id":"https://visdom-maturity-matrix.virtuslab.com#organization","name":"VirtusLab","url":"https://virtuslab.com","logo":"https://visdom-maturity-matrix.virtuslab.com/og-image.png","sameAs":["https://www.linkedin.com/company/virtuslab"]},{"@type":"WebSite","@id":"https://visdom-maturity-matrix.virtuslab.com#website","url":"https://visdom-maturity-matrix.virtuslab.com","name":"AI-Native SDLC Maturity Matrix | Visdom by VirtusLab","description":"Most teams stall at Level 2 and conclude AI doesn't work. It works great - at Level 4. The opinionated map there: 60 practices, 4 perspectives, 5 levels, built from years of real engineering engagements.","publisher":{"@id":"https://visdom-maturity-matrix.virtuslab.com#organization"},"inLanguage":"en"}]}</script>
1<script>(function(){ 2 if (typeof navigator === 'undefined') return; 3 var endpoint = '/api/mcp/matrix'; 4 var tools = []; 5 if (!navigator.modelContext) { 6 navigator.modelContext = { 7 tools: tools, 8 registerTool: function(t){ 9 tools.push(t); 10 return Promise.resolve(); 11 }, 12 provideContext: function(ctx){ 13 if (ctx && ctx.tools) ctx.tools.forEach(function(t){ tools.push(t); }); 14 return Promise.resolve(); 15 } 16 }; 17 } else if (!navigator.modelContext.tools) { 18 navigator.modelContext.tools = tools; 19 } 20 21 async function callMcp(name, args){ 22 var res = await fetch(endpoint, { 23 method: 'POST', 24 headers: { 'content-type': 'application/json' }, 25 body: JSON.stringify({ jsonrpc: '2.0', id: Date.now(), method: 'tools/call', params: { name: name, arguments: args || {} } }) 26 }); 27 var json = await res.json(); 28 if (json.error) throw new Error(json.error.message); 29 var content = json.result && json.result.content; 30 if (content && content[0] && content[0].text) { 31 try { return JSON.parse(content[0].text); } catch (e) { return content[0].text; } 32 } 33 return json.result; 34 } 35 36 function reg(t){ 37 if (typeof navigator.modelContext.registerTool === 'function') { 38 try { navigator.modelContext.registerTool(t); } 39 catch (e) { (navigator.modelContext.tools || tools).push(t); } 40 } else { 41 (navigator.modelContext.tools || tools).push(t); 42 } 43 } 44 45 reg({ 46 name: 'get_perspectives', 47 description: 'List the four perspectives of the AI-Native SDLC Maturity Matrix with their areas.', 48 inputSchema: { type: 'object', properties: {}, additionalProperties: false }, 49 execute: function(){ return callMcp('get_perspectives', {}); } 50 }); 51 reg({ 52 name: 'get_area', 53 description: 'Return one area of the matrix with all 5 levels and their bullet items.', 54 inputSchema: { 55 type: 'object', 56 properties: { 57 perspective: { type: 'string', enum: ['development','delivery','infrastructure','organization'] }, 58 area: { type: 'string' } 59 }, 60 required: ['perspective','area'], 61 additionalProperties: false 62 }, 63 execute: function(args){ return callMcp('get_area', args); } 64 }); 65 reg({ 66 name: 'get_gates', 67 description: 'Return Must/Should criteria for a perspective+area, at one level or all 5.', 68 inputSchema: { 69 type: 'object', 70 properties: { 71 perspective: { type: 'string', enum: ['development','delivery','infrastructure','organization'] }, 72 area: { type: 'string' }, 73 level: { type: 'integer', minimum: 1, maximum: 5 } 74 }, 75 required: ['perspective','area'], 76 additionalProperties: false 77 }, 78 execute: function(args){ return callMcp('get_gates', args); } 79 }); 80 reg({ 81 name: 'search_guides', 82 description: 'Substring search across guide title/slug/summary across 240 practice guides.', 83 inputSchema: { 84 type: 'object', 85 properties: { query: { type: 'string' }, perspective: { type: 'string' }, limit: { type: 'integer' } }, 86 required: ['query'], 87 additionalProperties: false 88 }, 89 execute: function(args){ return callMcp('search_guides', args); } 90 }); 91 reg({ 92 name: 'recommend_next_steps', 93 description: 'From a scores object (perspective -> array of per-area levels 1-5), return lowest-scoring areas with the L+1 gate to target.', 94 inputSchema: { 95 type: 'object', 96 properties: { scores: { type: 'object' }, limit: { type: 'integer' } }, 97 required: ['scores'], 98 additionalProperties: false 99 }, 100 execute: function(args){ return callMcp('recommend_next_steps', args); } 101 }); 102})();</script>
102<script>(self.__next_s=self.__next_s||[]).push(["https://policy.app.cookieinformation.com/uc.js",{"data-culture":"EN","data-gcm-version":"2.0","id":"cookie-information-consent"}])</script>
102<main class="min-h-screen"><header class="sticky top-0 z-50 print:hidden flex items-center justify-between gap-4" style="padding:22px clamp(20px, 5vw, 80px);background:color-mix(in oklab, var(--background) 72%, transparent);backdrop-filter:blur(14px);-webkit-backdrop-filter:blur(14px);border-bottom:1px solid transparent;transition:border-color .25s"><div class="flex items-center gap-1 min-w-0"><a aria-label="Visdom Maturity Matrix, home" class="flex items-center gap-2.5 no-underline mr-4 shrink-0" href="/"><svg viewBox="0 0 17.7 49.9" aria-hidden="true" fill="currentColor" class="h-9 w-auto text-foreground"><path fill-rule="evenodd" clip-rule="evenodd" d="M6.4097 14.78C6.20417 14.5233 6.01478 14.2725 5.85111 14.0381C5.64181 13.7386 5.45721 13.4229 5.41501 13.0599L5.10338 10.3676C5.06598 10.0441 5.10991 9.71645 5.23172 9.4145L7.18039 4.5806C7.19136 4.5535 7.20272 4.52677 7.21473 4.50035L8.77253 1.08452C8.81096 1.00012 8.89534 0.945801 8.98798 0.945801C9.08097 0.945801 9.165 1.00012 9.20378 1.08452L10.7606 4.49817C10.7726 4.52494 10.7843 4.55182 10.7952 4.57893L12.7424 9.41031C12.8652 9.715 12.9095 10.0458 12.8707 10.3722L12.5502 13.0798C12.5073 13.4421 12.3241 13.7581 12.1135 14.0559C11.9512 14.2851 11.7648 14.5291 11.5641 14.7778L9.23678 9.84005L10.3923 7.38842C10.4455 7.27553 10.5193 6.86091 10.4067 6.80773C10.2942 6.75488 10.0372 7.08337 9.98435 7.19592L8.98765 9.31096L7.99078 7.19592C7.9376 7.08337 7.69603 6.74918 7.58349 6.80236C7.47094 6.85521 7.52964 7.27553 7.58248 7.38842L8.73818 9.84005L6.4097 14.78ZM11.2464 15.1622C10.6515 15.8649 9.97986 16.5724 9.43498 17.0864C9.31077 17.2037 9.1592 17.2878 8.98832 17.2878C8.8171 17.2878 8.66674 17.202 8.54115 17.086C7.97466 16.5628 7.31 15.8598 6.72635 15.166L8.98765 10.3688L11.2464 15.1622ZM5.4971 29.0027C5.5297 28.8209 5.39581 28.4411 4.86363 28.3986C4.16503 28.3426 4.02234 29.4203 2.55274 29.5895C1.92825 29.6615 2.09535 29.561 2.25971 29.1242C2.45392 28.6088 2.88345 28.0068 3.24853 27.8177C3.94576 27.4564 4.30121 27.7373 4.90099 27.9429C5.57557 28.1738 5.98352 28.5563 6.09573 28.6819C6.93947 29.6268 6.83346 29.4694 6.80258 29.8238C6.69106 31.1058 7.04594 31.6444 7.14235 31.9481C7.21407 32.1739 7.11623 32.248 6.82252 32.7555C6.59262 33.1532 6.49865 33.4048 6.23445 33.0582C5.27302 31.7972 5.25863 30.3323 5.4971 29.0027ZM12.0679 29.2463C11.8243 29.3726 11.4446 30.0109 11.2799 30.2651C10.6452 31.2448 9.30086 32.8814 8.48972 34.1352C8.42315 34.2378 7.23907 35.8602 7.67312 38.2627C7.72288 38.5389 7.69503 38.5139 7.42499 39.1058C7.25137 39.4863 7.17799 39.7141 7.08706 39.7312C6.92442 39.7617 6.65505 39.2364 6.58025 39.0404C6.24639 38.1638 5.90948 36.7934 6.37853 34.8787C6.89562 32.7678 9.39038 29.9882 9.92119 29.2312C10.3624 28.6016 11.5515 26.485 11.0114 24.9869C10.9795 24.8987 11.0896 24.7892 11.3634 24.5604C11.9494 24.0711 12.2837 23.814 12.3633 23.9921C13.208 25.8772 12.5142 27.7019 12.0657 28.7886C12.0516 28.8226 12.0912 28.8546 12.1262 28.8357C12.3664 28.706 12.6542 28.5125 12.7444 28.2424C13.3984 26.2846 15.7049 26.312 15.5303 26.5844C15.4908 26.6459 15.321 26.908 14.6869 27.8725C14.4844 28.1803 14.0577 28.704 13.426 28.6971C12.8986 28.691 12.3626 29.0936 12.0679 29.2463ZM4.45315 23.8797C4.5379 24.0306 4.87076 23.9833 4.93115 23.6789C5.23378 22.1489 5.41741 22.2182 5.72931 21.3855C5.84666 21.0719 5.83732 21.0488 5.53949 20.2353C5.40739 19.875 5.14603 19.444 4.04837 18.4448C3.18096 17.6553 2.74279 18.1045 1.53224 16.1771C1.18294 15.6209 0.74552 15.1498 0.786352 15.0626C0.945218 14.7216 3.58069 15.1261 3.86274 17.1574C3.96671 17.9064 4.14858 17.8656 4.39803 18.1171C4.49308 18.2132 4.99299 18.7756 5.50699 19.1372C5.6686 19.2512 5.80485 19.2838 5.93455 19.8493C6.10886 20.609 6.42823 20.5399 6.50235 20.4823C6.5847 20.4174 7.04081 19.923 7.09159 19.8677C7.41104 19.5208 7.40033 19.3335 6.34385 18.2798C5.40335 17.342 3.81705 15.7794 3.14315 13.9203C2.33441 11.6904 3.37039 9.04323 3.8041 8.5282C4.28893 7.95279 4.2158 8.52922 4.12762 8.90768C3.40123 12.0349 4.23469 13.3336 4.94907 14.2995C5.96815 15.6782 7.21814 17.1121 7.94487 17.7373C8.60093 18.3017 9.13866 18.5955 9.96559 17.7703C10.9672 16.7704 11.6812 15.9857 13.0187 14.3665C14.6382 12.4049 14.1109 10.451 13.9256 8.90216C13.8
102638 8.3837 13.9774 8.00734 14.3325 8.68569C15.3324 10.5955 15.5386 12.4154 14.7985 14.1204C14.4766 14.8619 13.6013 16.3384 11.3662 18.429C11.0454 18.7293 8.48628 20.9864 8.1538 21.3148C6.86743 22.586 6.35967 23.7324 6.58545 25.1962C6.86338 26.9979 8.71654 28.3866 9.28235 28.9695C9.45563 29.1483 9.44904 29.1929 8.96352 29.7498C8.51403 30.2648 8.50343 30.4035 8.02546 29.9914C7.55984 29.5899 5.31069 27.6995 4.88625 25.0431C4.81488 24.596 4.80386 24.6503 4.68067 24.5024C4.55749 24.3545 3.00792 24.0807 2.68744 22.0446C2.58141 21.3714 2.34336 20.6641 2.44836 20.6391C2.72835 20.5722 3.42277 21.1021 3.56619 21.2486C4.7908 22.4986 4.22566 23.4748 4.45315 23.8797ZM10.9214 20.6391C11.1688 20.4181 11.3954 19.6623 11.8967 19.3192C12.0874 19.1881 12.3318 19.3356 12.2322 19.4584C11.9066 19.8599 11.7371 19.9406 11.3957 20.8999C11.196 21.4613 11.2461 21.5251 11.3758 21.6703C11.4619 21.767 11.5052 21.9423 11.7879 21.7375C12.4745 21.2399 13.0677 20.7084 13.3373 19.8091C13.7776 18.3398 13.1933 17.468 13.207 17.2556C13.2149 17.1276 13.2849 17.0998 13.4558 16.8898C13.5855 16.731 13.7608 16.5161 13.8418 16.5982C14.0745 16.8335 14.292 17.4758 14.4217 18.1881C14.4707 18.4571 14.5034 18.6757 14.6928 18.4811C14.9621 18.2039 13.9577 16.4019 16.2457 15.3821C16.8856 15.0967 16.9224 15.1439 16.8229 15.6284C16.5453 16.9824 16.4063 17.1571 15.8858 17.9274C15.7616 18.1113 15.1199 18.7983 14.8022 18.9404C14.493 19.0787 14.5902 19.3095 14.4615 20.051C14.3706 20.5767 14.0689 21.2372 14.0926 21.2667C14.374 21.6177 14.9862 20.179 16.0845 22.0092C16.1181 22.0652 16.2948 22.3592 16.3796 22.5335C16.455 22.6882 16.0521 22.7112 15.7464 22.6316C14.5139 22.3101 14.21 21.2828 13.5704 21.9958C12.5167 23.171 11.8421 23.5797 10.5622 24.6534C8.65923 26.2499 8.16889 27.1451 7.98257 26.9924C7.77532 26.8229 7.33072 26.2433 7.23567 25.9225C7.16705 25.6919 8.49447 24.301 10.0931 23.0935C10.512 22.7772 9.15647 21.4287 9.04461 21.229C8.98182 21.1171 9.17841 20.9826 9.50301 20.7016C9.62173 20.5986 9.87671 20.367 10.0301 20.2288C10.1821 20.0919 10.5141 21.0028 10.9214 20.6391ZM8.5817 42.8429C8.72581 41.2893 9.2082 41.3884 9.24114 41.7566C9.27374 42.1186 9.50506 43.1226 9.62481 43.4709C9.95043 44.4203 10.1543 45.6593 9.86707 47.0177C9.59497 48.3058 9.25151 48.9424 8.98765 48.9458C8.74026 48.9489 8.48261 47.5538 8.45688 46.2328C8.45036 45.8866 8.84631 45.695 8.79347 44.8431C8.73617 43.9207 8.53126 43.3844 8.5817 42.8429ZM8.3964 35.2401C8.58546 34.8191 8.72821 34.5078 8.99619 34.8053C11.3442 37.4107 12.1043 39.9312 10.8601 43.0197C10.5406 43.8123 10.3052 43.9634 10.1618 43.5382C10.0674 43.2583 9.86771 42.6947 9.8334 42.4487C9.81796 42.3379 9.88139 42.2171 9.95687 41.9759C10.3398 40.7503 10.4786 39.2406 8.36423 36.3765C8.10551 36.0255 8.10577 35.8879 8.3964 35.2401ZM8.53814 37.9514C8.70695 37.6402 9.08677 38.4249 9.20378 38.6531C9.32147 38.8837 9.41759 39.0831 9.20897 39.5658C8.98388 40.0874 8.67543 40.7643 8.45068 41.2914C7.77027 42.8889 8.20157 43.9908 8.22902 44.6842C8.25442 45.3255 7.64261 44.7 7.36468 44.057C7.07646 43.3899 7.01126 42.991 7.00028 42.5291C6.97626 41.4967 7.07094 40.6539 8.53814 37.9514ZM10.5999 36.1103C10.3319 35.6927 10.1769 35.5376 10.2177 35.3269C10.2641 35.085 10.6133 34.1967 10.3707 33.2232C10.1847 32.4766 10.1298 32.609 10.5203 32.081C10.7183 31.8133 11.0631 31.2993 11.2141 31.3652C11.2786 31.3937 11.3384 31.4528 11.3803 31.5423C12.1846 33.2703 11.8023 35.05 11.4671 35.7561C11.3621 35.9774 11.185 36.3124 11.0838 36.405C10.9191 36.5557 10.7999 36.4222 10.5999 36.1103Z"></path></svg><span class="flex flex-col"><svg viewBox="25.5 19 85.5 17" aria-hidden="true" fill="currentColor" class="h-[13px] w-auto text-foreground"><path fill-rule="evenodd" clip-rule="evenodd" d="M32.3025 35.1511L26.1785 20.3673H28.1552L32.9257 32.2718L37.7822 20.3673H39.6731L33.5488 35.1511H32.3025ZM42.6491 35.0009V20.3673H44.4542V35.0009H42.6491ZM52.8882 35.323C51.1908 35.323 49.2999 34.7216 47.9676 33.0239L49.2782 31.8204C50.4387 33.1312 51.8139 33.6902 52.9956 33.6902C54.9082 33.6902 56.1116 32.5726 56.1116 31.1116C56.1116 29.4567 54.5644 28.8553 53.0172 28.3825C50.589 27.6304 48.4616 26.7707 48.4616 24.1492C48.4616 21.8283 50.4603 20.0235 53.2753 20.0235C54.8437 20.0235 56.3908 20.5393 57.5941 22.0003L56.2835 23.2468C55.3382 22.0864 54.2635 21.6782 53.2108 21.6782C51.4056 21.6782 50.2883 22.774 50.2883 24.1063C50.2883 25.6319 51.7065 26.126 53.1463 26.5988C55.6387 27.351 57.9168 28.146 57.9168 31.0038C57.9168 33.4321 56.0687 35.323 52.8882 35.323ZM61.6019 35.0009V20.3673H65.964C70.7344 20.3673 73.6136 23.0533 73.6136 27.6731C73.6136 32.2933 70.6912 35.0009 65.9424 35.0009H61.6019ZM63.4068 33.3677H65.9853C69.574 33.3677 71.6798 31.2619 71.
1026798 27.6731C71.6798 24.1063 69.6598 22.0003 65.9853 22.0003H63.4068V33.3677ZM83.896 35.3875C79.5339 35.3875 76.6328 32.1214 76.6328 27.6947C76.6328 23.2681 79.5339 19.9806 83.896 19.9806C88.2582 19.9806 91.1161 23.2681 91.1161 27.6947C91.1161 32.1214 88.2582 35.3875 83.896 35.3875ZM83.896 33.7331C87.1835 33.7331 89.1821 31.1116 89.1821 27.7163C89.1821 24.2995 87.2051 21.6353 83.896 21.6353C80.5866 21.6137 78.5669 24.2782 78.5669 27.7163C78.5669 31.1116 80.6082 33.7331 83.896 33.7331ZM94.844 35.0009V20.3673H96.9286L102.58 31.6914L108.146 20.3673H110.251V35.0009H108.425V23.5906C108.188 24.1493 107.93 24.7296 107.716 25.1592L102.688 35.194H102.408L97.3798 25.2882C97.165 24.8586 96.907 24.2783 96.6493 23.7193V35.0009H94.844Z"></path></svg><span class="whitespace-nowrap text-muted-foreground" style="font-family:var(--font-geist-mono, ui-monospace, monospace);font-size:8.5px;letter-spacing:.148em;margin-top:3px">MATURITY MATRIX</span></span></a><nav class="hidden md:flex items-center gap-0.5" aria-label="Primary"><div class="relative"><button type="button" aria-expanded="false" aria-haspopup="menu" class="inline-flex items-center gap-1 rounded-full px-3 py-2 text-sm font-medium text-muted-foreground transition-colors hover:text-foreground">Matrix<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-down h-3.5 w-3.5 transition-transform" aria-hidden="true"><path d="m6 9 6 6 6-6"></path></svg></button></div><div class="relative"><button type="button" aria-expanded="false" aria-haspopup="menu" class="inline-flex items-center gap-1 rounded-full px-3 py-2 text-sm font-medium text-muted-foreground transition-colors hover:text-foreground">Resources<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-down h-3.5 w-3.5 transition-transform" aria-hidden="true"><path d="m6 9 6 6 6-6"></path></svg></button></div><a href="https://visdom.virtuslab.com/" target="_blank" rel="noopener noreferrer" class="rounded-full px-3 py-2 text-sm font-medium text-muted-foreground transition-colors hover:text-foreground inline-flex items-center gap-1.5">Platform <span aria-hidden="true">â</span></a></nav></div><div class="flex items-center gap-1 shrink-0"><a class="hidden sm:inline-flex items-center rounded-full px-3 py-2 text-sm font-medium text-muted-foreground transition-colors hover:text-foreground" href="/workshop">Start assessment</a><div class="hidden md:block ml-1"><button type="button" class="nav-cta">Log in</button></div><button type="button" aria-label="Open menu" aria-expanded="false" aria-controls="mobile-nav" class="md:hidden inline-flex items-center justify-center rounded-full text-foreground hover:bg-accent transition-colors" style="width:36px;height:36px;border:1px solid var(--border)"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-menu w-4 h-4" aria-hidden="true"><path d="M4 5h16"></path><path d="M4 12h16"></path><path d="M4 19h16"></path></svg></button></div></header><div class="flex items-center justify-center h-32 text-muted-foreground text-sm">Loading your progress...</div></main><!--$--><!--/$-->
102<script src="/_next/static/chunks/15d2e4c42c92d9d3.js" id="_R_" async=""></script>
102<script>(self.__next_f=self.__next_f||[]).push([0])</script>
102<script>self.__next_f.push([1,"1:\"$Sreact.fragment\"\n9:I[379532,[],\"default\"]\n:HL[\"/_next/static/chunks/4eaf45b13857288d.css\",\"style\"]\n:HL[\"/_next/static/chunks/e55c23589ab78d68.css\",\"style\"]\n:HL[\"/_next/static/media/248e1dc0efc99276-s.p.8a6b2436.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/media/797e433ab948586e-s.p.29207c2f.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/media/caa3a2e1cccd8315-s.p.3b6cae6d.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n2:Te5e,"])</script>
102<script>self.__next_f.push([1,"(function(){\n if (typeof navigator === 'undefined') return;\n var endpoint = '/api/mcp/matrix';\n var tools = [];\n if (!navigator.modelContext) {\n navigator.modelContext = {\n tools: tools,\n registerTool: function(t){\n tools.push(t);\n return Promise.resolve();\n },\n provideContext: function(ctx){\n if (ctx \u0026\u0026 ctx.tools) ctx.tools.forEach(function(t){ tools.push(t); });\n return Promise.resolve();\n }\n };\n } else if (!navigator.modelContext.tools) {\n navigator.modelContext.tools = tools;\n }\n\n async function callMcp(name, args){\n var res = await fetch(endpoint, {\n method: 'POST',\n headers: { 'content-type': 'application/json' },\n body: JSON.stringify({ jsonrpc: '2.0', id: Date.now(), method: 'tools/call', params: { name: name, arguments: args || {} } })\n });\n var json = await res.json();\n if (json.error) throw new Error(json.error.message);\n var content = json.result \u0026\u0026 json.result.content;\n if (content \u0026\u0026 content[0] \u0026\u0026 content[0].text) {\n try { return JSON.parse(content[0].text); } catch (e) { return content[0].text; }\n }\n return json.result;\n }\n\n function reg(t){\n if (typeof navigator.modelContext.registerTool === 'function') {\n try { navigator.modelContext.registerTool(t); }\n catch (e) { (navigator.modelContext.tools || tools).push(t); }\n } else {\n (navigator.modelContext.tools || tools).push(t);\n }\n }\n\n reg({\n name: 'get_perspectives',\n description: 'List the four perspectives of the AI-Native SDLC Maturity Matrix with their areas.',\n inputSchema: { type: 'object', properties: {}, additionalProperties: false },\n execute: function(){ return callMcp('get_perspectives', {}); }\n });\n reg({\n name: 'get_area',\n description: 'Return one area of the matrix with all 5 levels and their bullet items.',\n inputSchema: {\n type: 'object',\n properties: {\n perspective: { type: 'string', enum: ['development','delivery','infrastructure','organization'] },\n area: { type: 'string' }\n },\n required: ['perspective','area'],\n additionalProperties: false\n },\n execute: function(args){ return callMcp('get_area', args); }\n });\n reg({\n name: 'get_gates',\n description: 'Return Must/Should criteria for a perspective+area, at one level or all 5.',\n inputSchema: {\n type: 'object',\n properties: {\n perspective: { type: 'string', enum: ['development','delivery','infrastructure','organization'] },\n area: { type: 'string' },\n level: { type: 'integer', minimum: 1, maximum: 5 }\n },\n required: ['perspective','area'],\n additionalProperties: false\n },\n execute: function(args){ return callMcp('get_gates', args); }\n });\n reg({\n name: 'search_guides',\n description: 'Substring search across guide title/slug/summary across 240 practice guides.',\n inputSchema: {\n type: 'object',\n properties: { query: { type: 'string' }, perspective: { type: 'string' }, limit: { type: 'integer' } },\n required: ['query'],\n additionalProperties: false\n },\n execute: function(args){ return callMcp('search_guides', args); }\n });\n reg({\n name: 'recommend_next_steps',\n description: 'From a scores object (perspective -\u003e array of per-area levels 1-5), return lowest-scoring areas with the L+1 gate to target.',\n inputSchema: {\n type: 'object',\n properties: { scores: { type: 'object' }, limit: { type: 'integer' } },\n required: ['scores'],\n additionalProperties: false\n },\n execute: function(args){ return callMcp('recommend_next_steps', args); }\n });\n})();"])</script>
102<script>self.__next_f.push([1,"0:{\"P\":null,\"b\":\"Vv4HRFzSbAJZU99gF7vmh\",\"c\":[\"\",\"workshop\"],\"q\":\"\",\"i\":false,\"f\":[[[\"\",{\"children\":[\"workshop\",{\"children\":[\"__PAGE__\",{}]}]},\"$undefined\",\"$undefined\",true],[[\"$\",\"$1\",\"c\",{\"children\":[[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/chunks/4eaf45b13857288d.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"link\",\"1\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/chunks/e55c23589ab78d68.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/chunks/e9aa5be042e3b0a3.js\",\"async\":true,\"nonce\":\"$undefined\"}]],[\"$\",\"html\",null,{\"lang\":\"en\",\"suppressHydrationWarning\":true,\"children\":[\"$\",\"body\",null,{\"className\":\"geist_9c6cb61b-module__8NX9hq__variable geist_mono_d6617093-module__z61v7q__variable merriweather_519133d3-module__aYUzJa__variable\",\"children\":[[\"$\",\"script\",null,{\"type\":\"application/ld+json\",\"dangerouslySetInnerHTML\":{\"__html\":\"{\\\"@context\\\":\\\"https://schema.org\\\",\\\"@graph\\\":[{\\\"@type\\\":\\\"Organization\\\",\\\"@id\\\":\\\"https://visdom-maturity-matrix.virtuslab.com#organization\\\",\\\"name\\\":\\\"VirtusLab\\\",\\\"url\\\":\\\"https://virtuslab.com\\\",\\\"logo\\\":\\\"https://visdom-maturity-matrix.virtuslab.com/og-image.png\\\",\\\"sameAs\\\":[\\\"https://www.linkedin.com/company/virtuslab\\\"]},{\\\"@type\\\":\\\"WebSite\\\",\\\"@id\\\":\\\"https://visdom-maturity-matrix.virtuslab.com#website\\\",\\\"url\\\":\\\"https://visdom-maturity-matrix.virtuslab.com\\\",\\\"name\\\":\\\"AI-Native SDLC Maturity Matrix | Visdom by VirtusLab\\\",\\\"description\\\":\\\"Most teams stall at Level 2 and conclude AI doesn't work. It works great - at Level 4. The opinionated map there: 60 practices, 4 perspectives, 5 levels, built from years of real engineering engagements.\\\",\\\"publisher\\\":{\\\"@id\\\":\\\"https://visdom-maturity-matrix.virtuslab.com#organization\\\"},\\\"inLanguage\\\":\\\"en\\\"}]}\"}}],[\"$\",\"script\",null,{\"dangerouslySetInnerHTML\":{\"__html\":\"$2\"}}],\"$L3\",\"$L4\",\"$L5\"]}]}]]}],{\"children\":[\"$L6\",{\"children\":[\"$L7\",{},null,false,false]},null,false,false]},null,false,false],\"$L8\",false]],\"m\":\"$undefined\",\"G\":[\"$9\",[]],\"S\":true}\n"])</script>
102<script>self.__next_f.push([1,"a:I[906558,[\"/_next/static/chunks/e9aa5be042e3b0a3.js\"],\"\"]\nb:I[348930,[\"/_next/static/chunks/e9aa5be042e3b0a3.js\"],\"DevBanner\"]\nc:I[485585,[\"/_next/static/chunks/e9aa5be042e3b0a3.js\"],\"SessionProvider\"]\nd:I[485739,[\"/_next/static/chunks/e9aa5be042e3b0a3.js\"],\"AccessCodeProvider\"]\ne:I[750786,[\"/_next/static/chunks/e9aa5be042e3b0a3.js\"],\"ReducedMotionProvider\"]\nf:I[430453,[\"/_next/static/chunks/8a2c28f82c6d4f4c.js\",\"/_next/static/chunks/3e8b73794b0ba35e.js\"],\"default\"]\n10:I[599434,[\"/_next/static/chunks/8a2c28f82c6d4f4c.js\",\"/_next/static/chunks/3e8b73794b0ba35e.js\"],\"default\"]\n11:I[961928,[\"/_next/static/chunks/e9aa5be042e3b0a3.js\",\"/_next/static/chunks/18b6abd70bbeface.js\",\"/_next/static/chunks/8827248f36da4bf2.js\",\"/_next/static/chunks/47f82402146413fb.js\",\"/_next/static/chunks/b7312459423597a8.js\",\"/_next/static/chunks/705056c08acab783.js\",\"/_next/static/chunks/1a514990f47add69.js\"],\"PageNav\"]\n12:I[28485,[\"/_next/static/chunks/e9aa5be042e3b0a3.js\",\"/_next/static/chunks/18b6abd70bbeface.js\",\"/_next/static/chunks/8827248f36da4bf2.js\",\"/_next/static/chunks/47f82402146413fb.js\",\"/_next/static/chunks/b7312459423597a8.js\",\"/_next/static/chunks/705056c08acab783.js\",\"/_next/static/chunks/1a514990f47add69.js\"],\"WorkshopClient\"]\na6:I[44887,[\"/_next/static/chunks/8a2c28f82c6d4f4c.js\",\"/_next/static/chunks/3e8b73794b0ba35e.js\"],\"ViewportBoundary\"]\na8:I[44887,[\"/_next/static/chunks/8a2c28f82c6d4f4c.js\",\"/_next/static/chunks/3e8b73794b0ba35e.js\"],\"MetadataBoundary\"]\na9:\"$Sreact.suspense\"\n"])</script>
102<script>self.__next_f.push([1,"3:[[\"$\",\"$La\",null,{\"id\":\"cookie-information-consent\",\"src\":\"https://policy.app.cookieinformation.com/uc.js\",\"data-culture\":\"EN\",\"data-gcm-version\":\"2.0\",\"strategy\":\"beforeInteractive\"}],[\"$\",\"$La\",null,{\"id\":\"piwik-pro-loader\",\"strategy\":\"afterInteractive\",\"children\":\"\\n(function(window, document, dataLayerName, id) {\\n window[dataLayerName]=window[dataLayerName]||[],window[dataLayerName].push({start:(new Date).getTime(),event:\\\"stg.start\\\"});var scripts=document.getElementsByTagName('script')[0],tags=document.createElement('script');\\n var qP=[];dataLayerName!==\\\"dataLayer\\\"\u0026\u0026qP.push(\\\"data_layer_name=\\\"+dataLayerName);var qPString=qP.length\u003e0?(\\\"?\\\"+qP.join(\\\"\u0026\\\")):\\\"\\\";\\n tags.async=!0,tags.src=\\\"https://virtuslab.containers.piwik.pro/\\\"+id+\\\".js\\\"+qPString,scripts.parentNode.insertBefore(tags,scripts);\\n !function(a,n,i){a[n]=a[n]||{};for(var c=0;c\u003ci.length;c++)!function(i){a[n][i]=a[n][i]||{},a[n][i].api=a[n][i].api||function(){var a=[].slice.call(arguments,0);\\\"string\\\"==typeof a[0]\u0026\u0026window[dataLayerName].push({event:n+\\\".\\\"+i+\\\":\\\"+a[0],parameters:[].slice.call(arguments,1)})}}(i[c])}(window,\\\"ppms\\\",[\\\"tm\\\",\\\"cm\\\"]);\\n})(window, document, 'dataLayer', 'c2f01b3a-b794-4a7b-a19b-4985f746aa37');\\n\"}],[[\"$\",\"$La\",null,{\"id\":\"ga4-loader\",\"src\":\"https://www.googletagmanager.com/gtag/js?id=G-P67YVZS2LZ\u0026l=gtagLayer\",\"strategy\":\"afterInteractive\"}],[\"$\",\"$La\",null,{\"id\":\"ga4-init\",\"strategy\":\"afterInteractive\",\"children\":\"\\n window.gtagLayer = window.gtagLayer || [];\\n function gtag(){ window.gtagLayer.push(arguments); }\\n window.gtag = gtag;\\n gtag('js', new Date());\\n gtag('config', 'G-P67YVZS2LZ');\\n \"}]]]\n"])</script>
102<script>self.__next_f.push([1,"4:[\"$\",\"$Lb\",null,{}]\n"])</script>
102<script>self.__next_f.push([1,"5:[\"$\",\"$Lc\",null,{\"children\":[\"$\",\"$Ld\",null,{\"children\":[\"$\",\"$Le\",null,{\"children\":[\"$\",\"$Lf\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L10\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":[[[\"$\",\"title\",null,{\"children\":\"404: This page could not be found.\"}],[\"$\",\"div\",null,{\"style\":{\"fontFamily\":\"system-ui,\\\"Segoe UI\\\",Roboto,Helvetica,Arial,sans-serif,\\\"Apple Color Emoji\\\",\\\"Segoe UI Emoji\\\"\",\"height\":\"100vh\",\"textAlign\":\"center\",\"display\":\"flex\",\"flexDirection\":\"column\",\"alignItems\":\"center\",\"justifyContent\":\"center\"},\"children\":[\"$\",\"div\",null,{\"children\":[[\"$\",\"style\",null,{\"dangerouslySetInnerHTML\":{\"__html\":\"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}\"}}],[\"$\",\"h1\",null,{\"className\":\"next-error-h1\",\"style\":{\"display\":\"inline-block\",\"margin\":\"0 20px 0 0\",\"padding\":\"0 23px 0 0\",\"fontSize\":24,\"fontWeight\":500,\"verticalAlign\":\"top\",\"lineHeight\":\"49px\"},\"children\":404}],[\"$\",\"div\",null,{\"style\":{\"display\":\"inline-block\"},\"children\":[\"$\",\"h2\",null,{\"style\":{\"fontSize\":14,\"fontWeight\":400,\"lineHeight\":\"49px\",\"margin\":0},\"children\":\"This page could not be found.\"}]}]]}]}]],[]],\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]}]}]}]\n"])</script>
102<script>self.__next_f.push([1,"6:[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$Lf\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L10\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]\n13:T42a,Bob's team has been using autocomplete and chat for three months. A few developers have started using Cursor Composer, and Bob is hearing reports of \"I implemented an entire API endpoint in 20 minutes.\" He's excited but also nervous - what if the agent makes changes nobody reviews properly?\n\n**What Bob should do:** Bob should institutionalize the review gate before celebrating the speed, and he should treat the permission ruleset as a team artefact rather than a personal preference. His team policy for L2 should be: one reviewed `permissions.deny` and `permissions.ask` set committed to the repo, always on a branch, always a PR with diff review before merge, always run the test suite after agent changes. These rules preserve quality while unlocking speed. Bob should also establish agent office hours - a weekly 30-minute session where developers share what worked, what went wrong, and what they added to CLAUDE.md or to the deny rules as a result. This turns individual experiences into collective learning, which is exactly what moves the team toward L3.14:T46a,Victor has been running agents on whole tasks since day one and is already pushing it to its limits. He runs agents on significant refactoring tasks, architecture cleanups, and test suite expansions. His CLAUDE.md has grown to 400 lines of conventions and constraints. His productivity is measurably higher than anyone else on the team.\n\n**What Victor should do:** Victor's 400-line CLAUDE.md is a goldmine - it represents months of learned agent behavior encoded as rules. He should split it into a root-level CLAUDE.md (project-wide conventions) and per-directory CLAUDE.md files (module-specific patterns), and separate the parts that are advice to the agent from the parts that are boundaries, which belong in `permissions.deny` and `permissions.ask` where they are enforced rather than read. That separation is the artefact the rest of the team can adopt. Victor should also start experimenting with running agents on two tasks in parallel using separate git worktrees - that's the preview of L4 multi-agent workflows. His next leverage point is showing Bob the throughput numbers to build the case for team-wide L3 adoption.15:T456,Bob is frustrated that AI-generated code consistently fails code review on the same issues. Reviewers keep leaving comments like \"we don't use this pattern,\" \"this library is deprecated,\" and \"this should use the centralized error handler.\" The same AI mistakes repeat across developers and across weeks.\n\n**What Bob should do:** These repeating review comments are a diagnostic: they are exactly the implicit conventions that need stating. Bob should create a \"convention backlog\" - every repeating misfire observed in the last month of code review - and task one developer per team with converting the list into explicit rules in the team's instruction file. That is a two to three day effort per team and should eliminate the repeat comments within a fortnight. Bob should also watch for the subset of the backlog that documentation cannot fix: misfires where the agent read the code correctly and the code was misleading. Those are a signal to change the code rather than to write another rule, and a team that only ever adds rules will accumulate an instruction file that grows faster than its usefulness.16:T419,Sarah has been tracking code review cycle time and has noticed that PRs with significant AI involvement take longer to merge than expected - not because the code is functionally wrong, but because it consistently violates style and pattern conventions. Reviewers spend time on style corrections that feel avoidable.\n\n**What Sarah should do:** This is the explicitness gap in measurable form. Sarah should track \"AI convention"])</script>
102<script>self.__next_f.push([1," violation rate\" - the percentage of AI-involved PRs that receive at least one convention-related review comment. This metric directly measures how explicit your CLAUDE.md is. Set a target: reduce AI convention violation rate from current baseline to under 10% within 90 days by systematically making conventions explicit. Sarah can show this to stakeholders as \"context engineering quality improvement\" - a leading indicator that predicts faster review cycles and lower rework cost. The financial story: every convention violation caught in review represents time that could be eliminated by better CLAUDE.md documentation.17:T415,Victor has been doing this informally for months. When he notices the AI making the same mistake twice, he immediately adds an explicit rule to CLAUDE.md. His team's AI misfire rate is dramatically lower than other teams, and his code reviews are mostly about logic rather than style.\n\n**What Victor should do:** Victor should systematize his informal practice into a team process. His proposal: when any developer on the team gets an AI convention misfire in code review, they are responsible for adding an explicit rule to the team's CLAUDE.md before the PR merges. This creates a direct feedback loop: code review surfaces implicit conventions, convention violations trigger explicit documentation, explicit documentation prevents future misfires. Victor should also write a guide for the team on \"how to write a good convention rule\" - the four-part structure (rule, rationale, positive example, negative example) with examples from the team's actual CLAUDE.md. This institutionalizes his practice without depending on his individual effort.18:T45b,Bob's team has been delivering impressive results with L4 parallel agents, but he's seeing a pattern: the tasks that take the most effort are the ones that require coordination across multiple areas of the codebase. These are exactly the tasks that would benefit from orchestration, but Bob doesn't know if the team is ready for the L5 investment.\n\n**What Bob should do:** L5 orchestration is a significant infrastructure investment - likely a dedicated 2-4 week engineering project before it produces business value. Bob should only initiate this if the L4 foundation is solid: mature CLAUDE.md files, reliable agent sandboxes, established review protocols, and demonstrated parallel agent workflows. The signal for L5 readiness is when developers say \"the bottleneck is no longer implementation - it's the coordination overhead of managing multiple agents.\" When that statement is consistently true, orchestration automation is the right investment. Bob should frame it as infrastructure engineering: assign Victor to design and build the orchestration layer, with a 6-week timeline and specific success criteria.19:T407,Victor has been informally acting as the \"planner agent\" himself - decomposing complex tasks into subtasks, assigning them to different Claude Code instances, and synthesizing the results. He knows this manual orchestration is the bottleneck and wants to automate it.\n\n**What Victor should do:** Victor should document his manual orchestration process first, before automating it. For the last 10 complex tasks he coordinated: how did he decompose them? What criteria determined subtask assignments? How did he handle failures? What information did he pass to each agent? This documentation is the specification for the planner agent he'll eventually build. Victor should then propose a concrete L5 project: build a planner agent that can decompose a specific category of tasks (e.g., \"implement REST endpoint from spec\") into the same subtask structure he uses manually. Automate one task type end-to-end before generalizing. The first successful end-to-end orchestration is the proof of concept that justifies broader investment.1a:T418,Bob's team has successfully deployed CLAUDE.md files across most repositories (L2). Developers are reporting good results, but senior engineers are raising a new problem: the context in CLAUDE.md files keeps going stale. The service ownership sectio"])</script>
102<script>self.__next_f.push([1,"n was accurate six months ago; since then, three service owners have changed and two services were deprecated. Manually updating CLAUDE.md isn't keeping up with organizational change.\n\n**What Bob should do:** Bob has identified the natural L2-to-L3 transition trigger: static context that changes faster than humans can update it needs to become dynamic context from an MCP server. Bob should sponsor a proof of concept: one MCP server exposing service ownership from the existing service registry. This is typically low-effort to build (the data already exists; the MCP server just makes it queryable). Bob should track whether agents' suggestions about service interactions become more accurate after the MCP server is available, and use that as the business case for expanding MCP infrastructure.1b:T451,Victor has been manually building context packages for his agent sessions for months. He has a personal checklist: check the sprint board for related work, pull the current service ownership from the registry, look up recent ADRs for the relevant area, check deployment status. He does this before every session and it works well - but it takes 20 minutes each time and he never misses a step because he built the checklist.\n\n**What Victor should do:** Victor's checklist is the specification for a BYOC pipeline. He should spend two days building an automated version: a script that takes a Jira ticket ID, queries the relevant sources (sprint board, service registry, ADR repository, deployment dashboard), assembles the context into a structured prompt prefix, and outputs it ready to inject into an agent session. Once it works for his own workflow, he should propose it as shared infrastructure. The test: does his automated context package produce the same quality results as his manual 20-minute preparation? If yes, the ROI is his 20 minutes per session, multiplied by every developer on the team.1c:T402,Bob's team has built up a solid L4 context engineering infrastructure. Agents are performing well on individual tasks, but Bob is hearing a persistent complaint: \"Every agent session feels like the agent is starting from scratch.\" Developers spend time in each session re-explaining things they know the agent would have encountered before - codebase quirks, known gotchas, previous decisions. There's no continuity.\n\n**What Bob should do:** Bob should frame persistent memory as an investment in agent \"experience.\" Just as experienced human developers are more valuable because they've learned the codebase, experienced agents should be more valuable because their memory accumulates codebase knowledge. Bob should pilot the Beads memory system on one team's primary repository: configure agents to write memory records, build a review process, and measure whether suggestion quality improves over 6-8 weeks of use. The improvement should be measurable in reduced iteration counts and fewer constraint-violation corrections.1d:T400,Sarah tracks ITS (Iterations-to-Success) for agent workflows. She notices that ITS doesn't improve much over time - even teams that have been using AI agents for a year don't show significantly better per-task performance than they did in the first month. Agents don't seem to get better with use. She's been told this is expected (agents are stateless), but suspects there's a way to change it.\n\n**What Sarah should do:** Sarah should frame persistent agent memory as an ITS improvement initiative. The hypothesis: agents with access to accumulated codebase memory will have lower ITS than stateless agents working on the same codebase. She should fund a six-week pilot with memory-enabled agents on one team, tracking ITS before and after memory accumulation begins. If the hypothesis holds (ITS decreases as memory grows), she has evidence that memory investment improves agent ROI over time - and that the improvement compounds, which is a fundamentally different ROI story than \"agents are useful for individual tasks.\"1e:T41a,Bob's team has had three incidents in the past six months caused by direct database access in HTTP handlers - "])</script>
102<script>self.__next_f.push([1,"each resulted in N+1 queries under load that crashed the service. After each incident, they added a note to the architecture documentation. The notes are there, but developers keep making the same mistake because no one reads the architecture docs before writing code.\n\n**What Bob should do:** Bob should commission Victor (or a senior engineer with taste for tooling) to write a custom ESLint (or language-appropriate) rule: \"no direct database access in HTTP handler files.\" The rule takes half a day to write and test. Once deployed, it will catch every future violation at CI time, before it reaches production. Bob should present this to his team as a category of incident that has been permanently eliminated. He should also establish a process: after every incident, ask \"could a lint rule have prevented this?\" If yes, write the rule. This is the Bug â Codify â Lint Rule process that makes quality gates progressively stronger.1f:T441,Bob's team has adopted the AI review agent and linting, and is seeing good results. Human review time has dropped. But he's started wondering if some PRs could be merged without any human review at all - the AI consistently rates them as clean, the tests pass, and the human reviewer always approves them with no additional comments. He wants a way to identify and automate these PRs.\n\n**What Bob should do:** Bob is ready to implement the traffic-light evaluation. His first step: ask Victor to analyze the last 3 months of PRs and identify what proportion would have met a conservative Green definition (tests pass, lint clean, AI review no issues, \u003c300 lines, no high-risk files). If that proportion is above 20%, the system is worth implementing. Bob should establish the evaluation criteria with his tech leads, start with Green being very strict, and run it in \"observation mode\" for 30 days - compute the color for every PR, post it as a comment, but don't actually auto-merge. Then review the Green PRs: were any of them problematic? If not, he's ready to enable actual auto-merge.20:T43e,Bob is managing a fleet of Claude Code agents working on different parts of the codebase simultaneously. He's noticed that some agents produce clean Green PRs while others consistently produce Red PRs or PRs that require many human corrections after auto-merge. He wants to understand why the fleet's performance is uneven.\n\n**What Bob should do:** The variance Bob is seeing reflects differences in task type and quality gate calibration, not random variation. Bob should analyze the non-convergent agent tasks: what do they have in common? If they're consistently in one area of the codebase (say, the payment module), the problem is likely insufficient test coverage or unclear lint rules in that area. If they're consistent across task types (the agent always struggles with database migrations), the problem is likely a gap in the agent's context or a missing acceptance criteria template. Bob should direct Victor to improve the quality gate for the problem areas - better tests, clearer lint rules, more specific task templates - rather than trying to tune the agents themselves.21:T492,Victor is spending time each week reviewing the non-convergent agent tasks - the Red PRs that escalated to him because the agent couldn't converge. He's noticed that 40% of them failed because of a pattern the agents kept getting wrong: they were calling an external service directly in a unit test context instead of using the mock framework. The agents would iterate, fail the test, and not understand why.\n\n**What Victor should do:** Victor has identified a systemic gap in the agent context. The agents don't know about the mock framework convention because it's not in the CLAUDE.md or any instruction file the agents use. Victor should add a specific instruction: \"When writing tests that involve external service calls, use the `MockServiceFactory` rather than calling services directly. See `/tests/mocks/README.md` for patterns.\" He should then verify that this instruction resolves the convergence failures in the next batch of similar tasks. Vict"])</script>
102<script>self.__next_f.push([1,"or is doing exactly what an L5 staff engineer should be doing: identifying the systemic gap, fixing it once in the agent configuration, and watching the fleet's performance improve across all future similar tasks.22:T43c,Bob's team has 15 developers and just started rolling out AI coding tools. Three weeks in, he notices that agent adoption has stalled - developers are using the tools for chat and autocomplete but not for autonomous agent tasks. When he digs in, the answer is clear: no one wants to wait 18 minutes to see if the agent's code passes CI. The feedback loop is too slow to make agent iteration feel worthwhile.\n\nBob needs to treat CI speed as a prerequisite for AI tool ROI, not a nice-to-have. He should make \"CI under 10 minutes within 30 days\" an explicit team goal, assign one developer to own the CI pipeline for the sprint, and track progress publicly. The developer assigned to CI optimization will find that dependency caching and basic parallelization of the test suite can cut 8-12 minutes off the current pipeline within a week. Bob should frame this as infrastructure investment that pays for itself: if 15 developers are each waiting 18 minutes for CI 5 times per day, that's 22.5 developer-hours per day lost to waiting. Cutting that to 5 minutes recovers 16 hours per day.23:T438,Sarah tracks developer experience metrics and has noticed that \"CI wait time\" consistently appears in developer satisfaction surveys as a friction point. Now that AI agents are part of the workflow, the problem is acute: agents need to iterate quickly, but 18-minute CI turns every iteration cycle into a multi-hour affair. Her DORA metrics show lead time for changes is dominated by CI time, not code review time.\n\nSarah should create a CI feedback latency dashboard that tracks p50 and p95 CI run time per day, per team, and per pipeline stage. Making the data visible is step one. Step two is setting a team target (10 minutes within 30 days) and reporting progress weekly in the engineering all-hands. Sarah should also interview developers who are getting the most value from AI agents to understand how they handle the CI wait - most high performers have already developed workarounds (running a subset of tests locally, batching multiple changes before pushing). Systematizing those workarounds into pipeline infrastructure is the path from workaround to standard practice.24:T41e,Bob's team had a production incident last month where a deploy job silently stopped running its smoke tests. Nobody could establish when that changed or who changed it, because the pipeline is configured in the CI provider's web console and the console does not keep meaningful history. The post-incident review produced an action item Bob has not yet funded.\n\nBob should treat this as a governance problem rather than a tooling preference, because that is how it will be received by anyone asking why the incident happened. The ask is small: one engineer, roughly a sprint, to move the deploy pipelines for the top three services into their repositories and put them behind CODEOWNERS review. The outcome Bob can report is concrete - every future change to how software reaches production arrives as a reviewed diff with an author and a date. Bob should also close off the old path explicitly, revoking console edit rights once the files are authoritative, because a migration that leaves the back door open will quietly reverse itself within a quarter.25:T458,Sarah keeps hearing that CI is unpredictable, but the complaints do not resolve into anything she can act on: a job that passed yesterday fails today, a step that used to run does not, a build behaves differently on one repository than on its neighbour. She suspects the variability is not in the code.\n\nSarah should measure how much of the delivery process is invisible to version control. For each active repository she can record whether the pipeline is defined in a committed file, partly committed, or entirely console-configured - a half-day of clicking, and the resulting table is the argument. Repositories with committed pip"])</script>
102<script>self.__next_f.push([1,"elines will show a clean correspondence between pipeline changes and commits; the others will show unexplained behaviour changes. That contrast is what turns \"CI feels flaky\" into a specific, fundable piece of work. Sarah should also start tracking pipeline changes as a review category in their own right, so the team can see that these edits are happening at all - in most organisations moving to pipeline-as-code reveals a change rate nobody had realised was going unreviewed.26:T4a1,Victor has been letting agents propose changes across several services and has run into a consistent wall: an agent can write the code and the tests, but when the change needs a new CI job it stops, because the pipeline is not something it can see or edit. Victor ends up hand-translating the agent's suggestion into the console himself.\n\nVictor should make pipeline definitions part of the agent's working surface. That means committing the workflow files, then making sure the repository's agent instructions describe where they live, what the shared reusable workflows are, and which changes require a platform review. He should also add the pipeline linter to the pull request checks, because an agent proposing workflow edits will get the syntax wrong occasionally and the fast, mechanical feedback is what lets it self-correct without a human round trip. Victor should keep one guardrail deliberately human: changes that grant a job new credentials or add a new trigger stay under CODEOWNERS review regardless of who authored them. The goal is agents proposing pipeline changes freely and a human approving the privileged subset, not agents locked out of a third of the codebase.27:T502,Victor has implemented Nx's `affected` command on his team's JavaScript monorepo and the results are clear: p50 CI time dropped from 7 minutes to 2 minutes after the change, because most agent-generated commits touch 1-2 packages in a 10-package repo. He's using Turborepo remote caching with Vercel for the remote artifact store. His agents now iterate at 30 cycles per hour instead of 8.\n\nVictor should write up the configuration as a reference implementation: the `turbo.json` or `nx.json` configuration, the CI YAML changes, and the Vercel remote cache setup. He should include the before-and-after timing data and the agent iteration rate improvement. This reference implementation is what other JavaScript teams in the organization need to replicate the result. Victor should also investigate whether the pattern extends to the organization's Python services - Pants has equivalent affected-package detection and is worth evaluating for those teams. Before he publishes the reference implementation, he should run one grep across every workflow file for `${{ github.event.* }}` inside `run:` blocks and split the agent pass into its own job with its own token scope. A reference implementation that other teams copy is exactly the wrong place to standardise an injection path.28:T463,Bob has 10 developers each running agents, and they've started complaining about CI slowness during afternoon peak hours. Investigation reveals the problem: two developers are running agents that are iterating rapidly on failing tests, generating 30-40 CI jobs per hour each on the shared runner pool. The 8 other developers are waiting for their CI jobs to clear the queue caused by the agents.\n\nBob needs to implement the sandbox isolation before the organizational friction gets worse. His action: provision a dedicated agent sandbox runner pool (4 autoscaling runners), create an `agent-sandbox.yml` workflow that runs a 30-second subset of checks, and update the team's agent usage convention - agents use the sandbox workflow for iteration, the standard workflow only for final PR creation. Bob should implement this in a day and communicate the change to the team: \"agents now have their own CI infrastructure; your CI queue times will return to normal.\" The rapid response to the shared queue problem demonstrates that engineering leadership is actively managing the infrastructure implications of AI tool adoption.29:"])</script>
102<script>self.__next_f.push([1,"T4e1,Sarah has data showing that the two developers using agents most heavily are generating 80% of the team's CI load. She also has data showing that their agent sessions have an average of 28 CI runs per successful task completion - they iterate heavily before converging. She wants to understand: is 28 runs per task a sign of inefficiency, or is it the expected iteration rate for the complexity of tasks they're tackling?\n\nSarah should interview the two heavy agent users and review their agent session logs. If the 28-run average is driven by agents correcting type errors and lint violations through iteration (avoidable with better initial context), that's a prompt engineering problem. If it's driven by agents working through complex behavioral logic (expected iteration), it's a load planning problem. The distinction determines the intervention: improve agent instructions to reduce unnecessary iteration, or provision more sandbox capacity to support the expected iteration rate. Sarah should use this analysis to set a \"reasonable iteration target\" per task complexity category - a simple bug fix should converge in 5-10 iterations, a complex feature in 20-30 - and use deviations from this target as a signal for agent quality improvement.2a:T418,Victor already uses a custom bash script that connects his local Claude Code session to a GitHub Actions sandbox workflow. The script: (1) stages the current changes, (2) triggers the sandbox workflow via `gh workflow run`, (3) polls for completion every 10 seconds, (4) prints the result. His iteration loop is about 25 seconds end-to-end. He can attempt 2-3 iterations per minute while working on a problem.\n\nVictor should formalize his script into a proper CI-as-sandbox integration: a Claude Code MCP tool that agents can invoke natively, triggering a sandbox CI run and returning the results in the agent's context window. This transforms the ad-hoc script into a first-class agent capability: agents can decide when to trigger a sandbox run, receive results, and continue iteration without human involvement. Victor should also document the runner setup and the workflow configuration so other developers can replicate his setup. The MCP tool plus the runner setup is the complete CI-as-sandbox implementation that other developers can adopt.2b:T4fe,Bob's team is operating at L4 with 2-minute CI and running 30-40 agent-assisted PRs per day. The productivity gains are real and visible. But he's seeing a ceiling: agents still require human check-ins during long sessions because they can't iterate fast enough to complete complex tasks autonomously. The agents complete the easy parts quickly but stall on the parts that require 15-20 iterations to get right. Bob wants to reach the point where agents can complete full features autonomously, not just the easy subtasks.\n\nBob should recognize that sub-minute CI is the infrastructure prerequisite for autonomous agent operation and frame the investment accordingly. He needs to present it as: \"our agents are currently capable of autonomous operation, but our CI infrastructure requires human supervision because the feedback loop is too slow for agents to close it themselves.\" The investment - pre-warmed runner pools and distributed build execution - is an infrastructure cost, not a tooling cost. Bob should work with the platform engineering team to scope the work, establish a timeline, and define the SLO (95% of agent-triggered CI runs complete in under 60 seconds). Autonomous agent operation is a product outcome; sub-minute CI is the infrastructure that enables it.2c:T454,Sarah has been tracking agent iteration rate across the team and has clean data: at 5-minute CI, agents average 8 iterations per hour; at 2-minute CI after last quarter's infrastructure work, they average 22 iterations per hour. The relationship is clear and linear. She can model what sub-minute CI would unlock: 60+ iterations per hour, and the resulting completion rate for complex agent tasks.\n\nSarah should use this model to make a concrete prediction: with sub-minute CI, what percentag"])</script>
102<script>self.__next_f.push([1,"e of complex agent tasks (currently requiring human intervention to push through the 10-20 iteration phase) would complete autonomously? She should survey current agent users: how many times per session do they need to intervene to \"unstick\" an agent that's iterating on CI failures? That intervention rate is the cost of slow CI. Sub-minute CI eliminates most of those interventions, and Sarah can estimate the hours recovered per developer per week. A clear before-and-after prediction, validated after implementation, is how Sarah demonstrates the infrastructure ROI and makes the case for the next L5 investment.2d:T424,Victor is already running sub-minute feedback for his personal development workflow using a custom local pre-push hook that runs a minimal test set in 20 seconds. He's proven the concept works. But the local hook is a personal tool - it doesn't help the team or give agents the same benefit in CI.\n\nVictor should design the production sub-minute CI system starting from his local proof of concept. The local hook already has the test selection logic (which tests to run for which changes). Porting that logic to a CI step is the core work. Victor should then scope the runner infrastructure: how many pre-warmed containers are needed for the team's current push rate? What does the cluster cost per month? What's the RBE system that fits the team's build setup (EngFlow for Bazel, BuildBuddy as an alternative)? Victor should produce a technical design document, not a blog post: concrete infrastru
102cture choices, cost estimate, implementation timeline, and success metrics (the SLO). That document is the basis for an engineering project, not a research spike.2e:T4c6,Victor has noticed something odder than a slow review queue: changes he approved days ago are still open. Nobody is arguing about them. The author moved on to something else, and the merge click never happened. When several of them are eventually merged in a burst, main breaks, because no two of them were ever tested together.\n\n**What Victor should do:** Victor should measure the gap nobody is looking at: the time between a change being approved and the same change being merged. A week of data is enough, and on most teams the number is embarrassing and completely invisible in the existing dashboards. He should pair it with a count of how many merges happened without a rebase onto current main, because that is the mechanism behind the breakages. Together the two numbers make a specific proposal rather than a general complaint: require branches to be current before the merge button unlocks, which is a branch protection setting rather than a project. Victor should frame it as protecting main rather than as automation, since that is the version of the argument nobody objects to, and it establishes the principle - the conditions for merging belong in the tooling - that everything at the next level builds on.2f:T45c,Bob's team has a CD pipeline that deploys on merge but has no intermediate gates. Production incidents typically go undetected for 5-15 minutes after deploy (when users start reporting errors) rather than being caught by automated checks. The last three incidents could have been caught by a post-deploy error rate check within 2 minutes. Bob wants to improve the safety of deploys without slowing them down significantly.\n\n**What Bob should do:** Bob should invest in post-deploy observation windows as the first gate. This doesn't slow deploys - it runs in parallel with production receiving traffic. The implementation is: after deploy, monitor error rate and latency for 5 minutes. If either exceeds a threshold, alert the on-call and pause subsequent deploys. This catches the majority of bad deploys within minutes without adding gate latency. Bob should also commission a staging smoke test suite for the five most-critical user journeys - a one-sprint investment that prevents the most common categories of production incident. Together, these make the CD pipeline defensible before moving toward automation.30:T438,Sarah has observed that the team's \"mean time to d"])</script>
102<script>self.__next_f.push([1,"etect\" production issues is 12 minutes after deploy - users report problems before the team knows there's a problem. This creates reactive, high-stress incident response. She wants to reduce detection time to under 2 minutes through automated post-deploy checks, but the infrastructure team is concerned about alerting fatigue.\n\n**What Sarah should do:** Sarah should propose a pilot: instrument the post-deploy health check for one service with clear alert criteria (error rate 2x baseline, p99 latency 50% above baseline). Run it for 30 days and measure: how often does it fire? How many of those are real incidents? How quickly is the on-call notified? If the alert quality is high (few false positives, catches real incidents), expand to all services. The concern about alerting fatigue is valid - the fix is well-calibrated thresholds, not no thresholds. Sarah should own the threshold calibration process, since it's ultimately a developer experience problem (noisy alerts are worse than no alerts) that falls in her domain.31:T509,Victor has already implemented a sophisticated CD pipeline for his team's main service using ArgoCD with automated staging promotion and manual production approval. He wants to move the production gate from manual to automated but can't get sign-off without a defined criteria set.\n\n**What Victor should do:** Victor should analyze the last 20 manual production approvals: in how many cases did the approver actually look at the staging metrics before approving? What specific data did they use? What would have to be true for them to reject an approval? This analysis typically reveals that manual approvers are rubber-stamping when certain automated criteria are met. Victor should translate those criteria into an automated gate: if staging smoke tests pass, error rate delta \u003c 5%, and latency delta \u003c 10%, auto-promote to production. Present this to stakeholders as \"we're automating the easy approvals; anything outside these thresholds still requires human judgment.\" This framing preserves human oversight while automating the mechanical majority. Victor should also move his own agents off his personal credentials first: a bot identity per agent makes the automated gate's criteria enforceable for agent changes specifically, and makes him the example the rest of the team copies.32:T60c,"])</script>
102<script>self.__next_f.push([1,"Victor has been writing Mergify configurations for his own repositories and knows the tool deeply. He wants to implement a tiered auto-merge policy: agent-generated PRs that touch only test files can auto-merge; agent-generated PRs touching source code require one human approval; agent-generated PRs touching infrastructure require two approvals. This would let his parallel agent setup ship more work without manual per-PR approval.\n\n**What Victor should do:** Victor should implement the tiered auto-merge policy as a working prototype on a non-critical service, run it for 30 days, and measure: how many PRs auto-merged? How many auto-merged PRs caused issues? How much review time was saved? The data from the pilot is the proposal to the team. Victor should also add the branch-stacking instruction to the agent configuration before the pilot starts, so the tiered policy has something to discriminate on, and he should watch the PR size distribution for the split-to-game pattern Zalando reported. Victor should document the Mergify configuration as a template: other teams can adopt the same tiered policy for their agent workflows by copying the config and adjusting the path patterns. Making the pattern reusable is how a single prototype becomes a team-wide standard. Before the pilot auto-merges anything, Victor should also move his parallel agents to per-run GitHub App tokens and remove every deploy secret from their environments; tiered auto-merge is only as safe as the guarantee that agents cannot reach production any other way."])</script>
102<script>self.__next_f.push([1,"33:T5a1,"])</script>
102<script>self.__next_f.push([1,"Bob's team has the technical prerequisites for auto-merge (merge queue, policy rules, CD pipeline with gates) but the CTO is concerned about \"removing humans from the deploy loop.\" Bob needs to make the case that auto-merge â auto-deploy with good policy and automated rollback is safer than the current manual process, not riskier.\n\n**What Bob should do:** Bob should build the comparison with data. Manual deploy process: average time to detect production issue = 15 minutes (users report it), rollback time = 20-30 minutes. Proposed auto-deploy with health checks: average time to detect = 2 minutes (automated checks), rollback time = 3 minutes (automated rollback). Auto-deploy with automated health checks is objectively faster at detecting and recovering from failures. Bob should also calculate how many \"human-caused deploy errors\" occurred in the last year (wrong version deployed, deploy at wrong time, incomplete deployment steps). These are eliminated by automation. Present the comparison as: current process has human error rate X and detection time Y; proposed automation has near-zero human error and detection time Z. Bob should also show the CTO exactly where humans stay: not in front of each deploy, but in front of every token creation, production credential change and webhook edit, with a fresh re-authentication each time. That answers \"removing humans from the loop\" with a precise list rather than a reassurance."])</script>
102<script>self.__next_f.push([1,"34:T40e,Sarah has been measuring the gap between \"PR merged\" and \"code in production\" and finds it averages 4 hours on her team. Most of that gap is a human waiting to execute the deploy. For agent-produced code, this 4-hour gap is particularly problematic because agents can't observe the production feedback they need to validate their work.\n\n**What Sarah should do:** Sarah should instrument the merge-to-production gap as a primary metric alongside PR cycle time. The goal is to make this gap approach zero: merge to production in under 10 minutes. She should calculate the business value of this improvement: if each of the 30 PRs per week that have a 4-hour deploy gap represents a feature or fix that users can't access for 4 hours, the aggregate delay is 120 feature-hours per week. That's a concrete number that makes the case for auto-deploy investment. Sarah should also partner with Victor to design the health check criteria that make auto-deploy safe, since those criteria need to be calibrated against real user experience metrics.35:T46e,Bob leads a 40-person engineering organization where some teams have informal DORA tracking and most don't. He's invested in GitHub Copilot licenses and is getting pressure from leadership to show impact. When he asks his engineering managers what the team's lead time is, he gets different answers from every manager - because they're all measuring different things.\n\n**What Bob should do:** Bob needs to standardize before he can measure impact. The first step is a cross-team alignment on definitions: what counts as a deployment, how is lead time calculated, what's the definition of a change failure. This is a 2-hour workshop, not a month-long project. After alignment, Bob should designate one engineer or engineering manager as responsible for DORA instrumentation - not a committee, a single owner. The goal for Q1 is a single dashboard that shows deployment frequency for all teams with a consistent definition. In Q2, Bob adds lead time. By Q3, he has a baseline he can use to measure the AI investment impact. The ROI conversation with leadership becomes: \"Here's where we were before AI tooling, here's where we are now.\"36:T6eb,"])</script>
102<script>self.__next_f.push([1,"Bob approved an expansion of the agent program last quarter and is now seeing unexpectedly high cloud bills. The AI API costs are three times what he projected. He doesn't know which agents are consuming the budget or why.\n\n**What Bob should do:** Bob needs CPI instrumentation immediately. He should ask his platform engineer to add token logging to the agent orchestration layer and connect it to the billing data from the cloud provider. Within a week, he should have a report showing: which agent configurations are most expensive, what the per-PR cost distribution looks like, and which task types are producing the highest costs. The analysis will almost certainly reveal that a small number of high-ITS, high-context tasks are consuming a disproportionate share of the budget. Bob should put a temporary cap on context window size for new agent tasks (this is a single configuration change) and measure the impact over two weeks. The combination of CPI instrumentation and context window capping typically reduces AI API costs by 30-50% without significant quality impact. Bob should also brace for the second half of the bill: the C++ study's 5-8% production compute increase does not appear on the AI API line at all, and Uber exhausting its annual AI budget was the most-repeated enterprise cost story of the month for a reason. The follow-up is the useful part: Uber then held spend flat from April while agent requests grew 9.4x, using a context cap, cheap models for subagents and token-saving tool access. Bob should adopt the same three levers as default policy rather than leaving model choice to each developer, because Anthropic's data shows sessions get longer as tokens get cheaper, and a cheaper price list alone will not bring his bill back to plan."])</script>
102<script>self.__next_f.push([1,"37:T53c,Sarah is building the quarterly AI productivity report and wants to include unit economics alongside throughput metrics. She wants to show: \"Here is the cost of producing each agent PR and here is the value it delivers.\"\n\n**What Sarah should do:** Sarah should build a simple cost-vs-value model. Cost side: CPI * ITS = cost per PR. Value side: estimated time saved per PR (based on the type of task - a test-writing PR saves ~30 minutes of developer time, a bug fix PR saves ~90 minutes). The ratio of value to cost is the agent ROI per PR type. Sarah should add a third cost line to the model alongside tokens and CI: the runtime compute the merged code consumes, benchmarked against the 5-8% band the C++ study measured, even if her first version of it is an estimate rather than an instrumented number. A cost-versus-value model that omits the only cost component that recurs will systematically over-rate the cheapest-looking task types. Sarah should present this model with ranges and uncertainty estimates rather than false precision. The goal isn't an exact ROI number - it's a framework that the team can use to make decisions about which task types to prioritize for agent automation. High-value, low-cost tasks (test writing) should be automated first. High-cost, low-value tasks (simple boilerplate) should be the last priority.38:T6ab,"])</script>
102<script>self.__next_f.push([1,"Victor tracks CPI for all his agent workflows and has achieved sub-$0.30 CPI through a combination of model tiering (Haiku for simple tasks, Sonnet for complex tasks, Opus reserved for architectural reasoning), optimized context windows (only the most relevant files included, not the whole codebase), and a fast CI pipeline (2-minute test runs via incremental test selection).\n\n**What Victor should do:** Victor should publish his model tiering configuration and context window strategy as a platform template. The specific configuration choices - which model for which task type, how to determine context window contents, how to structure the agent's working directory to avoid loading unnecessary files - are the optimizations that took Victor months to develop. Packaging them as a template that other developers can adopt with minimal modification is the highest-leverage contribution Victor can make. Victor should also set up a monthly CPI review where he helps other teams analyze their own CPI data and identify their highest-cost outliers. The combination of a good template and ongoing coaching is how the team moves from \"Victor's CPI is 0.30\" to \"the team's median CPI is 0.45.\" Victor should also test the newest tier: decision models such as TypeSafe's Jev ($0.042 per 1M input tokens, output free; vendor-reported speed and cost claims not yet independently reproduced) return typed choices, scores and booleans rather than prose, and can take over routing, \"compact now?\" and \"is this PR risky?\" calls that currently burn frontier tokens. The three-tier template - a decision model decides, a cheap model executes, the frontier plans - is worth measuring on CPI before it is worth adopting."])</script>
102<script>self.__next_f.push([1,"39:T421,Victor has achieved 98% TORS for his agent workflows by systematically eliminating flaky tests over the past year. He considers TORS the most important infrastructure metric for agent-driven development. He's seen firsthand how a bad test suite makes agents effectively useless.\n\n**What Victor should do:** Victor should build the team's flaky test detection and quarantine tooling as a platform contribution. The tooling should: automatically identify flaky tests from CI retry data, add them to a quarantine dashboard, create GitHub issues for each quarantined test with the flaky failure data, and send a weekly digest to the responsible team. Victor should also write up the playbook for fixing the most common types of flakiness in the team's tech stack: how to fix database isolation issues in tests, how to replace sleeps with proper async waits, how to mock external service calls reliably. The playbook turns TORS improvement from an expert task (only Victor knows how to fix these tests) into a systematized process that any developer can execute.3a:T614,"])</script>
102<script>self.__next_f.push([1,"Bob is presenting the AI program's value to the board. He has detailed cost-per-PR data, throughput metrics, and CI efficiency numbers. But the board is asking: \"What does it cost to ship a feature now compared to two years ago?\" Bob doesn't have a direct answer because he's never tracked at the feature level.\n\n**What Bob should do:** Bob should build a retrospective cost-per-feature estimate for three representative features: one from two years ago (before AI tooling), one from one year ago (early AI adoption), and one from the current quarter (mature L4 workflows). For each, he should estimate the total cost using the available data: engineering hours (from Jira tickets and calendar records), PR counts (from git history), and today's AI/CI costs for the current feature. The trend line across the three features - even with rough estimates - will show meaningful cost reduction. Bob should also bring the counterweight, unprompted, because the board will otherwise ask for it later under worse circumstances: incidents-per-merged-change and firefighting hours across the same three periods. A cost-per-feature trend that improves while incident load quietly rises is the pattern Meta's internal telemetry described, and a board that hears it first from you rather than from a journalist is a board that keeps funding the programme. Bob should present this at the board with honest caveats about estimation methodology and a commitment to systematic cost-per-feature tracking going forward. An imperfect retrospective is far better than silence."])</script>
102<script>self.__next_f.push([1,"3b:T436,Sarah has been tasked with designing the L5 metrics framework for the engineering team. She wants to move from activity metrics (PRs, commits) to outcome metrics (features shipped, user value delivered). Cost-per-feature is the bridge between engineering activity and business outcomes.\n\n**What Sarah should do:** Sarah should build a pilot cost-per-feature tracking system for one team over one quarter. The system needs three inputs: feature tagging in Jira (feature ID applied to all tickets), PR tagging in GitHub (same feature ID applied to all PRs), and a simple time estimation survey (developers estimate hours spent on each feature at sprint close). From these three inputs, Sarah can compute a rough but directionally accurate cost-per-feature for every feature shipped in the quarter. After one quarter of data, Sarah should identify: which feature types have the highest cost, which have the highest cost-to-value ratio (using any available product metrics), and where the biggest cost reduction opportunities are. This analysis is the foundation for the L5 roadmap.3c:T481,Victor is already thinking at the feature level. He tracks, informally, how long it takes him to deliver complete features from task specification to production: 2-3 hours for small features with his agent workflows, 1-2 days for medium features. He knows these numbers are dramatically better than the team average but doesn't have the data to prove it systematically.\n\n**What Victor should do:** Victor should instrument cost-per-feature tracking for his own work for one quarter. He should tag every PR he works on with the feature it belongs to, log his agent session times and costs using Claude Code's built-in usage tracking, and record his time estimates per feature. At the end of the quarter, he'll have a per-feature cost breakdown that shows: AI compute cost, CI cost, and his own time (the expensive part). Victor should then analyze whether there are patterns in which features cost more or less than expected and what drove the variance. This personal cost-per-feature analysis is the most credible possible argument for the L5 investment: real data, from a real feature portfolio, showing what autonomous feature delivery actually costs.3d:T42d,Bob suspects that half his team is using ChatGPT and Claude through personal accounts, but he doesn't have visibility into how or how often. He's received an informal inquiry from the CISO about AI tool data handling and doesn't know how to answer it. His immediate concern is that he'll be caught flat-footed in an audit with no documentation of what AI tools are being used on the codebase.\n\n**What Bob should do:** Bob should run an anonymous survey this week - three questions: what AI tools do you use for coding, are they personal or company subscriptions, what would you need to switch to company-provided tools? The results give him the actual state of shadow AI on his team and a clear procurement brief. Bob should then take the survey results to procurement and security with a simple ask: approve and fund GitHub Copilot Enterprise or Claude for Teams within 30 days, with appropriate data handling agreements. The survey results are his business case: \"X% of developers are using personal subscriptions today; this is the compliance risk we need to close.\"3e:T464,Sarah is trying to measure developer productivity and AI adoption but her data is incomplete. When she looks at GitHub Copilot usage through the enterprise dashboard, she sees low numbers - but she knows anecdotally that most developers are using AI heavily. The discrepancy tells her that most of the AI use is happening through channels she can't see.\n\n**What Sarah should do:** Sarah should treat the measurement gap as the primary problem. She can't measure what she can't see, and right now she can only see the official tools. Sarah should make the case that closing the shadow AI gap is a prerequisite for any meaningful productivity measurement program - not because measurement is the goal, but because visibility is. Once developers are using approved, o"])</script>
102<script>self.__next_f.push([1,"bservable tools, Sarah can start tracking actual AI-assisted throughput, lead time, and quality metrics. Until then, her productivity data has a systematic blind spot that will make any conclusions unreliable. Sarah should frame this to Bob as: \"we can't demonstrate AI's value to leadership if we can't measure it, and we can't measure it if we can't see it.\"3f:T413,Victor uses Claude Code on a company-provided API key and has set up a sophisticated local workflow. But he knows that three of his colleagues are still on personal ChatGPT accounts because the official tooling doesn't cover a specific use case they need - interactive architecture discussion with a model that has full codebase context. Victor can see exactly why they're using shadow tools.\n\n**What Victor should do:** Victor should bring the specific unmet use case to Bob with a concrete proposal: \"here's what's driving shadow AI, here's the tool that would address it, here's the data handling model that would make it compliant.\" Victor understands both the technical requirements and the compliance constraints well enough to design the solution. He should also volunteer to be the internal champion for whatever tool gets approved - running onboarding sessions, documenting the workflow, and helping teammates migrate off their personal subscriptions. Victor making shadow AI migration easy is more effective than any policy document.40:T448,Sarah needs to start measuring AI adoption and productivity impact, but she can't measure reliably until there's an official policy that defines what AI use is legitimate. Right now, any measurement she does captures only the fraction of AI use that happens through official tools, which dramatically understates real adoption.\n\n**What Sarah should do:** Sarah should treat the official policy publication as a measurement milestone. On day one of the official policy, she sets a baseline: here is the current state of official AI tool adoption, here are the disclosure rates in PR templates, here is the current throughput baseline. She then tracks these metrics over 90 days: how did adoption change, did throughput change, what's the quality trajectory? The policy creates the measurement infrastru
102cture she needs - every PR with an AI disclosure field gives her a data point she didn't have before. Sarah should use the first policy review meeting to present these first 90 days of data, which gives the policy review a quantitative foundation rather than being a purely process conversation.41:T561,"])</script>
102<script>self.__next_f.push([1,"Victor wants the policy to enable the advanced workflows he's using - multiple parallel agents, MCP server integrations, autonomous code generation - but he worries that a conservative first policy will prohibit these patterns and create friction for the team's most sophisticated AI use.\n\n**What Victor should do:** Victor should participate actively in the policy drafting, not just the review. He should bring specific use cases that need to be explicitly in scope: \"AI agents that can run CLI tools in sandboxed environments,\" \"MCP servers that provide codebase context to AI tools,\" \"AI-generated PRs that are reviewed and merged by human developers.\" For each use case, he should also bring a proposed governance control: what approval is needed, what gets logged, what's off-limits. Victor as policy co-author rather than policy critic produces a document that actually enables sophisticated use rather than inadvertently prohibiting it. The single most useful artefact he can contribute is the baseline deny/ask ruleset itself: a reviewed, version-controlled file that lets his parallel agents run without a prompt in front of every command, precisely because the small set of things they must never do is stated rather than adjudicated. That file buys autonomy for the team and gives the CISO something more durable than an assurance that engineers read their prompts."])</script>
102<script>self.__next_f.push([1,"42:T428,Bob's SOC2 auditors have asked specifically about AI change management controls - they want to see evidence that AI-generated changes are reviewed by a human and that the review is documented. Bob has the PR template disclosure fields from L2, but the auditors want something more structured and queryable than free-text PR comments.\n\n**What Bob should do:** Bob should implement the MVAT schema in the two weeks before the audit evidence window opens. He needs two artifacts: the schema implementation (git trailers with the four fields), and a sample audit report generated from the schema (showing 30 days of AI-assisted changes with model, timestamp, context, approver). The sample report demonstrates to auditors that the system is queryable and that the approver field documents human review. Bob should also note that the approver is captured both in the MVAT (the developer who ran the AI session) and in the PR approval (the reviewer who approved the PR) - two distinct human accountability records per AI-assisted change, which is a strong control story.43:T4d2,Victor runs complex multi-session agent workflows where a single feature involves multiple Claude Code sessions: one for the implementation, one for tests, one for documentation updates, and sometimes a separate session for a security review. The four-field MVAT schema covers single sessions well but doesn't naturally capture multi-session work.\n\n**What Victor should do:** Victor should extend the schema with a session chain concept: an `AI-Session-Chain` field that groups related AI sessions under a single feature or task identifier. When he starts an agent workflow for a feature, he generates a UUID for the chain; every session in that workflow includes the chain ID. This allows the audit trail to say \"this PR involved four AI sessions over three days; here they are in sequence.\" Victor should prototype this extension, use it for a month, then bring the usage data to the team with a proposal to standardize it. The chain concept is likely to become relevant to other developers as multi-session agent workflows become more common. Each session in the chain should also record the agent principal it ran as, so a chain that mixes a planning agent and a coding agent shows two actors, not one developer doing four things.44:T450,Bob is preparing for a major enterprise customer's security review, which includes a software supply chain assessment. The customer is asking for evidence that every change to the software that runs their data has complete provenance - from ticket to production. Bob has the MVAT audit trail from L3, but the customer wants to see the ticket linkage and the deployment record, not just the commit metadata.\n\n**What Bob should do:** Bob should scope a provenance graph MVP that connects the three systems the customer cares about: the issue tracker (Jira or Linear), the code repository (GitHub), and the deployment platform (Kubernetes with deployment records). The MVP is a script that, given a ticket ID, traverses the graph and outputs: the ticket, the PRs that closed it, the commits in those PRs with AI provenance data, and the deployment record. This traversal report is the supply chain evidence the customer wants. Bob doesn't need to build the full graph database in the first iteration - a script that queries three APIs and joins the results is enough to demonstrate provenance for the audit.45:T421,Victor's workflow involves complex multi-agent orchestration: a planner agent creates a task breakdown, multiple worker agents implement the tasks in parallel, and Victor reviews the combined output. The provenance graph needs to capture this hierarchical agent structure, not just flat session-to-commit links.\n\n**What Victor should do:** Victor should propose and implement a hierarchical provenance model that captures agent orchestration. The graph extension adds: a \"workflow\" node that represents the top-level task, \"session\" nodes for each agent session, and parent/child edges between orchestrator and worker sessions. The commit provenance then links t"])</script>
102<script>self.__next_f.push([1,"o the workflow node rather than individual sessions. This model makes it possible to answer: \"what was the top-level task that motivated this change?\" and \"which agent sessions contributed to this PR and in what roles?\" Victor should prototype this with his own workflows, validate that the resulting graph is queryable and useful, and then propose it as the standard for multi-agent provenance.46:T501,Bob's company operates across multiple EU jurisdictions and is trying to keep up with the rapidly evolving EU AI Act implementing regulations. His compliance team spends three days per month reading regulatory publications and assessing their impact, and the lag between publication and action is still averaging six weeks. Bob wants to reduce the monitoring burden and the compliance lag simultaneously.\n\n**What Bob should do:** Bob should frame this as an engineering problem with a build plan. The minimum viable continuous compliance agent for his situation is: (1) a scheduled job that pulls EU AI Act publications from EUR-Lex weekly, (2) a Claude call that extracts material requirements from each new publication and compares them to the current compliance posture document, (3) a Jira ticket creator that writes the impact assessment as a structured ticket in the compliance backlog. This is a two-week engineering investment. It doesn't replace the compliance team's judgment - it replaces their reading time and first-pass assessment. Bob should pilot this for 90 days and measure: did the agent correctly identify the same material changes the compliance team would have identified? Were there any the agent missed? Were there any the team missed that the agent caught?47:T419,Sarah is responsible for the compliance dashboard that leadership reviews monthly. Currently, the dashboard is assembled manually from a spreadsheet that the compliance team updates after each quarterly review. The dashboard is accurate as of the last review but can be meaningfully stale between reviews.\n\n**What Sarah should do:** Sarah should use the compliance posture database (built as part of the continuous compliance system) to automate the dashboard. Instead of a quarterly manual update, the dashboard reflects the current state of the compliance posture database, which is updated continuously as controls are implemented and requirements are added. Sarah should also add a \"compliance velocity\" metric: how quickly is the gap between new requirements and implemented controls closing? This metric makes the efficiency of the compliance process visible to leadership and creates accountability for the review and remediation SLAs. A dashboard that's always current is fundamentally more useful than one that's accurate four times a year.48:T460,Victor wants to take the monitoring agent beyond passive monitoring to active remediation. When the agent identifies a change to an OPA compliance rule that's required by a new regulatory requirement, he wants the agent to draft the rule change, write the tests, and submit a PR - not just file a ticket.\n\n**What Victor should do:** Victor should implement the active remediation capability for the narrowest possible scope first: OPA rule additions for new audit trail requirements. When the impact assessment identifies a new required field in the AI audit trail (a common type of requirement from evolving regulatory guidance), the agent should be able to: look up the current OPA rule, add the required field check, write a test case, and submit a PR to the compliance repository with a clear description of the regulatory requirement it implements. Victor should validate this capability with a historical test: take a real regulatory change from the past year that required an OPA rule update and verify that the agent produces the correct PR. Once validated, expand the remediation capability to other rule types.49:T42f,Bob's AI program has grown from the original 2-team pilot to 8 teams over the past six months. Each team has a different tool configuration, different security review status, and different levels of champion support. Th"])</script>
102<script>self.__next_f.push([1,"e operational overhead of managing the inconsistency is growing, and Bob has had three security-related questions in the past month that he didn't have consistent answers to.\n\n**What Bob should do:** Bob should formalize the platform team's ownership of AI tooling as an explicit decision, not as an organic evolution. This means: identifying who on the platform team is responsible for AI tooling (with allocated time, not just \"also do this\"), defining the scope of their ownership, and commissioning the three initial deliverables that prove the platform is working - a security clearance that applies org-wide, a self-service provisioning mechanism, and a 30-minute onboarding guide. Bob should also communicate the transition to team leads clearly: the platform team is taking over the infrastructure decisions; teams keep their workflow autonomy.4a:T433,Sarah has been tracking adoption metrics for 8 teams and the inconsistency is making her job increasingly difficult. Team A uses one tool, Team C uses another, Teams D and E have different API configurations, and two teams haven't completed security review. She can't produce a coherent org-level adoption picture from this data.\n\n**What Sarah should do:** Sarah should use her measurement challenges as the business case for platform team ownership. She should document the specific costs of the current inconsistency: the time she spends normalizing incompatible data formats, the gaps where no data exists, the security questions that don't have consistent answers. This is not a complaint - it is a resource allocation argument. Sarah should propose that the platform team take ownership of the tooling infrastructure, with a specific focus on the four things that would most improve her ability to measure adoption: standardized tooling configuration, consistent authentication approach, a usage telemetry pipeline, and a single security clearance that covers all teams.4b:T405,Victor has been the de facto platform for AI tooling in addition to his champion role. Developers across multiple teams come to him for setup help, configuration questions, and security escalations. He's spending 40% of his time on infrastructure problems rather than the workflow knowledge transfer that is actually his highest-value contribution.\n\n**What Victor should do:** Victor should document everything he's been doing informally and propose that it be formally owned by the platform team. He should write a handoff document: here is what developers ask for, here is how I answer it, here is what the platform team needs to build to make my answers unnecessary. Victor should also be explicit with Bob that his current role is not sustainable - he is doing both champion and platform engineer work with time allocated for neither. The solution is not for Victor to work harder; it is for the organization to clarify ownership so that infrastru
102cture questions go to the platform team and workflow questions come to Victor.4c:T486,Bob's organization has 80% AI tool adoption (weekly active users) and a successful platform team owning the tooling. By usage metrics, the program is succeeding. But when Bob reviews PRs and attends technical discussions, he notices that the most senior developers rarely mention AI in their design conversations, and the hardest technical problems are still being approached manually. The tools are being used for the easy stuff.\n\n**What Bob should do:** Bob needs to lead the cultural shift on the hard problems, not the easy ones. He should start by modeling the behavior he wants to see: at the next architecture review, Bob should share how he used Claude Code to explore the design space for a complex decision, including where the agent's output was wrong and how he corrected it. This normalizes AI use on hard problems and signals that using AI assistance on difficult work is what sophisticated engineers do, not something to hide. Bob should also restructure performance conversations to explicitly include \"how are you using AI tools to multiply your output on your hardest problems?\" - not as a gotcha, "])</script>
102<script>self.__next_f.push([1,"but as a genuine development conversation.4d:T435,Sarah's adoption metrics look strong but her qualitative surveys reveal a concerning pattern: developers report using AI tools on \"routine tasks\" at high rates, but on \"complex architectural decisions\" and \"difficult debugging sessions\" at very low rates. The culture has adopted AI for the easy stuff and is leaving the highest-leverage applications untouched.\n\n**What Sarah should do:** Sarah should reframe her measurement around AI use on high-complexity tasks, not just overall usage. She should design a survey dimension specifically asking about AI use on the developer's hardest recent work: \"In the last two weeks, did you use AI assistance on a problem that you genuinely weren't sure how to approach? If yes, what was the outcome?\" The answers to this question reveal whether the cultural shift has reached the hard problems. Sarah should share this measurement gap with Bob and propose a specific intervention: a series of case studies from developers who used agents on genuinely hard problems, shared at the monthly all-hands, to normalize AI use on complex work.4e:T43a,Victor uses AI assistance on genuinely hard problems - complex refactors, ambiguous architectural decisions, difficult debugging sessions - and gets significant value from it. But he's noticed that his colleagues still treat AI as \"for the easy stuff.\" When he mentions using Claude Code to explore three architectural alternatives before a design meeting, colleagues express surprise, as if this is unusual or excessive.\n\n**What Victor should do:** Victor should make his AI use on hard problems visible and teachable. He should write a detailed internal post about a specific instance: \"Here is a hard problem I recently faced. Here is how I used Claude Code to explore the solution space, including the three approaches the agent suggested, the two that were wrong and why, and the one I ended up using with modifications. Here is what I learned about how to use AI assistance effectively on genuinely hard engineering problems.\" This is the content that shifts culture - not \"AI is great\" enthusiasm, but concrete, honest accounts of how agents change the approach to hard work.4f:T43d,Bob's engineering organization is running at L4/L5 boundary - agent use is pervasive, cost is becoming a budget concern, and three coordination failures in the past month (agents conflicting on the same files) have created significant rework. Bob knows that the current fleet management approach won't scale to the next level but isn't sure whether to invest in centralized orchestration or buy an emerging commercial solution.\n\n**What Bob should do:** Bob should commission a 60-day investigation before committing to either building or buying. The investigation should produce three things: a workload map (what task types are running, what models, what cost), a failure taxonomy (what went wrong in the three coordination incidents and what infrastru
102cture would have prevented it), and a build-vs-buy analysis of the top two or three commercial orchestration platforms against an internal build. This investigation will be cheaper than getting the build-vs-buy decision wrong and will produce the requirements document that makes whichever option Bob chooses more likely to succeed.50:T47b,Victor has become the de facto expert on agent orchestration in the organization. He has been building increasingly sophisticated ad-hoc coordination mechanisms - shared state files, agent dependencies tracked in a spreadsheet, manual cost monitoring. He knows these approaches are not sustainable at the next scale, and he has a clear picture of what the centralized orchestration system needs to provide.\n\n**What Victor should do:** Victor should write the technical requirements document for the centralized orchestration system based on his operational experience. This document should cover: the workload characteristics (task types, sizes, dependencies), the failure modes he has encountered and their frequency, the coordination mechanisms he has been implementing manual"])</script>
102<script>self.__next_f.push([1,"ly that the orchestration layer should automate, and the observability requirements for the measurement systems Sarah needs. Victor should propose to Bob that he lead the orchestration system design - not the implementation (that should be a full team), but the architectural design and requirements specification where his operational experience provides unique input.51:T432,Bob has a team of 40 engineers across 6 teams. He estimates that onboarding takes 3 months to full productivity, but he has never quantified how much of that time is spent discovering folk traditions versus genuinely learning the domain. He suspects it's a lot. When he tries to push for better documentation, he gets pushback from seniors who say they don't have time.\n\n**What Bob should do:** Bob should run the traps audit as a structured initiative, framing it as a productivity investment rather than a documentation chore. The output is a count: how many folk traditions exist, how often they fire, and who owns them. This count is the business case. If 40 engineers each lose 2 hours per month to folk tradition traps (a conservative estimate for a 3-year-old codebase), that's 80 hours of productivity lost monthly. Documenting the 20 highest-impact traps is a day of work that pays back in weeks. Bob should make \"eliminate one folk tradition per sprint\" a team-level commitment for every team, with the count tracked as a leading indicator of documentation health.52:T416,Sarah has been measuring time-to-productivity for new engineers and has noticed that the fastest onboarders are the ones assigned to seniors who proactively share tribal knowledge. The slowest onboarders are left to figure things out from the documented procedures. This tells her the documented procedures are missing critical information, but she hasn't been able to quantify exactly what's missing.\n\n**What Sarah should do:** Sarah should instrument the onboarding process with a structured \"trap journal\" - a doc that new engineers fill out during their first 60 days, recording every time they got stuck on something that wasn't in the docs. After 3-4 new hires complete this, Sarah will have a ranked list of the highest-impact folk traditions. She should then track documentation coverage: what percentage of the items in the trap journal have been added to official documentation? This metric - trap documentation rate - is a leading indicator for whether the organization is closing the gap or letting it grow. Share it with Bob monthly.53:T434,Victor is trying to integrate AI agents into the team's development workflow, but he keeps hitting a wall: agents fail in confusing ways because they encounter undocumented preconditions that every human engineer knows about but nobody has written down. The agents aren't broken - the documentation is. Victor needs to fix the documentation before the agents can work reliably.\n\n**What Victor should do:** Victor should treat every agent failure as a documentation bug. When an agent fails because it didn't know to stop Redis first, that's a missing prerequisites section, not an agent failure. Victor should keep a running list of agent failures caused by missing context and use it as a prioritized documentation backlog. Each item he documents not only fixes the agent workflow - it also fixes the human onboarding experience. Victor should make this case explicitly to the team: \"every fix I make to help agents work reliably is also a fix that makes onboarding faster.\" This framing gets documentation work done faster than framing it as infrastructure investment alone.54:T443,Bob's team just finished a 3-week sprint that was entirely consumed by a docs refresh after a new engineer spent a week following stale runbooks and causing two minor incidents. The refresh was painful and expensive. Bob wants to avoid doing it again, but he's not sure what structural change would prevent the next round of drift.\n\n**What Bob should do:** Bob should use the refresh as a forcing function for exactly one structural change: adding a \"docs updated?\" checkbox to every PR template, wi"])</script>
102<script>self.__next_f.push([1,"th the expectation that any PR changing behavior includes a docs update. This single change prevents the most common source of drift - the gap between code changes and documentation changes. Bob should also assign quarterly documentation ownership for the 5 most critical doc categories (runbooks, onboarding, API docs, architecture, deployment). Each owner is responsible for one verification per quarter - not a full refresh, just a verification pass that takes an hour. These two changes (PR requirements and quarterly ownership) eliminate most drift without requiring another full sprint.55:T43e,Sarah can see that the docs refresh significantly reduced the \"time lost to stale documentation\" metric she tracks - engineers stopped asking certain recurring questions in Slack for about 6 weeks after the refresh. But then the questions started coming back, the pattern resumed, and within a quarter the team was back to the pre-refresh baseline. She needs to break the cycle.\n\n**What Sarah should do:** Sarah should present the decay curve to Bob and engineering leads: documentation accuracy peaks after a refresh and then decays on a predictable curve. The half-life of documentation accuracy in your codebase is approximately X weeks (Sarah can calculate this from her data). To maintain 70% accuracy without structural changes, you'd need a full refresh every Y weeks - at a cost of Z engineer-days. The alternative is structural changes that extend the half-life: PR requirements, automated staleness detection, co-located docs. Sarah should frame this as a calculation, not a complaint. The structural investment is the cheaper option once you account for ongoing refresh cost.56:T45b,Victor has been trying to use the refreshed documentation to improve agent performance, but he's watching the accuracy decay in real time. He can see which docs are drifting - the ones in active development areas are already diverging from reality two weeks after the refresh. He needs a way to maintain accuracy in the high-traffic areas that matter most for agent workflows.\n\n**What Victor should do:** Victor should identify the 10 documents most critical for agent workflows - the runbooks, setup guides, and architectural overviews that agents consult most often - and propose co-locating them in the repository alongside the code they describe. When these docs live in the repo, PRs that change the corresponding code will be visible alongside the docs, making it natural to update them simultaneously. Victor should also write a simple CI check: for each \"agent-critical\" doc, verify that the commands it contains actually execute without error in a fresh environment. This check catches the most common form of staleness - commands that no longer work - automatically, without requiring manual verification.57:T46c,Bob has been asking his team to write better documentation for years, with limited success. Seniors are busy and deprioritize documentation when sprint work is heavy. Juniors write documentation that drifts from reality as the code evolves. Bob measures test coverage because it is visible in the CI dashboard, but he has never measured documentation coverage because he has never had a way to do it.\n\nThe shift Bob needs to make is from treating documentation as a cultural output to treating it as an engineering deliverable with the same visibility and accountability as test coverage. He should work with Victor to define three documentation metrics: API coverage rate, ADR coverage rate for significant architectural choices, and onboarding path freshness score. He should add these to the engineering dashboard next to test coverage and build health. He should allocate a fixed percentage of sprint capacity to documentation maintenance and track it quarterly. None of this requires motivating individual engineers to document more - it requires building the systems that make documentation creation and maintenance automatic.58:T477,Sarah's perspective on documentation has always been organizational: documentation helps people work better, onboarding faster, and collabo"])</script>
102<script>self.__next_f.push([1,"rate more effectively. What shifts at L3 is her understanding of documentation as a technical artifact that requires the same engineering discipline as the code it describes. This changes her role from documentation cheerleader to documentation product manager - someone who defines the requirements, measures the outcomes, and holds teams accountable to coverage standards.\n\nSarah should partner with Victor to define what \"documentation-as-infrastructure\" looks like for her organization specifically: which documentation types are in scope, what coverage standards are acceptable, and how accuracy is measured. She should own the documentation metrics dashboard and present it monthly to Bob as part of engineering health reporting. She should also connect the documentation infrastructure investment to concrete outcomes she can measure: onboarding time, time-to-first-independent-PR, agent output quality scores. These connections make the infrastructure investment legible as a business investment.59:T411,Victor has been arguing for better documentation as an AI quality issue for months, but the framing of \"we need better docs for the agents\" has not gained traction. The infrastructure framing is more effective: documentation quality is a reliability concern with measurable cost and measurable return, just like test coverage or build stability. Victor should reframe his advocacy accordingly.\n\nConcretely, Victor should build the CI checks and automation that make documentation-as-infrastructure operational. This means lint rules for required docstrings, a bot that flags PRs changing architecture without an ADR, and agent pipelines that generate documentation first drafts from code changes for human review. He should instrument the correlation between documentation coverage and agent output quality, and present the data to Bob and Sarah: \"teams with higher documentation coverage get measurably better agent suggestions on the same tasks.\" This data converts the philosophical argument into an engineering decision with a clear ROI.5a:T41e,Bob's agents are producing inconsistent output. The same agent on the same task produces excellent results when an experienced engineer provides rich context and mediocre results when a junior engineer provides minimal context. The quality gap is not in the agent - it is in the context assembly. Bob wants to standardize agent output quality across his teams without requiring every engineer to become an expert context assembler.\n\nThe Context Fabric is the structural answer to Bob's problem. By making context delivery automatic and systematic, it removes context quality as a variable in agent output. Bob should fund a dedicated Context Fabric team - or give Victor a quarter of his time - to build and maintain the organization's MCP server infrastructure. He should measure the before/after impact on agent output quality, using peer evaluation of agent-generated PRs as the measurement instrument. He should treat Fabric coverage as a key engineering metric reported monthly, with a target to cover all major context sources within four quarters.5b:T4e0,Victor is the natural owner of the Context Fabric as a technical infrastructure project. He understands which MCP servers are available, which context sources matter most, and how to configure agents to use them effectively. What he needs from Bob and Sarah is explicit resource allocation: building and maintaining the Fabric is engineering work that competes with feature development, and it will not happen without clear prioritization.\n\nVictor should start with the three MCP servers that will have the highest immediate impact on agent output quality: the knowledge graph MCP server (structural codebase context), the ADR MCP server (architectural decision history), and the ticket tracker MCP server (current work context). He should measure output quality improvement after each deployment using a consistent evaluation rubric. He should document the Fabric architecture - which servers exist, what they cover, how they are connected - and make that documentation av"])</script>
102<script>self.__next_f.push([1,"ailable to all engineers so they understand what context agents have access to. Over time, he should build a Fabric expansion process: a lightweight proposal format for adding new MCP servers, with clear criteria for when a context source is worth the maintenance investment.5c:T4d5,Bob has invested in documentation infrastructure over the past two years: ADRs are now consistently written, onboarding paths are maintained, and the auto-update pipeline keeps API reference docs current. What he is still fighting is the decay of the documentation that requires ongoing maintenance and human judgment - architecture overviews that drift as the system evolves, runbooks that become inaccurate after incidents reveal gaps, design rationale that was never written down in the first place.\n\nA self-evolving knowledge base addresses the maintenance problem Bob cannot solve with human effort alone. Bob should work with Victor to define which documentation types are candidates for self-evolution, set accuracy requirements for each type, and establish the governance model for human review of agent proposals. He should treat the self-evolving knowledge base as a strategic infrastructure investment with a multi-year payback horizon: the value compounds with age, and the compounding starts from the moment the first self-evolution loop closes. He should also identify a documentation accuracy metric - perhaps measured by quarterly human spot-checks - and track it over time as the primary health indicator for the system.5d:T450,Sarah has been the most consistent advocate for documentation quality in the organization, and she has seen the limits of what human effort alone can achieve. Teams that are conscientious about documentation during calm periods let it slip during crunch. The self-evolving knowledge base is the answer she has been looking for: a system that maintains documentation quality independent of team bandwidth and individual discipline.\n\nSarah should own the accuracy measurement process for the self-evolving system. She should run quarterly documentation accuracy audits: sample 20-30 documentation artifacts across all types, verify accuracy against the codebase, and produce an accuracy score by documentation type. She should present this score to Bob as part of engineering health reporting and use it to identify which self-evolution mechanisms are working well and which need improvement. She should also track the engineer time saved by self-evolution: hours not spent writing documentation updates, multiplied by the engineering cost. This is the business case for continued investment in the system.5e:T4aa,Victor has been building toward self-evolution incrementally for the past year. The Context Fabric is mature, the auto-update pipeline is operational, and the knowledge graph is providing structural context. The remaining work is connecting these components into a closed loop: detection triggering correction, correction triggering validation, validation confirming accuracy. This is primarily integration work, not new infrastructure work - the components exist; they need to be composed.\n\nVictor should approach self-evolution as a series of closed loops, each validated before the next is added. Loop one: code change triggers documentation update. Loop two: staleness detection triggers correction proposal. Loop three: new pattern detection triggers standards proposal. He should run each loop independently for one quarter before combining them, using the accuracy metrics from each to build confidence before expanding scope. He should document the self-evolution architecture - which loops exist, what triggers each, what the human review surface is - and make that documentation part of the system's own self-evolving knowledge base, as a concrete demonstration that the system works.5f:T48d,Bob runs a team of twelve: eight developers, two QA engineers, and works closely with two PMs. The team delivers consistently but velocity has been flat for two years. He's heard that AI tools can improve productivity but hasn't made any formal moves yet"])</script>
102<script>self.__next_f.push([1," - a few developers use Copilot individually but it's not standardized or measured.\n\n**What Bob should do:** Bob should start by documenting the current state with precision: draw the actual workflow map, measure the actual cycle times, and inventory the informal AI tool use. This is not a bureaucratic exercise - it's the baseline that will let Bob demonstrate concrete improvement when AI adoption begins. The documentation exercise itself often surfaces inefficiencies: requirements that get revised three times between PM and dev, QA sign-off that takes 48 hours because of queue management rather than actual testing time. Bob should share the baseline with his team and frame the next six months as \"let's figure out where AI tools can improve each of these specific pain points.\" This is more compelling than \"we're going to adopt AI\" - it's a concrete problem-solving exercise with measurable outcomes.60:T47e,Sarah has been asked to lead the AI tooling initiative for the engineering organization. She's looking at a landscape of 40 developers, 8 QA engineers, and 6 PMs across four teams - all operating in traditional role structure with ad-hoc AI use. She needs to create a transition plan that doesn't create role anxiety or organizational disruption.\n\n**What Sarah should do:** Sarah should run a role impact assessment before recommending any tooling. For each traditional role, map: what will AI tools change about this job in the next 12 months? What new skills will be needed? What existing skills become less central? Present this to each role group honestly - QA engineers need to know that AI-assisted testing will change their job, not eliminate it, but the nature of the work shifts from manual test execution to test strategy and agent supervision. PMs need to know that writing agent-readable requirements is a skill they'll need to develop. Sarah should build the transition plan around skill development paths, not just tool adoption targets. This frames AI adoption as a growth opportunity for every role rather than a threat to some roles.61:T41a,Sarah is designing the formal AI champion program for an engineering organization of 120 developers across 15 teams. She needs to define the role, create a selection process, design the champion community structure, and establish success metrics - all in a way that doesn't create bureaucratic overhead that crowds out the actual champion work.\n\n**What Sarah should do:** Sarah should design a lightweight program: one-page role definition, three criteria for champion selection (credibility with team, curiosity about AI, available bandwidth), a monthly 90-minute champion sync, and a shared wiki where champions post their best practices. The success metrics should be outcome-based: adoption rate on each team, number of reusable prompt templates created, and a quarterly self-reported effectiveness score from each champion. Sarah should resist the temptation to create elaborate reporting and certification structures - the champion program is valuable because it's close to the work, not because it generates comprehensive data for a dashboard.
10262:T45e,Victor has been doing champion-level work informally for six months. He's written a comprehensive CLAUDE.md for his team's codebase, created a dozen prompt templates for common tasks, and answered hundreds of AI questions from his colleagues. He's tired and starting to wonder if this work is valued, because it's not reflected in his performance review or his official responsibilities.\n\n**What Victor should do:** Victor should make the value visible. He should prepare a one-page summary of his champion contributions: the CLAUDE.md that reduced senior debugging time by X%, the prompt templates that saved the team Y hours over the last quarter, the colleagues he trained who are now using AI effectively. He should share this with Bob and ask for explicit recognition - either as part of his performance review, as a formal role addition to his job description, or as part of the case for his next promotion. Victor's work is genuinely valuable and should "])</script>
102<script>self.__next_f.push([1,"be recognized as such. If it isn't recognized after he makes it visible, that's a useful signal about whether this organization is serious about AI adoption.63:T471,Bob's team is producing more code volume than ever, but review has become a bottleneck. His senior developers are drowning in PR reviews and complaining about the volume. The junior developers, who are generating most of the AI-assisted code, are frustrated by the slow review cycle. Bob suspects the review process is not adapting to the new reality.\n\n**What Bob should do:** Bob should run a review retrospective: how are people currently reviewing AI-generated PRs? What's taking the most time? Where are the most back-and-forth review cycles happening? This retrospective will almost certainly reveal that reviewers are applying traditional line-by-line standards to AI-generated code that mostly fails on intent alignment rather than individual line correctness. Bob should introduce the intent-first protocol and the AI code review checklist, run a workshop to walk through the approach with examples, and measure review cycle time before and after. The expectation: review time per PR drops by 20-30% and the quality of review feedback improves (more \"this doesn't match the intent\" comments, fewer \"move this variable\" comments).64:T457,Sarah's developer survey shows a split: AI tool users rate code review as \"more frustrating\" than non-AI users, despite generating code faster. When she digs into the comments, the pattern is clear: AI-generated PRs are getting the same level of detailed stylistic feedback as hand-written PRs, and developers feel the standard is unreasonably high. She needs to recalibrate the review culture without reducing quality.\n\n**What Sarah should do:** Sarah should make the review standard explicit. Currently it's implicit and inconsistent - different reviewers apply different standards to AI-generated code. She should work with the engineering leads to define the tiered review standard (blocking/suggested/optional) for AI-generated PRs, create example review comments for each tier, and share this standard in the engineering handbook. She should also create an internal guide to intent-first review that developers can reference. The goal is not lower standards - it's appropriately targeted standards that focus review energy on the defects that actually matter and don't create friction on matters of style.65:T474,Victor reviews more AI-generated code than anyone on the team and has developed strong intuitions for what to look for. He can scan a 200-line AI-generated PR in 10 minutes and find the one or two issues that actually matter. He can also spot when an AI has misunderstood the spec before reading a single line of code, just by reading the diff size versus the spec complexity. He hasn't articulated how he does this.\n\n**What Victor should do:** Victor should externalize his review process. He should write up the three or four heuristics he applies when reviewing AI-generated PRs and share them as the team's AI code review guide. He should also create a \"before/after\" example: take a real PR review from the past month, show the original review comments, and then show how the intent-first review would have approached the same PR. This makes the skill concrete and teachable. Victor should also advocate for the team to track review metrics separately for AI and human code - not to apply different standards, but to learn where AI-generated code is and isn't working so that context infrastructure can be improved in the right places.66:T44b,Bob is doing sprint planning and trying to estimate velocity for the upcoming quarter. His team has 8 developers operating at various levels of AI fluency. He's trying to figure out how many parallel agent workstreams the team can sustain and what the actual throughput multiplier is compared to pure human development.\n\n**What Bob should do:** Bob should survey each developer's current effective span and use this to estimate total concurrent agent capacity. If six developers are effectively running 4-5 agents each "])</script>
102<script>self.__next_f.push([1,"and two are at 2-3 agents, the team has roughly 28-36 concurrent agent workstreams available. This is a more accurate planning number than \"we have 8 developers with AI tools.\" Bob should also distinguish between developer time (thinking, specifying, reviewing) and agent time (executing tasks). A developer managing 5 agents is still working a full day - they're just doing different work. The throughput multiplier is in the agent execution, not in the developer headcount. Bob should communicate this distinction to stakeholders who ask \"how many developers does this replace.\"67:T5f4,"])</script>
102<script>self.__next_f.push([1,"Victor runs at a span of 6-7 agents effectively and has developed the most sophisticated fleet management workflow on the team. He's been asked to help two other developers increase their effective span from 3 to 5. He's not sure how to coach this transition because most of what he does has become intuitive.\n\n**What Victor should do:** Victor should work backwards from his current practice to make the implicit explicit. He should spend a day narrating his decisions aloud as he manages his agent fleet: \"I'm launching three agents now instead of five because these tasks are all in the same service and I want to sequence them to avoid conflicts. I'm checking in on agent two first because it's working in an area with unclear requirements. I'm not interrupting agent four because it's on track with a well-defined task.\" This narration reveals the judgment calls that experienced fleet management requires. Victor should record this or take notes, then work with the two developers he's coaching to help them make the same judgments consciously before they become automatic. He should also be careful about what he coaches toward. Most of what makes his practice work is the four rules rather than his personal capacity - batches held at two to four, no polling, no concurrent repo-wide git operations, overlapping file ownership merged rather than arbitrated - and those transfer to a developer on their first week. Coaching someone from three agents to five without those rules just moves the failure one agent later."])</script>
102<script>self.__next_f.push([1,"68:T559,"])</script>
102<script>self.__next_f.push([1,"Bob's organization is approaching L5. The AI infrastructure is mature, the developers are operating effectively as fleet managers, and the organization is producing at 4-5x the throughput of two years ago. Bob is starting to think about what the next step looks like and realizes he needs someone who can design the organizational-scale orchestration system that turns individual team-level agent workflows into a coordinated, supervised, architecturally-coherent enterprise capability.\n\n**What Bob should do:** Bob should identify whether the Agentic Engineer he needs is someone to develop from within or hire from outside. The internal candidate - likely Victor or someone on Victor's trajectory - has the domain knowledge and organizational context that an external hire lacks. An external hire might bring more sophisticated orchestration experience but will need 6-12 months to develop the domain context. Bob should have a frank conversation with his best L4 fleet managers about whether the Agentic Engineer role interests them and what they would need to grow into it. If the internal path is viable, investment in training and mentorship is more effective than an external search. If it isn't viable, define the hire's first 90 days explicitly: spend time with each team understanding the domain before designing the organizational orchestration architecture."])</script>
102<script>self.__next_f.push([1,"69:T4b9,Sarah is trying to define the hiring profile for Agentic Engineers for an organization that has never hired for this role. She's looking at job postings from other companies and finding wildly inconsistent definitions - some are really just senior developers with AI interest, others are ML engineers who have wandered into agent systems, and a few are the genuine role she's trying to hire for.\n\n**What Sarah should do:** Sarah should define the Agentic Engineer role by its outputs, not by its inputs. The outputs: a production agent orchestration system that runs reliably; a supervision framework that maintains quality without requiring constant human review; architectural standards that make agent operation safe in the organization's specific technical environment; and a capability-building program that helps L4 developers develop toward L5. Any candidate who can demonstrate these outputs from previous work is the right hire, regardless of how they got there. Sarah should design the interview process around these outputs: ask candidates to describe a production agent workflow they designed end-to-e
102nd, including the failure modes they designed for and how they verified output quality at scale.6a:T523,Victor has been on the journey from AI champion through context engineer to L4 fleet manager. He's now being given the opportunity to define and inhabit the Agentic Engineer role for the organization. He's uncertain because the role is new and he's not sure he's fully qualified for it - nobody is fully qualified for a role that didn't exist 18 months ago.\n\n**What Victor should do:** Victor should recognize that being on the frontier of a new role is not the same as being unqualified for it. The skills he has - deep codebase knowledge, fleet management experience, context engineering expertise, and the ability to see systems problems across the organization - are exactly the skills the Agentic Engineer role requires. What he doesn't have yet is experience with organizational-scale orchestration systems; he should invest in this deliberately by designing a small production workflow end-to-end, studying how other practitioners are approaching the orchestration and supervision problems, and connecting with the external community of agentic engineering practitioners. Victor should also be generous with his knowledge as he develops this expertise - the Agentic Engineer who builds the field rather than just claiming a title is more valuable to the organization and to the broader engineering community.6b:T458,Victor has worked in this codebase long enough to have a mental map of every danger zone. He knows which files have implicit dependencies on global state, which modules silently ignore errors, and which abstractions were designed for a requirements set that no longer exists. That knowledge lives entirely in his head, and he is increasingly the person who has to review any change touching those areas - a bottleneck that scales badly as the team grows.\n\nVictor should externalize his mental debt map into a structured form, even if it starts as a commented list in a markdown file. That map has two values: it tells other developers where to be careful, and it tells an AI agent where to start when automated debt reduction becomes available at higher maturity levels. Victor should also advocate for the 10% debt budget, because he is the one who pays the highest cost when debt-laden code breaks - the midnight pages, the complex debugging sessions, the hours explaining context to developers who are new to the area. His credibility is the most effective argument for treating debt as a first-class concern.6c:T41d,Bob has established that debt is a real cost and has the team's buy-in on addressing it. The problem now is that without a structured inventory, every sprint planning session involves a different set of developers advocating for different debt items based on their personal frustration. There is no shared view of what should be addressed next, and the selection process feels arbitrary.\n\nThe debt inventory solves this problem "])</script>
102<script>self.__next_f.push([1,"by making prioritization a structured process rather than a political one. Bob should run the initial inventory sprint himself or with his tech leads, establish the scoring rubric, and then present the top 10 items to the full team. The conversation shifts from \"I think we should fix the authentication module\" to \"the authentication module is P1 security debt with a score of 9; the reporting module is P2 architectural debt with a score of 6; based on our capacity, we can address the authentication module this quarter.\" This is a more productive conversation, and it produces more defensible commitments to stakeholders.6d:T440,Bob has been running continuous modernization for six weeks. The agent has generated 34 PRs: 28 have been merged, 4 are in review, and 2 were closed because they touched code areas that were in active feature development. The merged PRs have addressed 14 items from the debt inventory - more progress than the team made in the previous year through manual effort. Bob is converting from skeptic to advocate.\n\nThe next challenge for Bob is organizational: other teams have heard about this and want to adopt it, but they do not have a structured debt inventory, and their test suites are less reliable than his team's. Bob should create a \"continuous modernization readiness checklist\" - the prerequisites that teams must meet before enabling an agent-based migration workflow. The checklist is: structured debt inventory with 20+ items, test coverage above 70% in the areas to be migrated, and a designated PR reviewer who commits to the 48-hour SLA. Teams that meet the checklist get access to the shared agent configuration. Teams that do not have a clear improvement path to readiness.6e:T653,"])</script>
102<script>self.__next_f.push([1,"Sarah now has the data she has been waiting for: a clear before-and-after comparison for debt reduction velocity. Before continuous modernization: the team addressed 2 debt items per quarter through manual effort. After continuous modernization: the team is addressing 14 items in six weeks. The engineering time per item has dropped from approximately 40 hours (manual) to approximately 2 hours (review only). That is a 20x productivity improvement on debt reduction specifically.\n\nSarah should publish this comparison as a case study. It is the clearest possible demonstration of AI maturity ROI, and it will be more persuasive to other teams and to leadership than any abstract argument. She should also start tracking \"debt backlog age\" - the average age of items in the inventory - as a leading indicator of organizational technical health. As continuous modernization runs, this metric should decline. If it does not, the agent's output rate is not keeping up with new debt creation, which is itself a useful signal.\n\nThe stronger version of the case study adds a second number in a currency finance already understands. Alongside items closed and hours saved, Sarah should record the input tokens an agent needs for a representative change in each area of the codebase, before and after the agent works it. The controlled experiment behind this practice reported an 83% reduction on a single decomposed file, roughly $0.40 per change, applying to every subsequent modification of that code. That is an argument for refactoring that does not rely on anyone believing an engineer's intuition about maintainability."])</script>
102<script>self.__next_f.push([1,"6f:T495,Victor is the one who configured the agent, wrote the task specifications for the debt inventory items, and is the primary reviewer of agent-generated PRs. He has developed a strong sense for the difference between agent output that is clearly safe to merge quickly and agent output that needs careful review. Dependency bumps with green tests: quick merge. API migration in a module with low test coverage: careful review. Framework upgrade touching 300 files: line-by-line review of a sample, test run in staging.\n\nVictor should codify this review heuristic as a written protocol - the \"agent PR review guide\" for his team. The guide specifies: for each category of agent-generated PR, what is the appropriate review depth? Which categories can be merged by any developer? Which require Victor's review? Which require staging deployment before merge? This protocol scales Victor's knowledge to the entire team and means that when Victor is on vacation, the agent's output does not pile up unreviewed. The protocol is also the foundation for the higher-autonomy configuration at L4, where the lowest-risk categories move from \"requires review\" to \"auto-merge on green CI.\"70:T409,Bob has a list of seven \"dead\" internal tools that finance, HR, and operations still depend on. They run on Java 8, Spring Boot 1.5, and a deprecated internal authentication library. Nobody has touched them in three years. Bob has been carrying them as a liability on his tech debt register, marked as \"too expensive to address,\" because manual modernization would take a developer six months across all seven projects.\n\nBob should run a dead project modernization sprint using agents. The prerequisites: generate smoke tests for each tool, then run the migration agent. Bob's estimate for the agent-based approach: two weeks of elapsed time, approximately 20 hours of human review time, and perhaps $300 in API costs. Compare to the six-month manual estimate. Even if the agent approach takes twice as long as estimated due to complications, it is still dramatically more economical. Bob should treat this as a pilot for agent-based technical debt reduction and plan to expand the approach to the rest of his dead project inventory.71:T450,Victor ran the pilot on the first dead project: the internal HR tool that had not been touched since 2020. He spent a morning generating smoke tests using an agent (the tool had no tests at all), then ran the migration agent overnight. The next morning he had a PR with 847 file changes. He spent six hours reviewing the PR: spot-checking the most complex transformations, reviewing the agent's custom fixes, running the application locally, testing the core workflows manually.\n\nThe total elapsed time from \"dead project on Java 8\" to \"migration PR ready for merge\" was 18 hours, almost entirely unattended. Victor's hands-on time was the morning of test generation and the six-hour review. He estimates equivalent manual effort at four to five weeks. He is now running the agent on all seven dead projects in parallel. He expects all seven PRs within three days. The most important lesson Victor has documented: the test generation step, which he almost skipped, turned out to be essential - the agent found two genuine behavioral regressions that would have reached production without the smoke tests.72:T452,Victor manages the agent fleet that maintains steady state. His weekly routine has changed completely from L1-L2: instead of firefighting debt-related incidents and manually executing migrations, he reviews agent activity logs, tunes agent configurations, and handles the escalated items that agents could not resolve autonomously. His workload related to technical debt has dropped from 60% of his time to approximately 15%.\n\nThe primary challenge at L5 for Victor is maintaining the agents themselves. The agent configurations, recipe libraries, and task specifications that power the maintenance system are themselves code that ages. New framework versions require new recipes. New debt patterns emerge that the existing agents do not recognize. Victo"])</script>
102<script>self.__next_f.push([1,"r's job has evolved from \"staff engineer who also does debt management\" to \"platform engineer for the debt maintenance system.\" He is now building and maintaining the infrastructure that maintains the codebase, rather than directly maintaining the codebase. This is the characteristic shift of L5: the humans build the systems; the systems do the work.73:T40a,Bob's team adopted Cursor six months ago and everyone has it running in their IDE. Agents are helping with autocomplete and small refactors. But Bob has started hearing stories: one developer accidentally let an agent push to the wrong branch, another ran an agent that made API calls to a staging database because its connection string was in a local `.env` file. Bob is not sure whether these are one-off mistakes or signs of a systemic problem.\n\n**What Bob should do:** Bob should treat these incidents as early signal, not aberrations. They are the natural consequence of running capable agents with no environmental guardrails. Bob should commission a lightweight audit: ask developers to document the three most surprising things an agent did in the last month. The resulting list will reveal the real risk surface and make the case for investing in isolated environments. Bob does not need to stop IDE agents - he needs to start the journey toward sandboxing so that the foundation is in place before a serious incident occurs.74:T43d,Sarah tracks how developers are using AI tools and notices that Cursor usage is high but agent mode (autonomous task execution) is low. Most developers use Cursor for autocomplete and chat, not for multi-step agent tasks. When she asks why, the common answer is \"I don't trust it to not break things.\" That lack of trust is directly connected to the shared-environment model: developers instinctively know that an agent with access to their full environment is risky, so they keep it on a short leash.\n\n**What Sarah should do:** Sarah should recognize that the developer instinct is correct - the shared environment is risky - and invest in the infrastructure that makes it safe to use agents more ambitiously. Isolated environments (L2 and beyond) are not just a security feature; they are a productivity feature that unlocks the trust developers need to let agents run autonomously. Sarah should calculate the opportunity cost: if 20 developers are constraining agent use because of justified safety concerns, fixing the safety concern is worth significant infrastructure investment.75:T405,Victor has been running Claude Code in his IDE for months and has pushed it further than anyone on the team. He uses it for multi-step tasks, lets it run tests, and sometimes runs overnight agent sessions. He is also the person who discovered that the agent once committed a file containing a local API key to a feature branch (caught before push, but still). Victor understands the risks better than anyone because he has hit them directly.\n\n**What Victor should do:** Victor should document his near-misses and share them with the team without waiting for a formal incident review. His operational experience is the most credible data the organization has about what IDE-resident agents actually do in practice. Victor should also prototype a Docker-based agent sandbox - even a simple one that mounts only the project directory and excludes the home directory - to demonstrate that isolation is achievable without breaking the development workflow. That prototype becomes the proof of concept that justifies the L2 investment.76:T454,Victor has been running agents locally and has built up a sophisticated local setup. He is resistant to cloud dev environments because he feels his local setup is faster and more customized than anything a Codespace could offer. He is not wrong about his personal experience, but he also recognizes that his setup is not reproducible by other developers.\n\n**What Victor should do:** Victor should invest a weekend in building the best possible devcontainer configuration for the team's primary repo. This means pre-installing every tool, pre-configuring every exte"])</script>
102<script>self.__next_f.push([1,"nsion, and specifically optimizing for agent workflows: Claude Code pre-installed and authenticated, MCP servers configured, test runners ready to go. When the Codespace cold-start time is under 60 seconds and the environment is better configured than most developers' local setups, the resistance to cloud environments largely disappears. Victor should also document the specific cases where local development is still preferable (e.g., hardware-intensive tasks, specific hardware dependencies) so the policy is nuanced rather than absolutist.77:T412,Bob's team has adopted Docker sandboxing at L2 and it is working well for individual developers. But as the team starts running more parallel agent tasks, they are seeing conflicts: two developers running agents on the same file at the same time, an agent in one context picking up changes from an agent in another context. The shared-environment model is starting to show its limits.\n\n**What Bob should do:** Bob should recognize that the shared-environment conflicts are the forcing function to move to the devbox model. He should assign an infrastructure engineer to design the per-task isolation architecture using the team's existing Docker infrastructure as a starting point. The design does not need to be perfect - a simple per-task container manager with basic credential injection and automatic cleanup is enough to solve the conflict problem. Bob should time-box the design to two weeks and the initial implementation to four weeks, with a goal of running the team's most active agent pipeline in devboxes by the end of the sprint.78:T445,Victor has been running multi-agent workflows using git worktrees for isolation and it works reasonably well for his personal workflow. But he can see the fundamental limitation: worktrees are isolated in terms of the codebase, but the surrounding environment (credentials, network access, running services) is still shared. Two agents running in different worktrees can still interfere at the environment level.\n\n**What Victor should do:** Victor should build a local devbox manager as a weekend project - a simple script that creates a Docker container for each agent task with its own isolated filesystem (not just a separate worktree) and its own credentials. The script should accept a task specification, spin up the container, run the agent, and clean up. Running this for a week will reveal the operational challenges (what happens when a task fails? how do you inspect a running devbox? what are the right resource limits?) that the infrastructure team needs to solve for the org-wide implementation. Victor's operational learnings should feed directly into the infrastructure design.79:T48b,Bob's team has devboxes running but the startup time is 4 minutes. Developers are using them but grumbling about the wait, and Bob knows the team is not getting the full parallelism benefit because the startup cost makes it hard to justify dispatching tasks for short work. Bob has heard about the 10-second benchmark and wants to achieve it but does not know whether it requires a full infrastructure rewrite.\n\n**What Bob should do:** Bob should commission a startup time analysis sprint: one infrastructure engineer spends one week profiling the current initialization sequence and identifying the top three slowest steps. The output is a prioritized optimization plan with estimated effort and expected impact for each step. Most teams will find that the top two steps (dependency installation and codebase cloning) account for 80% of the startup time, and both can be addressed with pre-warming and snapshot techniques. Bob should then allocate a two-week infrastructure sprint specifically for startup time optimization, with the explicit target of under 30 seconds (a major improvement from 4 minutes) as the first milestone, 10 seconds as the stretch goal.7a:T4c5,Sarah has measured that developers dispatch agent tasks less frequently than expected based on the scale of work they are doing. When she follows up, developers consistently say that they factor in the startup time when de"])</script>
102<script>self.__next_f.push([1,"ciding whether to use an agent: \"it is not worth waiting 4 minutes for a task that will take 8 minutes.\" If the startup were under 30 seconds, developers would dispatch many more short tasks to agents. Sarah sees startup time reduction as a direct productivity lever.\n\n**What Sarah should do:** Sarah should estimate the value of the suppressed agent tasks. If developers are passing on 3 agent tasks per day because the startup cost is too high, and each of those tasks would take 15 minutes manually, that is 45 minutes of daily agent leverage not being captured. For a team of 20 developers, that is 900 person-minutes of productivity per day. The math will justify a significant infrastructure investment in startup time reduction. Sarah should use this calculation to fund the infrastructure work and set a success metric: after startup time drops below 30 seconds, measure whether the agent task dispatch rate increases. The before/after comparison becomes the ROI evidence for the investment.7b:T452,Victor has been tracking Stripe's engineering blog and read the Minions post about 10-second devbox spin-up. He has replicated several of the techniques in his personal setup: a pre-warmed Docker container with the codebase at HEAD and dependencies installed, achieving ~15-second startup times. He is confident that 10 seconds is achievable with Firecracker snapshots and a proper pool manager.\n\n**What Victor should do:** Victor should propose a concrete architecture for the team's path to 10-second devbox startup: switch from Docker cold-start to Docker pre-warmed pools as an immediate improvement (30-second startup), then implement Firecracker with snapshots as the next step (10-second target). Victor should write the proposal as a one-page architecture document with estimated implementation effort for each phase. The proposal should include the expected startup times at each phase, grounded in his own prototype measurements. Victor should also identify which team members have the skills to implement Firecracker integration (it requires Linux systems knowledge) and propose a staffing plan.7c:T447,Bob's organization has reached the scale where agent workloads are competing with CI and application workloads for shared compute. CI times are getting longer, agent tasks are getting slower, and the operations team is complaining about resource contention they cannot explain. Bob needs to make the case for dedicated agent compute but is having trouble quantifying the business case.\n\n**What Bob should do:** Bob should instrument the shared compute to measure resource contention attributable to agent workloads. The metrics to collect: CI queue time before and after major agent usage spikes, application latency during peak agent hours, agent task completion time variance (high variance indicates resource contention). A one-week measurement period that captures a range of agent load conditions will produce the data Bob needs. If the contention is real and measurable, the dedicated compute business case writes itself: dedicated agent compute costs X per month, it eliminates Y hours of CI delay per month across the organization, and it removes Z% of application performance incidents.7d:T46f,Sarah has been tracking agent task throughput and has noticed that performance degrades during peak hours - afternoons when most of the engineering team is online and running agents simultaneously. The per-task completion time is 30-40% longer during peak hours than off-peak hours. This degradation is directly impacting developer productivity: developers are running agents off-peak (evenings, early mornings) to get reasonable performance, which is not a sustainable workflow.\n\n**What Sarah should do:** Sarah should use the peak-hour performance data to make the case for dedicated compute. The calculation is: if X developer-hours per week are shifted to off-peak to work around peak contention, and dedicated compute would eliminate that contention, the value of those developer-hours is the ROI of the infrastructure investment. Sarah should also survey developers "])</script>
102<script>self.__next_f.push([1,"directly about how often they choose not to dispatch an agent task during peak hours because of expected performance. The subjective opportunity cost - agent tasks not run because of expected slowness - is likely larger than the objective performance degradation.7e:T481,Victor has been watching the agent infrastructure evolve and is ahead of the curve on the dedicated compute question. He has been advocating for dedicated agent compute for three months based on his own performance analysis. He has data showing that disk IOPS are the binding constraint at the current shared infrastructure scale: when more than 20 agents run concurrently on the shared nodes, disk wait time accounts for 40% of agent task duration.\n\n**What Victor should do:** Victor should present his disk IOPS analysis as the primary technical argument for dedicated compute with NVMe-optimized instances. He should also prototype a Kubernetes node pool configuration for agent workloads: node type selection (i3en or similar), taint/toleration configuration, resource limits for agent pods, and a simple Prometheus dashboard for fleet-level disk IOPS monitoring. The prototype does not need to be production-ready; it needs to demonstrate that the solution is feasible and the approach is sound. Victor should estimate the cost of the dedicated fleet (cloud pricing is public) and model the cost per developer-hour of agent productivity it enables.7f:T46a,Bob's team has been using AI tools for six months and sees mixed results. Some developers swear by Copilot; others think the AI is more trouble than it's worth. Bob suspects the variance is about tool quality but hasn't diagnosed the root cause. His team is almost certainly at Zero MCP without knowing it.\n\n**What Bob should do:** Bob should run a structured evaluation to identify where his team sits on the MCP maturity scale. The fastest way is to ask one senior developer to spend two hours trying to answer five specific codebase questions using only their current AI tools - no copy-paste, no manual context loading. Document what the agent gets right, what it gets wrong, and what it simply cannot answer without context. This exercise almost always reveals that the team is at Zero MCP for anything beyond the current file. That concrete evidence is the business case for the first MCP investment. Bob doesn't need to understand MCP deeply; he needs to understand that his team is operating AI tools with no codebase context and that fixing this is a one-time infrastru
102cture investment with permanent throughput benefits.80:T44a,Sarah tracks developer productivity metrics and is frustrated that AI tool adoption hasn't moved the needle on ticket throughput the way she expected. Her developers use AI tools, but the context-loading overhead - copying error messages, pasting schemas, explaining the codebase structure - eats most of the time they'd otherwise save.\n\n**What Sarah should do:** Sarah should quantify the context-loading tax. For one week, ask five developers to log every time they manually copy something into an AI chat window: error messages, stack traces, documentation, schema definitions, ticket text. At the end of the week, count the instances and estimate the time. This number will be surprising. The context-loading tax is typically 20-40% of total AI tool usage time at Zero MCP. That number is your ROI calculation for the first MCP server: if one MCP server eliminates 50 context-paste operations per day across the team, and each paste takes two minutes, that's 100 minutes per day recovered. Sarah should use this calculation to justify a one-sprint investment in setting up the first MCP server.81:T4e2,Victor has been experimenting with Claude Code and Cursor and knows there's a better setup available. He's read about MCP servers and wants to implement them, but he's working alone on this initiative and isn't sure where to start or how to get organizational buy-in.\n\n**What Victor should do:** Victor should build a proof of concept that is impossible to ignore. Pick the single most painful context-loading operation "])</script>
102<script>self.__next_f.push([1,"his team does every day - probably something like \"paste the full error stack trace into the chat\" or \"explain this internal API to the agent.\" Build one MCP server that eliminates that specific operation. Make it work end-to-end: the agent automatically has the context, gives a better answer, and saves two minutes per occurrence. Demo it at the next team meeting with a live before/after comparison. One well-chosen MCP server demo, where the agent visibly knows something it couldn't know without the connection, converts skeptics faster than any presentation. Victor should then document the server as a template: a README with setup instructions, a configuration example, and a clear explanation of what the server exposes. This template is what makes the jump from Zero MCP to 1-3 MCP servers achievable for the whole team.82:T403,Victor already has Git and docs MCP servers running locally. He configured them by hand, knows the edge cases, and has tuned the server settings to return focused data. Now he needs to help the rest of the team get to the same setup without requiring each person to repeat his experimentation.\n\n**What Victor should do:** Victor should build a team setup script that configures all three servers in a single command. The script should: check for required credentials, write the MCP server configuration to the correct location for the team's primary AI client, and run a smoke test that verifies each server is returning data. This setup script reduces the deployment time for each additional team member from two hours to fifteen minutes. Victor should also write the tool descriptions carefully - the natural language description of each tool that the agent uses to decide when to call it. Good tool descriptions are the difference between an agent that reliably uses MCP tools and one that ignores them in favor of guessing.83:T40a,Bob is spending an hour per week on MCP-related support: token expiry problems, developers who can't replicate the correct setup, and onboarding new team members to AI tools. He knows manual MCP setup doesn't scale but has been deferring the platform investment because it feels like infrastructure overhead.\n\n**What Bob should do:** Bob should calculate the cumulative cost of manual MCP management. One hour per week of his own time plus one hour per quarter per developer for setup and maintenance, across a 15-person team, adds up to more than 15 hours per quarter of MCP overhead. A one-sprint investment in a basic centralized platform - configuration repository, centralized secrets, deployment script - eliminates most of that overhead permanently. Bob should present this as an infrastructure efficiency project to the team, not as an AI initiative. The output is faster developer onboarding, fewer support tickets, and better AI tool consistency. These are engineering infrastructure outcomes that stand on their own merits.84:T419,Sarah has been tracking AI tool adoption and sees that new team members take 4-6 weeks to reach the same AI tool effectiveness as senior developers. Part of this gap is skill, but part is setup: new developers don't have the full MCP configuration that experienced developers have accumulated over months.\n\n**What Sarah should do:** Sarah should use the onboarding gap as the primary justification for the centralized platform. If a platform reduces the time for new developers to reach full AI tool effectiveness from 6 weeks to 2 weeks, and the team hires 8 developers per year, that's 32 developer-weeks of productivity recovered annually. Sarah should calculate this number and present it to Bob and the engineering leadership as the ROI case for the platform investment. She should also define \"full AI tool effectiveness\" concretely: not a vague quality judgement, but a specific list of configured MCP servers that each developer should have. The platform's job is to close the gap from \"just onboarded\" to \"full configuration\" automatically.85:T40b,Victor has been the de facto MCP platform administrator - the person everyone asks when their MCP setup breaks. He's spendin"])</script>
102<script>self.__next_f.push([1,"g 2-3 hours per week on this support role and it's taking time away from product engineering. He knows what a proper platform would look like but hasn't been given the time to build it.\n\n**What Victor should do:** Victor should document the cost of his current informal support role and present it as a project justification. The setup: \"I spend 2-3 hours per week on MCP support for the team. A centralized platform would reduce this to near zero. Here's my proposal for what to build in one sprint.\" Victor already knows what the platform needs because he's been the informal support layer for it. He can build the first version in one sprint if given the explicit allocation. He should scope the v1 narrowly: centralized credentials, a shared configuration repository, and a setup script. That's enough to eliminate 80% of the current support overhead and create the foundation for RBAC and governance at L4.86:T44f,Bob's organization has accumulated 20+ MCP servers over two years of L2 and L3 investment. Each team built their own server; some are well-maintained, others are broken and abandoned. The credential management is a mess, the operational overhead is real, and nobody has a complete picture of what's actually running. Bob needs to rationalize this before adding more.\n\n**What Bob should do:** Bob should run a MCP infrastructure audit as a prerequisite to Toolshed adoption. The audit: inventory all running MCP servers, test each one, document what it exposes, and assign ownership. Remove or decommission servers that aren't working or aren't owned. This audit will reveal both the current state and the requirements for the Toolshed gateway. Bob should then fund a platform team sprint to build a basic Toolshed gateway that aggregates the surviving servers. The Toolshed doesn't need to have 400 tools on day one; it just needs to be the canonical single endpoint that agents connect to, with the organizational commitment that new tools are added to the Toolshed, not deployed as standalone servers.87:T436,Sarah is onboarding new teams to AI agent workflows and finds that the current multi-server setup creates a significant learning curve. Each team needs to know which servers exist, how to configure them, and which tools are in which server. The cognitive overhead of the multi-server model is slowing adoption.\n\n**What Sarah should do:** Sarah should champion the Toolshed model as an adoption enabler, not just an infrastructure improvement. The key insight for Sarah: a single endpoint with 400 discoverable tools and a browsable catalog is dramatically easier to onboard to than 20 separate servers with undocumented tools. Sarah should drive the tool catalog UI as a priority deliverable alongside the gateway: a web page where any developer can see every tool available to agents, with examples of when to use each. This catalog is both the developer documentation and the agent's tool discovery mechanism. Sarah should measure onboarding time before and after the Toolshed + catalog: the reduction in \"how do I give an agent access to X?\" questions is the adoption impact.88:T488,Victor is already connecting agents to many tools and finding the multi-server complexity limiting. He wants to build more sophisticated agent workflows that use 10-15 tools in a single session, but the current setup requires connecting to multiple MCP servers, managing multiple authentication contexts, and handling failures from each independently.\n\n**What Victor should do:** Victor should prototype the Toolshed gateway as a personal project first. Build a simple proxy that aggregates his current MCP servers, implement tool namespace conventions, and connect his most sophisticated agent workflows through it. The goal is to validate that the single-endpoint model simplifies his complex workflows before proposing it as an organizational investment. Victor should document what he built, the complexity reduction he observed, and the specific agent capabilities it unlocked. A working prototype that Victor can demo - an agent that calls 15 tools in sequence through a sin"])</script>
102<script>self.__next_f.push([1,"gle endpoint, versus the same workflow requiring 5 separate MCP server connections - is the most effective argument for the organizational investment in a proper Toolshed platform.89:T49a,Bob's team currently responds to production incidents by receiving an alert, manually gathering context from multiple monitoring tools, opening an agent session with that context, and asking the agent to help diagnose the issue. This process takes 15-30 minutes before the agent is even contributing to the investigation. Bob wants to cut that time dramatically.\n\n**What Bob should do:** Bob should fund a proof-of-concept for event-driven agent incident response. The target: when a production alert fires, an agent automatically receives the full alert context via MCP notification and begins a structured investigation - querying logs, checking recent deployments, identifying anomaly patterns - before a human has even acknowledged the alert. The agent's investigation summary is ready when the on-call engineer opens their incident channel. The PoC doesn't need to resolve incidents autonomously; it just needs to do the first 10-15 minutes of investigation automatically. This time savings compounds across every incident, and the structured investigation summaries improve the quality of human incident response even when the agent doesn't resolve the issue autonomously.8a:T432,Sarah wants to measure the impact of moving from reactive to proactive operations - agents that respond to events rather than waiting to be invoked. She needs a before/after framework that captures the productivity and reliability improvements.\n\n**What Sarah should do:** Sarah should instrument the current state before building the bidirectional infrastructure. Measure: median time from alert firing to agent session started (currently 15-30 minutes for most teams), median time from agent session started to root cause identified, and rate of incidents resolved without escalation. After deploying event-driven agent workflows, measure the same metrics. The expected outcome: time to agent involvement drops from 15+ minutes to under 1 minute; root cause identification time improves because the agent starts with structured context rather than assembled context; resolution rate without escalation increases because the agent has more time and better context. These three metrics make the case for the bidirectional MCP investment to engineering leadership and finance.8b:T4ca,Victor has built agent workflows that are triggered manually - he invokes an agent and provides context. He wants to make his most common agent workflows event-driven: automatically triggered when specific conditions occur, without requiring him to notice the condition and manually start a session.\n\n**What Victor should do:** Victor should pick one high-frequency agent workflow he currently triggers manually and build an event-driven version. The ideal candidate: a workflow he triggers multiple times per day in response to a specific, identifiable event (a test failure, a deployment completion, an error rate spike). He builds an MCP server that subscribes to the relevant event source, translates events into MCP notifications, and triggers his standard agent workflow automatically. He runs this in parallel with his manual workflow for two weeks: does the automated version trigger correctly? Does it produce equivalent results? After validation, he removes the manual trigger. Victor then documents this pattern as the template for event-driven agent development - how to identify automatable trigger events, how to build the notification MCP server, and how to validate that automated triggers are well-calibrated.8c:T411,Bob's team completed the Bazel migration 3 months ago and is running with a remote cache. He's looking at the metrics: average agent incremental build time is 8 seconds, but there are outliers at 2-3 minutes. These outliers happen when agents change shared proto files or base utility libraries. Bob wants to understand the pattern and reduce the outlier frequency.\n\n**What Bob should do:** Bob should have"])</script>
102<script>self.__next_f.push([1," his infrastructure team run a \"change type vs. rebuild scope\" analysis: for each type of file that agents commonly change (service code, shared libraries, proto files, test utilities), measure the average number of targets affected and the average build time. This analysis will identify the high-impact change types. For those types - especially proto files - Bob should fund targeted refactoring work to reduce the downstream dependency fan-out. Splitting a large proto file into smaller domain-specific protos can reduce a \"3-minute outlier build\" to a \"15-second normal build\" by reducing the affected target set from 2,000 to 50.8d:T430,Victor has set up automated affected-target analysis as part of his agent workflow. Before submitting any agent-generated branch to CI, a pre-CI step runs `bazel query 'rdeps(//..., //path/to:changed_targets)'` and reports the affected target count. If it's over 500, the agent adds a comment to the PR explaining the wide impact and requests human review before CI runs. This prevents agents from accidentally submitting wide-impact changes that saturate CI.\n\n**What Victor should do:** Victor should contribute this affected-target analysis tool as a team standard. It should run automatically as a GitHub Actions check on every PR - both human and agent-generated. PRs that affect over 500 targets get a \"wide impact\" label and a comment explaining what they're changing. This creates organizational awareness of change scope and makes wide-impact refactoring a deliberate decision rather than an accident. Victor should also explore whether the 500-target threshold is calibrated correctly for the team's CI capacity, and adjust it based on observed CI queue behavior.8e:T437,Bob's team runs 50 agent CI iterations per day and 30 human pre-merge CI runs. Each CI run takes 12 minutes. His CI cost is substantial. His infrastructure lead has proposed creating an agent-specific build profile that would cut agent CI time to 90 seconds. Bob is concerned that a reduced CI profile might miss bugs that the full pipeline catches.\n\n**What Bob should do:** Bob should approve a 2-week experiment: run the proposed agent profile in parallel with the full pipeline for all agent branches. Compare results: how many agent builds pass the agent profile but fail the full pipeline? The expected failure rate is 3-7% for a well-designed agent profile - these are typically integration test failures unrelated to the specific change the agent made. Bob should set an explicit acceptance criteria: if the agent profile's false-positive rate (passes agent profile, fails full pipeline for a reason relevant to the agent's change) is under 5%, the profile is acceptable. This experiment provides the data Bob needs to make an informed decision rather than a gut-feel one.8f:T437,Victor has been running a custom agent build profile for 3 months. He has three profile levels: \"quick\" (compilation only, 8 seconds), \"iteration\" (compilation + changed-target unit tests, 90 seconds), and \"full\" (complete pipeline, 12 minutes). He's instrumented his agent workflow to automatically choose the right profile: \"quick\" for every push, \"iteration\" every 5 pushes or when the agent signals it wants test feedback, \"full\" before marking a PR ready for review.\n\n**What Victor should do:** Victor should codify this three-tier profile system as a team standard. The key insight - that different stages of agent work need different feedback types - is generalizable. He should write the CI configuration so the three profiles are available to all agents and documented in the team's agent workflow guide. Victor should also track which agents use which profiles and how often: if agents rarely use the \"quick\" profile, it's not providing value; if agents frequently run \"full\" profiles during iteration (not just pre-merge), they're over-testing and wasting CI capacity.90:T448,Bob's team has been on the L4/L5 build journey for 18 months. Build times are consistently fast for most work, but he occasionally hears complaints about slow builds when teams work on shared infrast"])</script>
102<script>self.__next_f.push([1,"ructure. He's asked whether the team has truly achieved \"commodity\" status.\n\n**What Bob should do:** Bob should commission a 30-day build performance audit that tracks p50, p95, and p99 build times for agent iteration builds, segmented by target area. The audit should identify whether the tail events are randomly distributed or consistently concentrated in specific areas of the codebase. If they're concentrated - say, proto changes and base library changes account for 80% of tail events - those specific areas need targeted investment (fine-grained proto structure, stable/volatile library splitting). Bob should set a formal commodity status criterion: \"p99 agent iteration build time under 30 seconds for 30 consecutive days\" and track it monthly. When that criterion is met and sustained, formally declare commodity status and redirect the infrastructure investment to the next bottleneck.91:T407,Sarah's DevEx metrics show that \"build time\" has dropped out of the top 5 developer friction points for the first time. It's been replaced by \"PR review turnaround\" and \"context switching between agent sessions.\" This is exactly the productivity evolution she was targeting - build infrastructure has been solved, and new, higher-level friction points have surfaced.\n\n**What Sarah should do:** Sarah should document the build journey as a case study: what investments were made, in what order, over what timeline, and what the measurable outcomes were at each stage. This documentation serves two purposes: it justifies the infrastructure investment to leadership, and it provides a roadmap for other teams in the organization that are at earlier maturity levels. Sarah should also redirect her optimization focus to the new top friction points - PR review turnaround and agent session context switching are now the limiting factors on agent productivity, and they're both addressable with process and tooling investments at L4/L5.92:T444,Victor's build infrastructure is running at what he considers commodity status: p99 under 15 seconds, flat scaling up to 30 concurrent agents, 97% cache hit rate. He's run out of obvious build optimizations and is now focused on the next frontier: making agent sessions themselves more efficient, since build time is no longer the constraint.\n\n**What Victor should do:** Victor should document the full technical architecture of the build infrastructure as a reference implementation. The architecture document should cover: Bazel BUILD file conventions, remote execution cluster configuration, cache tier design, agent build profile definitions, disk I/O optimization choices, and the monitoring stack. This document should be detailed enough that any senior engineer could replicate it from scratch. Victor should also contribute to the open-source Bazel ecosystem: the configurations, tools, and learnings from achieving commodity build status at this scale are valuable to the broader community, and open-source contributions build Victor's reputation as a technical leader in this space.93:T410,Bob's team has production issues that take hours to diagnose because log data is scattered across individual server instances and nobody has consistent access. Post-mortems reference \"couldn't find the relevant logs\" as a recurring theme. Bob knows the situation is bad but isn't sure how to frame the investment.\n\n**What Bob should do:** Bob should frame the logging upgrade not as an infrastructure project but as an incident response cost reduction. Track the time spent in the next three production incidents on pure log retrieval and correlation - the time developers spend SSHing into servers, grepping files, and trying to piece together what happened. That time cost, multiplied by developer hourly rate and incident frequency, is the ROI case for centralized structured logging. Bob should approve a two-sprint initiative: sprint one to push all logs to a centralized system (even unstructured), sprint two to move to structured JSON. This is a low-risk, high-return investment that unblocks every future observability improvement.94:T421,Sara"])</script>
102<script>self.__next_f.push([1,"h sees developers spending significant time on production debugging and wants to reduce the friction. She also knows the team plans to introduce AI agents for incident investigation, but the current logging state makes that impossible - agents have no queryable interface to production data.\n\n**What Sarah should do:** Sarah should treat centralized logging as a prerequisite for any AI agent investment in observability. She should work with the team to instrument one service end-to-end as a reference implementation: structured JSON logs, pushed to a central system, with a dashboard showing error rates. Then document the before/after debugging experience - how long did it take to diagnose the last three incidents with the old approach versus with centralized logs? That concrete comparison is the most persuasive argument for the rest of the team. Sarah should also note that every future observability investment (OTel, alerting, agent investigation) requires this foundation, so the upgrade pays dividends across every subsequent maturity step.95:T40b,Victor wants to give AI agents access to production signals so they can investigate anomalies autonomously. He knows that the current basic logging setup is the primary blocker - there is no structured, queryable interface for agents to consume.\n\n**What Victor should do:** Victor should design the logging architecture with agent consumption in mind from the start. This means choosing a log aggregation platform that exposes a query API (Datadog Logs API, Elasticsearch REST API, Loki HTTP API), not just a human-readable UI. The agent's investigation path starts with a query like \"show me all ERROR-level logs from the payment service in the 10 minutes before the alert fired\" - and that query must be answerable programmatically. Victor should also push for log correlation IDs (trace IDs) as part of the basic logging setup, because without them, even structured logs from multiple services cannot be linked to a single request. The correlation ID is the foundation of distributed tracing, and adding it now costs almost nothing.96:T497,Bob's team has centralized logging but it is still unstructured text. Post-mortems show that even with logs available, investigation takes a long time because finding relevant entries requires pattern-matching against free-form text. The team is also being asked to provide audit trails for compliance, and the current logs do not have consistent enough fields to serve as reliable audit records.\n\n**What Bob should do:** Bob should frame structured logging as both an operational efficiency investment and a compliance requirement. The compliance angle provides budget justification that the operational angle alone may not. Bob should commission a two-week migration project per service, starting with the highest-risk services (payment, authentication, data mutations). Each service migration follows the same playbook: adopt the structured logging library, define the field schema, validate in the aggregation platform, update alerting to use field-based queries. Bob should also make structured logging a standard criterion in the team's definition of done for new services: no service goes to production without structured JSON logs flowing to the aggregation platform.97:T42f,Victor wants to build an agent that can investigate production incidents by querying the log system, correlating errors with deployments, and proposing root causes. He knows this requires structured logs with consistent fields and a query API. His current setup has neither.\n\n**What Victor should do:** Victor should design the log schema with agent consumption as the primary use case. Agents need to ask questions like: \"What errors occurred in the payment service in the 10 minutes before the alert?\" and \"Which user IDs were affected by this error?\" These questions require specific fields (`service`, `level`, `timestamp`, `user_id`) with consistent types and names. Victor should also select a log aggregation platform based on its API quality, not just its UI quality. Grafana Loki's HTTP API, Datadog'"])</script>
102<script>self.__next_f.push([1,"s Log Query API, and Elasticsearch's REST API are all suitable for agent queries; a platform with a beautiful UI but no programmatic API is a dead end for agent-assisted investigation. Victor should prototype the agent query path before committing to a platform.98:T482,Bob has invested in structured logging and basic tracing at L2, but the tools are fragmented: engineers use different dashboards, switch between Sentry, Grafana, and CloudWatch during incidents, and lose time translating between different query languages and mental models. He wants to unify the stack and establish SLOs as the team's reliability language.\n\n**What Bob should do:** Bob should make the unified observability stack a Q1 infrastructure investment. The business case is incident resolution time and SLO accountability. Bob should commission two workstreams in parallel: one team consolidates telemetry into the Grafana stack (Prometheus, Loki, Tempo), another defines SLOs for all customer-facing services and presents them to product leadership. The SLO definition exercise is not purely technical - product and business stakeholders need to agree on what reliability means. Getting that agreement early creates the organizational foundation for the error budget policy. Bob should also ensure that the observability stack is designed from day one to support agent queries, since the long-term goal is agent-assisted incident investigation.99:T424,Sarah's developers spend the first 15 minutes of every incident figuring out which tool has the relevant data and how to query it. The fragmented tool stack is a developer experience problem that compounds under incident stress. She wants to reduce the tool-switching overhead and create a single starting point for all incident investigation.\n\n**What Sarah should do:** Sarah should work with the team to designate Grafana as the single starting point for all production investigation. This means Grafana dashboards that link out to other tools, not standalone tools that link to Grafana occasionally. Every alert notification should include a direct link to a Grafana dashboard pre-filtered to the relevant service and time window. Every runbook should reference Grafana dashboards for its investigation steps. Sarah should also track \"time to first relevant data\" during incidents as a developer experience metric: from alert receipt to seeing the relevant metric/trace/log, how long does it take? A unified stack should reduce this from minutes to seconds.9a:T4af,Victor wants agents to be able to query the observability stack using all three telemetry types. He knows that Prometheus exposes PromQL, Loki exposes LogQL, and Tempo exposes TraceQL - all via HTTP APIs. He wants to build MCP tools that wrap these APIs so agents can query any pillar of the observability stack programmatically.\n\n**What Victor should do:** Victor should build a `observability-query` MCP server with four tools: `query_metrics(promql, time_range)`, `query_logs(logql, time_range)`, `query_traces(traceid)`, and `find_traces(service, time_range, filters)`. These tools expose the full Grafana stack to agents without requiring the agent to know the underlying query language syntax. When an alert fires, an agent can call `find_traces(service=\"payment\", time_range=\"last_10m\", error=true)` to retrieve the relevant traces, then `query_logs(logql='{service=\"payment\"} |= \"error\"', time_range=\"last_10m\")` to find the corresponding log lines. The agent synthesizes this data into a preliminary root cause analysis before any human engages. Victor should run this as a live demo with the team, showing an incident investigation that takes an agent 30 seconds versus a human 20 minutes.9b:T450,Bob's on-call engineers are burning out. The team's services are mature and incidents are infrequent, but each incident requires deep investigation that can take hours and happens at unpredictable times. He wants to reduce the on-call burden without reducing service quality.\n\n**What Bob should do:** Bob should position the agent investigation pipeline as an on-call load reduction invest"])</script>
102<script>self.__next_f.push([1,"ment. The goal is to shift the on-call engineer's role from investigator to decision-maker: the agent does the investigation, the human makes the call on remediation. This shift reduces the cognitive and time burden of on-call, making rotation more sustainable and accessible to more team members. Bob should measure on-call experience with a simple monthly survey: mean pages per week, mean investigation time per incident, subjective experience rating. These metrics should improve as the agent pipeline matures. Bob should also ensure the pipeline has a clear human override: on-call engineers should always be able to bypass agent recommendations and investigate manually. The agent is a tool, not an authority.9c:T473,Victor is building the investigation agent pipeline. He has the observability stack (metrics, traces, logs via MCP), the incident history MCP, and the runbook MCP. He is now designing the agent's investigation protocol and the ticket writing format.\n\n**What Victor should do:** Victor should start with a narrow, high-frequency incident type for the first production deployment of the investigation agent. Pick the alert that fires most often, has the most consistent root causes, and has the most detailed runbook. Build an investigation agent specifically for this alert type, test it against the last 20 historical incidents (were the root causes it would have identified correct?), and deploy it to production for that alert type only. This narrow deployment validates the pipeline with real incidents before expanding to all alert types. Victor should also design the investigation protocol as data (a JSON schema) rather than code, so that runbook owners can update the investigation steps without modifying the agent code. The protocol schema is the bridge between human-written runbooks and agent-executed investigation procedures.9d:T4f1,Bob's team is operating a complex microservices system with dozens of services. He has a small SRE team that is stretched thin handling ongoing reliability work, and a backlog of performance optimization work that never gets prioritized over product features. He wants to change the economics: move reliability and optimization from human-executed work to agent-executed work.\n\n**What Bob should do:** Bob should make the transition to the full production-agent loop a 6-month strategic initiative, not a sprint project. The prerequisites need to be validated in months 1-2 (observability stack audit, false positive rate measurement, investigation pipeline accuracy). Policy framework and governance model in months 2-3. First loop deployment on optimization-only work in months 3-4. Gradual expansion to remediation work in months 4-6, with weekly review. Bob should also plan for the organizational change that accompanies this shift: SRE team members whose time is freed from repetitive operational work need a clear direction for that reclaimed time. The answer is: higher-level reliability engineering, loop governance, policy tuning, and the novel incident types that the loop cannot yet handle. The loop should augment the SRE team, not make them redundant.9e:T4cd,Victor is the technical architect of the full production-agent loop. He has built the component pieces at lower maturity levels and is now integrating them into a coherent, self-monitoring system. His biggest concern is reliability: a loop that breaks silently is worse than no loop.\n\n**What Victor should do:** Victor should treat the production-agent loop as a production system with its own SLOs. The loop should have defined reliability targets: 99% of work items that enter the queue should be resolved or escalated within 24 hours, the deployment-and-verify cycle should have a 95% success rate (5% rollback rate is the acceptable ceiling), and the loop health dashboard should be reviewed daily. Victor should also build the loop's self-monitoring carefully: the observability stack monitors production services, but who monitors the observability stack? Victor should implement a separate, simple heartbeat monitoring system for the loop infrast"])</script>
102<script>self.__next_f.push([1,"ructure itself - if the OTel collector stops receiving data, or the anomaly detection system stops generating alerts, or the agent fleet stops processing work items, the heartbeat monitor pages a human immediately. The loop's most dangerous failure mode is silent degradation."])</script>
102<script>self.__next_f.push([1,"7:[\"$\",\"$1\",\"c\",{\"children\":[[\"$\",\"main\",null,{\"className\":\"min-h-screen\",\"children\":[[\"$\",\"$L11\",null,{}],[\"$\",\"$L12\",null,{\"perspectives\":[{\"slug\":\"development\",\"title\":\"Development\",\"areas\":[{\"name\":\"Coding Agent Usage\",\"description\":\"How your team uses AI coding assistants - from autocomplete to autonomous agent fleets.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Copilot autocomplete\",\"summary\":\"How to use IDE autocomplete as your first step into AI-assisted development.\"},{\"text\":\"Chat in sidebar, ad-hoc questions\",\"summary\":\"How to use the AI chat panel for one-off code questions and explanations before any systematic workflow exists.\"},{\"text\":\"Agent runs without codebase context\",\"summary\":\"Understanding why AI tools at L1 see only what you show them - and why that's the core limitation the entire maturity journey addresses.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Agent in IDE; autonomy set by a written rule, not a per-prompt click - and increasingly by the organisation rather than the developer (GitHub's enterprise-managed Copilot agent permissions for shell, files and network domains cannot be overridden by users)\",\"summary\":\"How to run an IDE agent whose autonomy is set by a written, version-controlled permission ruleset rather than a per-prompt click, so multi-file work runs without constant confirmation interrupts.\"},{\"text\":\"An agent instruction file ships with every active repository\",\"summary\":\"An agent instruction file ships with every active repository, teaching AI tools that project's conventions, patterns and constraints - the single highest-leverage action at L2.\"},{\"text\":\"Copilot + Claude Code in parallel\",\"summary\":\"How to use inline autocomplete and an agentic CLI tool simultaneously, each at the granularity it handles best.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Code is written to be read by agents: explicit over implicit, obvious over clever\",\"summary\":\"Code is written to be read by agents: explicit over implicit, obvious over clever, with the conventions that govern it stated precisely rather than absorbed by osmosis.\"},{\"text\":\"Rules files per-team/per-repo\",\"summary\":\"How to evolve from a single project-level CLAUDE.md to a layered system of context files tailored to each team's tech stack and conventions.\"},{\"text\":\"CLI agents as primary (Claude Code with Opus 5.5 or Sonnet 5.5, Codex on GPT-6 Sol, Cursor, Gemini CLI) with cheap and open-weight workers (GPT-6 Luna, DeepSeek V4.1 Flash, MiMo-V2.6, GLM-5.3, Kimi K3) for execution\",\"summary\":\"How shifting from IDE plugins to CLI-based agents makes AI a programmable, scriptable part of your development workflow rather than a typing assistant.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Scheduled / unattended agents + three-tier routing: a System One decision model decides (TypeSafe Jev - routing, compaction, safety gates at milliseconds), a cheap model executes, the frontier plans; re-costed monthly per task, not per token (Copilot Auto tiers, Uber's cheap subagents, Stripe Minions)\",\"summary\":\"How to launch AI agents that run to completion autonomously in a sandbox - writing code, running tests, fixing errors and opening a PR without supervision - under a run-status taxonomy and a model-routing policy you re-cost on a schedule.\"},{\"text\":\"Slack/CLI/Web/PagerDuty invocation â PR\",\"summary\":\"How to trigger AI agent tasks from natural language interfaces - a Slack message, a CLI command, a web form - and have the agent autonomously produce a pull request.\"},{\"text\":\"3-5 parallel agents per developer + merge queues for agent fleets; the ceiling is orchestrator context pollution, not token cost (cap batches at 2-4, no concurrent repo-wide git ops)\",\"summary\":\"How to shift from sequential AI assistance to managing multiple concurrent agent instances - transforming the developer's role from implementer to orchestrator, with the ceiling set by orchestrator context pollution rather than by token cost.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Multi-agent orchestration (Claude Code dynamic workflows - the Bun-in-Rust model, Gas Town / custom)\",\"summary\":\"How to build systems where specialized agents collaborate - a planner decomposes tasks, workers execute them, and reviewers validate results - to handle complex engineering tasks end-to-end.\"},{\"text\":\"Planner â Worker hierarchy\",\"summary\":\"How to structure a two-tier agent architecture where a planner decomposes engineering tasks and workers execute them in parallel - the canonical L5 pattern for complex autonomous development.\"},{\"text\":\"Fleet size bounded by compute and review capacity, not by tooling\",\"summary\":\"The frontier of AI-assisted development: massive agent parallelization where hundreds of concurrent agents produce thousands of commits per hour on a single codebase.\"}]}]},{\"name\":\"Context Engineering\",\"description\":\"What information agents receive about your codebase, architecture, and conventions.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Agent works from the currently open file\",\"summary\":\"Why AI agents at L1 are flying blind - they see only the open file, with no awareness of your project's architecture, conventions, or dependencies.\"},{\"text\":\"Project knowledge lives with people, not docs\",\"summary\":\"At L1, the rules governing your codebase live in developers' memories - invisible to AI agents, fragile under turnover, and impossible to scale.\"},{\"text\":\"Onboarding leans on the existing README and people\",\"summary\":\"Stale documentation is more than a developer experience problem - it actively prevents AI agents from using docs as context and signals that the team hasn't invested in machine-readable knowledge.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"CLAUDE.md pruned to repo-specific gotchas and committed to the repo (repos without committed agent config saw twice the cognitive-complexity growth: +53% vs +27%)\",\"summary\":\"A lean, repo-specific CLAUDE.md committed to version control is the single highest-leverage context engineering investment - it gives every AI agent the minimum viable information about what your project is and how it works.\"},{\"text\":\"Written coding conventions\",\"summary\":\"Documenting your team's semantic coding decisions - not just style rules - gives AI agents the judgment framework they need to suggest code that fits your architecture, not just code that c
102ompiles.\"},{\"text\":\"Agent instruction files + Skills as the unit of reuse (SKILL.md alongside CLAUDE.md, AGENTS.md, llms.txt; cross-agent registries)\",\"summary\":\"Agent instruction files - CLAUDE.md, .cursorrules, copilot-instructions.md - have become a standard software project artifact, with 60,000+ repositories on GitHub already containing them.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Architecture, ownership and operational context reach the agent through governed servers rather than pasted text\",\"summary\":\"Model Context Protocol servers are the infrastructure layer of context engineering at L3 - dedicated services that give AI agents structured, real-time access to your organization's knowledge.\"},{\"text\":\"Retrieval is deterministic and cheap: the agent searches and extracts on demand instead of being fed a pre-built index\",\"summary\":\"Retrieval is deterministic and cheap: the agent searches and extracts what it needs on demand, rather than being fed a pre-built index that was assembled before anyone knew what the task was.\"},{\"text\":\"Context budgeting with a standing eviction rule and a hard cap (Uber: 400K tokens with auto-compaction); agent files hand-written and audited, not generated - machine-generated context files did worse than none at 20%+ more cost; and any security rule in CLAUDE.md is backed by a technical control (only 4.4% are)\",\"summary\":\"AI models have finite context windows - context budgeting is the practice of deliberately allocating that budget across context types, and evicting from it on a schedule, to maximize agent effectiveness per token spent.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"BYOC: org PUSHES context to the agent\",\"summary\":\"At L4, organizations flip the context model - instead of agents pulling context on demand, the org pre-assembles and pushes a rich context package to agents at task start, eliminating discovery latency and ensuring consistency.\"},{\"text\":\"A queryable map of code structure, ownership and change history\",\"summary\":\"Semantic knowledge graphs of codebase
102s - built by tools like Graph Buddy and CodeTale - give AI agents structural understanding of your codebase without requiring them to read every file.\"},{\"text\":\"Spec-Driven Development: AGENTS.md as shared spec interface (Spec Kit constitution.md, Andrew Ng + JetBrains, Thoughtworks)\",\"summary\":\"At L4, AI agents transform Jira or Linear tickets into structured specifications and failing acceptance tests before a developer touches the implementation - creating a context artifact that guides the entire downstream workflow.\"}]},{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Persistent agent identity + memory (Beads/Git, Open Memory Protocol; treat auto-memory as an exfiltration surface - \\\"Memory Heist\\\")\",\"summary\":\"At L5, agents maintain persistent identity and memory across sessions using structured memory files committed to git - accumulating codebase-specific institutional knowledge that makes them progressively more effective over time.\"},{\"text\":\"Production telemetry â context auto-update\",\"summary\":\"At L5, agent context updates automatically based on production signals - when a service degrades, agents working on related code receive updated operational context without manual intervention.\"},{\"text\":\"Stale context is detected and refreshed before an agent runs on it\",\"summary\":\"Context is verified against reality at the moment it is about to be used: stale material is detected and refreshed before the agent runs on it, rather than after the mistake it caused.\"}]}]},{\"name\":\"Code Review \u0026 Quality\",\"description\":\"How AI-generated code is reviewed, validated, and approved before merging. Model upgrades do not buy security: the average GenAI security pass rate held flat year on year at 56%, and coding-specialised models scored no better than general-purpose ones.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Every PR gets human review\",\"summary\":\"Why relying entirely on human code review creates a quality bottleneck that scales poorly and establishes the baseline every higher maturity level is designed to escape.\"},{\"text\":\"Review turnaround measured in hours; 78.9% of agentic PRs pass through a single reviewer\",\"summary\":\"The 2-hour average wait for code review feedback is a measurable symptom of L1's structural review problem - and a concrete baseline to improve against.\"},{\"text\":\"AI and human code share one review path\",\"summary\":\"At L1, teams can't see how much of their codebase is AI-generated - making it impossible to measure adoption, calibrate review depth, or understand quality patterns.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"AI-assisted review suggestions (CodeRabbit, Qodo, Claude Security beta)\",\"summary\":\"Using AI tools to generate a first-pass review frees human reviewers from routine checks and focuses their attention on architecture and business logic.\"},{\"text\":\"Basic linter rules\",\"summary\":\"A shared linter configuration enforced in CI is the fastest, cheapest quality investment a team can make - and the essential foundation for every higher-level quality automation.\"},{\"text\":\"Diff awareness - reviewer knows it's AI code; reject code the human can't understand even if CI is green; humans write the PR description (the why)\",\"summary\":\"When reviewers know which parts of a PR are AI-generated, they can calibrate review depth to match the actual risk - spending more time on business logic and less on syntax.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Lint-as-architecture (standards = enforced rules; Vercel Konsistent for agents and humans)\",\"summary\":\"Custom lint rules that enforce architectural decisions turn code review comments into machine-checked constraints - so architectural violations are caught at CI time, not weeks later in production.\"},{\"text\":\"AI review agent as first pass (self-verification; adversarial verification pass on whole-repo scans)\",\"summary\":\"A dedicated AI review agent that automatically reviews every PR before human reviewers are notified transforms review from a bottleneck into a parallel, always-available quality gate.\"},{\"text\":\"Architecture guardrails: Bug â Codify â Lint Rule\",\"summary\":\"A systematic process for converting recurring bugs into permanent lint rules turns each incident into an improvement to the quality gate, progressively hardening the codebase against known failure modes.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Green/Yellow/Red auto-evaluation\",\"summary\":\"A traffic-light quality evaluation system that replaces binary pass/fail CI with a nuanced, policy-driven assessment - enabling selective automation and focusing human review where it genuinely adds value.\"},{\"text\":\"Classification is fully algorithmic: the same change always lands in the same class\",\"summary\":\"The verdict on a change is produced by algorithm rather than by whoever happened to look at it, and it is reproducible: the same change always lands in the same class.\"},{\"text\":\"Policy-based auto-approval driven by a risk classifier trained on your own incident history (Zalando: 33% of PRs auto-approved as low-risk, 20-40% lead-time reduction) - and every agent PR carries a named human owner, or it does not merge (agentic PRs merge at 79% elite vs 37% fair, and the gap is ownership)\",\"summary\":\"Setting a 60%+ Green rate as policy turns code quality into a measurable team KPI, with a risk classifier trained on your own incident history deciding what qualifies and a named human owner on every agent PR.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Agent fleet self-reviews (error â fix â converge), bounded: a drafting agent never approves its own work, and LLM defect detection degrades across successive review rounds\",\"summary\":\"At fleet scale, agents review and correct their own work in tight feedback loops - running tests, observing failures, fixing root causes and iterating to a passing state - bounded by a separate verifier, because a drafting agent must never approve itself and defect detection decays with each review round.\"},{\"text\":\"Human review only for Red (architectural)\",\"summary\":\"At L5, human engineering attention is reserved exclusively for Red PRs - architectural changes, security-sensitive modifications, and business logic decisions that automated systems can't confidently evaluate.\"},{\"text\":\"Continuous auto-refactoring in background\",\"summary\":\"Background agents that continuously identify and execute code quality improvements - extracting duplication, simplifying complexity, updating deprecated APIs - eliminate technical debt accumulation without dedicated refactoring sprints.\"}]}]},{\"name\":\"Testing Strategy\",\"description\":\"How tests are written, maintained, and validated in an AI-assisted workflow. A test \\\"oracle\\\" is the source of truth for what a test's correct result should be.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Tests written by hand\",\"summary\":\"The baseline testing state at L1 - manual test writing, chronic under-coverage, and the compounding debt that makes AI-generated code increasingly risky to ship.\"},{\"text\":\"Flaky tests are a recurring cost\",\"summary\":\"Flaky tests silently consume a sixth of your engineering capacity - Google's internal research quantified the cost, and at L1 they're accepted as normal instead of eliminated.\"},{\"text\":\"AI tests treat current output as \\\"correct\\\" - no independent oracle for the intended result\",\"summary\":\"When AI generates tests by reading your implementation, it encodes existing behavior as correct - including bugs - giving you coverage numbers that feel like safety but provide none.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Agent-generated unit tests + human acceptance tests\",\"summary\":\"A hybrid testing strategy at L2 that uses AI to generate unit test scaffolding at scale while keeping business-behavior verification in human hands.\"},{\"text\":\"Flaky test quarantine\",\"summary\":\"A systematic L2 process for removing flaky tests from the main CI signal without deleting them - creating accountability to fix them while keeping builds reliable.\"},{\"text\":\"Humans define expected results for key paths (acceptance tests are the oracle)\",\"summary\":\"Fixing flaky tests at the root by replacing fragile implementation-coupled assertions with stable, behavior-level oracles that reliably distinguish real failures from noise.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Expected results come from requirements (tickets/specs are the oracle, not the code); test *process* is not mandated to agents - outcomes are measured instead\",\"summary\":\"TORS quantifies what percentage of test failures are real bugs - at L3, 90%+ is the prerequisite for trusting automated quality gates, with expected results drawn from requirements rather than from what the code already does.\"},{\"text\":\"Acceptance tests from tickets (Autonomous Requirements)\",\"summary\":\"At L3, AI agents read requirements tickets and generate failing acceptance tests before implementation begins - making requirements machine-executable and eliminating circular testing at scale.\"},{\"text\":\"Incremental test selection (only changed paths)\",\"summary\":\"Running only the tests affected by a given code change - using dependency graph analysis - so CI feedback stays fast as the codebase grows to millions of lines.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Held-out oracles the agent never sees gate releases (\\\"Building to the Test\\\": with oracle access agents ship dead code passing all 222 tests); property-based testing + fuzzing over LLM-written tests\",\"summary\":\"At L4, raising the Test Oracle Reliability Score to 95%+ is the prerequisite for trusting automated merge decisions - where 1-in-20 false positives is the maximum the system can tolerate.\"},{\"text\":\"Agent iterates tests to green in sandbox (doesn't block team CI)\",\"summary\":\"AI agents fix failing tests in isolated sandboxes - running their own private CI loop - so agent work-in-progress never pollutes the shared pipeline or slows the team.\"},{\"text\":\"Mutation testing on high-risk paths as the real coverage signal (one component: 100% line coverage, 61% mutation strength)\",\"summary\":\"At L4, mutation testing on high-risk paths is the real coverage signal - it verifies that agent-generated test suites actually catch bugs rather than merely executing code, before those tests enter the shared codebase.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Self-healing test suite\",\"summary\":\"At L5, AI agents continuously maintain the test suite - detecting and fixing flaky tests, updating tests broken by intentional refactors, and generating tests for uncovered paths - so humans set quality policy rather than doing quality work.\"},{\"text\":\"Production logs â auto-generated regression tests\",\"summary\":\"At L5, agents mine production errors to automatically generate regression tests - capturing the exact inputs that caused real failures so they can never reach production again.\"},{\"text\":\"Agent detects edge case â writes test â fixes bug â ships\",\"summary\":\"The fully autonomous quality loop at L5: an agent finds an edge case, writes a failing test, fixes the bug, verifies all tests pass, and submits the PR without any human involvement in the cycle.\"}]}]}]},{\"slug\":\"delivery\",\"title\":\"Delivery Management\",\"areas\":[{\"name\":\"CI/CD Pipeline\",\"description\":\"Speed and reliability of your build-test-feedback loop for AI-generated code.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"CI runs on every change\",\"summary\":\"CI pipelines that take longer than 15 minutes are a defining characteristic of the Assisted maturity level.\"},{\"text\":\"Agent waits for CI feedback\",\"summary\":\"The L1 state where an agent ships its changes and then waits, blind: it cannot read CI, tests or lint output, so a human has to relay every result back.\"},{\"text\":\"Shared runner, queue\",\"summary\":\"The L1 CI default: one fixed pool of runners for everyone, first in first out, so a quick lint check queues behind a full integration suite.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Pipeline definitions live in the repo and are reviewed like application code\",\"summary\":\"Pipeline definitions live in the repository next to the application they build, and changes to them go through the same review, the same checks and the same history as any other change.\"},{\"text\":\"Dedicated runners per team\",\"summary\":\"Dedicated runners per team means each engineering team has its own isolated pool of CI runners, not shared with other teams.\"},{\"text\":\"CI \u003c 10 minutes\",\"summary\":\"CI under 10 minutes is the first meaningful milestone on the path to AI-native delivery infrastructure.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Agent CI treated as internet-facing: no `${{ github.event.* }}` interpolated into `run:`, agent passes split into separate jobs with per-job token scope\",\"summary\":\"Incremental builds are a build strategy where only the changed files, modules, or packages are recompiled and only the tests covering changed code re-run, with each agent given its own worktree pipeline and the agent CI itself hardened as an internet-facing surface.\"},{\"text\":\"Per-worktree pipelines, so parallel agents do not serialise on one runner\",\"summary\":\"Every worktree - every agent, every branch, every parallel line of work - gets its own pipeline instance, so parallel agents do not serialise behind each other on a single shared runner.\"},{\"text\":\"CI \u003c 5 minutes\",\"summary\":\"CI under 5 minutes is the Systematic (L3) milestone where CI speed becomes a first-class engineering concern, not a background project.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"CI as Sandbox: 50 attempts in 5 min without blocking team; merge queues for parallel agent fleets (auto-merge only what builds and passes tests); scheduled / async agents land PRs overnight\",\"summary\":\"\\\"CI as Sandbox\\\" is a configuration pattern where the CI system is intentionally designed to support rapid, high-frequency iteration by AI agents, isolated from the normal developer CI workflow.\"},{\"text\":\"Every CI run for an agent gets its own disposable environment, so a poisoned run cannot reach the next one\",\"summary\":\"Every CI run triggered by an agent executes in its own short-lived, fully isolated environment that is destroyed afterwards, so a poisoned run cannot reach the next one.\"},{\"text\":\"CI \u003c 2 minutes\",\"summary\":\"CI under 2 minutes is the Governed (L4) milestone and represents a qualitative shift in how CI is used.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Sub-minute feedback\",\"summary\":\"Sub-minute CI feedback is the Self-improving (L5) frontier - a pipeline that returns meaningful quality signal to an agent in under 60 seconds.\"},{\"text\":\"Runner capacity auto-scales with agent load, with no manual capacity planning\",\"summary\":\"Runner capacity auto-scales with agent load: the CI system observes its own queue, provisions and releases runners as demand moves, and nobody does manual capacity planning.\"},{\"text\":\"Production feedback â CI auto-adjusts test suite; every production failure becomes a permanent regression test; verification moves before the PR opens and CI checks the evidence instead of re-running it (main-branch success fell to a five-year low of 70.8% across 28M workflows)\",\"summary\":\"The test suite stops being a static artifact: an incident in production generates its own regression test and adds it to CI automatically.\"}]}]},{\"name\":\"Merge \u0026 Deploy\",\"description\":\"How PRs flow from creation to production - throughput, automation, and conflict handling.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Merging is a manual act: someone clicks the button on every change\",\"summary\":\"Nothing moves unless a person moves it: every change reaches main because someone decided the moment had come and clicked the button.\"},{\"text\":\"Merge capacity set by how fast humans can review\",\"summary\":\"Ten PRs per day is the typical throughput ceiling for a manual review-and-merge process on a team of 6-10 developers.\"},{\"text\":\"Manual deploy or simple CD\",\"summary\":\"L1 deployment, from SSH and git pull to a pipeline that fires on merge. Either way there are no gates, no progressive rollout and no automated rollback.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"CD pipeline with gates; agent commits come from a distinct bot identity, never a developer's account\",\"summary\":\"A CD pipeline with gates is a deployment pipeline that has explicit checkpoints between stages.\"},{\"text\":\"Basic merge queues\",\"summary\":\"A merge queue serializes pull requests that are ready to merge, ensuring that each PR is tested against the latest state of the target branch before it actually merges.\"},{\"text\":\"Auto-rebase\",\"summary\":\"PR branches are kept current with the target branch automatically, so nobody runs git rebase main by hand every time another PR lands.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Policy-based merge rules; agents push with short-lived GitHub App or OIDC tokens and never hold deploy secrets; agent output lands as a stack of dependent, independently reviewable branches rather than one 1,000-line PR\",\"summary\":\"Policy-based merge rules replace ad-hoc human judgment about when and how to merge with codified, machine-enforced criteria, and they expect agent output to arrive as a stack of dependent, independently reviewable branches rather than one thousand-line pull request.\"},{\"text\":\"Deterministic ordering + conflict detection (cross-vendor agent PR pairs conflict at 41.7% vs 19.8% intra-vendor - standardize the fleet or serialize the merges)\",\"summary\":\"Deterministic ordering means the merge queue processes PRs in a defined, predictable sequence rather than in arbitrary arrival order.\"},{\"text\":\"A published cap on CI rounds per PR, and the team tracks it\",\"summary\":\"Stripe's efficiency benchmark: a PR should reach green within two CI runs. A third round means the agent is guessing rather than reasoning.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"A green verdict flows straight to production without a second queue or a second approval - while high-impact actions (token creation, production credentials, webhook edits) require a fresh human re-authentication (GitHub proof of presence)\",\"summary\":\"A green verdict flows straight to production: no second queue, no second approval, no waiting for someone to act on a decision that has already been made.\"},{\"text\":\"Throughput well above the pre-agent baseline, with the merge path no longer the constraint\",\"summary\":\"At 50+ PRs a day, reviewing every one by hand stops being possible: the constraint moves to automated policy, merge queues and selective human review.\"},{\"text\":\"Canary/progressive deployment auto; verifiable provenance per change, with AI-assistance disclosure enforced in CI rather than by convention (Linux 7.2 carried 1,111 `Assisted-by` commits, and at least one maintainer strips the tags)\",\"summary\":\"Changes go to a slice of traffic first and metrics decide the rest - automatic promotion or automatic rollback with no human in the loop - and every change carries verifiable provenance, with AI-assistance disclosure enforced by CI rather than left to convention.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Merge volume limited by product decisions, not by the merge path\",\"summary\":\"1000+ merges per week is the throughput level that Stripe's engineering organization achieved with their AI-assisted development program, published as the \\\"Minions\\\" model.\"},{\"text\":\"Agent produces PR â CI passes â merge â deploy â observe (n8n model: release lifecycle fully delegated to bots)\",\"summary\":\"The full L5 delivery loop: an agent implements a spec, opens a PR, CI validates it, the merge queue lands it, CD deploys it and observability watches.\"},{\"text\":\"Rollback is agent-driven\",\"summary\":\"An agent detects the regression, finds the PR that caused it, rolls it back and tells the team, without waiting for a human to make those calls.\"}]}]},{\"name\":\"Metrics\",\"description\":\"What you measure to understand AI-assisted engineering productivity and quality.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Delivery performance measured at all (DORA, SPACE or an equivalent set), if tracked\",\"summary\":\"At L1 (Assisted), most engineering teams track DORA metrics inconsistently or not at all.\"},{\"text\":\"Standard delivery metrics (not yet AI-specific)\",\"summary\":\"At L1, engineering teams that have adopted AI tools - GitHub Copilot, Cursor, Claude Code - are tracking those tools with zero AI-specific metrics.\"},{\"text\":\"ROI of AI not yet measured\",\"summary\":\"\\\"How much did we save with AI?\\\" is the question every engineering leader eventually faces from finance, from the CTO, or from the board.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"A delivery-performance baseline plus basic AI tracking; per-session token spend; input tokens (context), not output, drive spend - watch power users well above the median\",\"summary\":\"At L2 (Delegated), teams have moved past the L1 silence on metrics.\"},{\"text\":\"Licenses vs usage rate - but never token spend or seat activity as an adoption target; both are gamed within weeks, and Meta scrapped its 85k-employee token leaderboard and pulled AI usage from performance reviews\",\"summary\":\"The first uncomfortable AI metric is that 30-50% of the AI coding licenses an organization pays for sit unused in any given month - a diagnostic worth measuring, and a target that is gamed within weeks of being set.\"},{\"text\":\"PR throughput per dev as a proxy, never a target - and never suggestion acceptance rate, of which 31% is deleted within 15 minutes\",\"summary\":\"PRs merged per developer per week is a crude proxy, since PRs vary wildly in size, but it is the first output signal available without real instrumentation - useful as a diagnostic, ruinous as a target, and far better than suggestion acceptance rate.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Cost per iteration (CPI) measured per task, not per token: cheaper models can cost more per task (Gemini 3.8 Flash: same token price, $0.40 -\u003e $0.58 per task), and cheaper tokens made sessions 3.3x longer; include CI compute and the production compute the generated code burns (+5-8%)\",\"summary\":\"Cost-per-Iteration (CPI) measures what a single agent CI attempt costs across three components - model tokens, CI compute, and the compute the generated code goes on to burn in production - with cost per merged PR tracked as a trend.\"},{\"text\":\"Iterations to success (ITS): counted per change, limit set by the team\",\"summary\":\"Iterations-to-Success (ITS) is an AI-native metric that measures how many CI attempts it takes for an agent's PR to pass.\"},{\"text\":\"Push-to-result time: median recorded per reporting period\",\"summary\":\"CI Feedback Latency is the time from when an agent pushes a commit to when CI produces a result (pass or fail) that the agent can act on.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Test-oracle reliability tracked as a metric, alongside model-regression signals (thinking length, files read before edit)\",\"summary\":\"The Test Oracle Reliability Score is the share of CI failures that are real defects rather than flake. Above 95% is what makes auto-merge trustworthy.\"},{\"text\":\"Every agent run terminates in a classified state - success / flawed / blocked / manual - and each non-success class routes to a different fix; only the classification rate justifies expanding the automation boundary\",\"summary\":\"Every agent run terminates in one of four classified states - success, flawed, blocked or manual - only success ships, each of the other three routes to a different fix, and the classification rate is what justifies expanding the automation boundary.\"},{\"text\":\"Agent Autonomy Score: % tasks without human intervention\",\"summary\":\"The share of tasks an agent takes from assignment to merge with no human intervention: no clarifying questions, no course corrections, no manual fixes.\"},{\"text\":\"Merge Queue Wait \u003c 10 min\",\"summary\":\"Merge Queue Wait is the time a PR spends waiting in the merge queue after all gates pass (CI green, reviews approved, policy rules satisfied) before it is actually merged.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Cost-per-feature (not cost-per-PR); CFO scorecard: Useful Work, Cost per Successful Task, Return on Compute - paired with incidents-per-merged-change and firefighting hours, which move in the opposite direction\",\"summary\":\"Cost-per-feature is the total cost - AI compute, CI infrastructure, human review and product management time - to deliver a complete user-facing feature, reported on a CFO scorecard of Useful Work, Cost per Successful Task and Return on Compute, and always paired with incidents-per-merged-change and firefighting hours.\"},{\"text\":\"Business value throughput, not activity metrics\",\"summary\":\"Once agents make engineering activity cheap, PRs merged stops measuring anything useful and revenue, churn and conversion become the primary metrics.\"}]}]},{\"name\":\"Governance \u0026 Compliance\",\"description\":\"Controls around AI-generated code - licensing, security scanning, and audit trails.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Individual devs use their own AI subscriptions\",\"summary\":\"Shadow AI refers to the use of AI tools by developers through personal subscriptions and accounts that operate entirely outside the organization's awareness, approval, or oversight.\"},{\"text\":\"AI usage not yet audited\",\"summary\":\"A zero audit trail state means that when an auditor, security team, or incident investigator asks \\\"what AI systems were involved in producing this code change?\\\" there is no answer.\"},{\"text\":\"AI usage is informal, policy not yet defined\",\"summary\":\"In 2023 and early 2024, many organizations responded to AI coding tools by banning them outright.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Official AI tool policy; per-session spend caps, short-lived keys and kill switches; autonomy set by a declarative deny/ask ruleset that is reviewed and version-controlled, not by a human clicking approve; a deliberate default for new vendor features instead of letting them auto-enable (Copilot's \\\"Default policy for new features\\\")\",\"summary\":\"An official AI tool policy is the organization's first structured governance response to AI in the delivery pipeline, and at L2 its sharpest edge is a declarative deny/ask ruleset - reviewed, version-controlled and enforced before any classifier runs - rather than a human clicking approve.\"},{\"text\":\"Basic audit: who uses what - and every agent has a named human owner, because 51% of organisations cannot say who owns their AI identities\",\"summary\":\"Basic audit at L2 means the organization has established visibility into which developers are using which AI tools, at what frequency, and for what purposes.\"},{\"text\":\"Regulatory obligations for your jurisdiction and sector are identified and owned (EU AI Act Article 50 transparency applicable since Aug 2 2026, high-risk duties deferred to Dec 2027 / Aug 2028; China tiers agents by decision authority); data-residency routing\",\"summary\":\"The regulatory obligations that apply to your jurisdiction and sector are identified and owned by a named person - starting with the EU AI Act, whose Article 50 transparency regime has been applicable since 2 August 2026 while the high-risk duties stand deferred to December 2027 and August 2028.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Minimum viable audit trail: model, timestamp, context, approver - and the agent's own identity, registered as a distinct principal in the IdP (Entra Agent ID, Okta, Google Agent Identity), not a developer's token\",\"summary\":\"The four fields that make an AI-assisted change defensible after the fact: which model, when, what it was asked to do, and who approved the result.\"},{\"text\":\"Policy-as-code; all model traffic through an AI gateway (LiteLLM, Kong AI Gateway 2.0, Agent Router, Claude apps gateway, Cloudflare): central keys with BYOK-only, per-team budgets, model allowlist with exact-version pinning and hard denies (`availableModelsMatch`, `deniedModels`, Codex `requirements.toml`); repository-supplied agent config (`.claude/`, `.vscode/`, `.cursor/`, `.git/config`, `build.rs`) on a mandatory-diff path, because it executes on folder open\",\"summary\":\"Policy-as-code means expressing compliance rules as executable code that runs in the pipeline rather than as documents people are asked to read - now extending to enterprise managed settings for agent clients, an org-wide default model admins can actually enforce, and repository-supplied agent config placed on a mandatory-diff path because it executes the moment a folder is opened.\"},{\"text\":\"Compliance gates in CI\",\"summary\":\"Compliance gates in CI are automated checks that must pass before a pull request can be merged, specifically focused on governance and compliance requirements rather than functional correctness.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Full provenance tracking per change (cryptographic agent traces;
102 commit-to-prompt lineage via gateway prompt IDs; prompt and response logs in the SIEM with retention beyond vendor defaults - Cursor keeps 30 days); agent access certified like human access and revoked when the owner leaves\",\"summary\":\"For every change in production you can reconstruct the whole lineage: the requirement, the ticket, the AI sessions, the review, the CI run, the release.\"},{\"text\":\"Automated compliance checks; the AI gateway run as critical infrastructure with a patch SLA (LiteLLM's MCP auth bypass reached the CISA KEV list), DLP at the gateway, and unapproved agents detectable - 26% of large firms cannot; skills and MCP servers allowlisted and pinned rather than scanned, assessed against OWASP Agentic Skills Top 10\",\"summary\":\"Compliance checks that judge substance rather than paperwork - prohibited patterns, regulatory boundaries crossed, known issues in the model that wrote it - with skills and MCP servers allowlisted and pinned rather than scanned, and assessed against the OWASP Agentic Skills Top 10.\"},{\"text\":\"AI code vs human code distinction in VCS (Kubernetes model: disclosure mandatory, AI commit messages banned; humans write the why)\",\"summary\":\"Version control tags which commits and lines came from AI, so \\\"how much of the payments module was AI-generated?\\\" has a computable answer.\"}]},{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Continuous compliance: agent monitors regulatory changes\",\"summary\":\"An agent tracks regulatory change continuously and maps each one onto the policies, gates and code it affects, instead of just alerting a human.\"},{\"text\":\"Self-documenting audit trail: every agent action traces to one agent identity and one accountable human, and the urgent fast path runs through the same controls (47% of firms with written policies skipped them for urgent deployments)\",\"summary\":\"A self-documenting audit trail is one where the documentation of AI involvement in a change is generated automatically by the AI system itself, without requiring human effort to produce it.\"},{\"text\":\"Enterprise-grade RBAC per agent: task-scoped permissions with no standing access, new agents start on probation and earn scope from their track record\",\"summary\":\"Every agent gets an audited identity of its own, so what it may do is set by its role rather than by the permissions of whoever launched it.\"}]}]}]},{\"slug\":\"organization\",\"title\":\"Organization\",\"areas\":[{\"name\":\"AI Adoption Model\",\"description\":\"How your organization rolls out AI tools - from individual experiments to org-wide strategy.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Adoption via bulk license purchase\",\"summary\":\"The big-bang license purchase is the most common first move in enterprise AI adoption, and the most reliable predictor of failure.\"},{\"text\":\"Initial enthusiasm fades to low usage\",\"summary\":\"Enthusiasm â Silence â Shelfware is the name for the failure arc that almost every unstructured AI tool deployment follows.\"},{\"text\":\"Powerful tools on an unprepared process\",\"summary\":\"The Ferrari Engine in Fiat 126p is a metaphor for a specific and common failure pattern: installing powerful AI capability into an engineering process that cannot take advantage of it.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Pilot teams (2-3 teams)\",\"summary\":\"A structured pilot is the antidote to the big-bang license deployment.\"},{\"text\":\"Adoption spreads through peer networks rather than mandates\",\"summary\":\"Adoption spreads through peer networks rather than mandates: practice travels between people who work together, along the relationships that already exist, and a directive from above does not move it.\"},{\"text\":\"Pilot metrics; track cost per task from day one - cheaper tokens lengthen sessions rather than shrink bills\",\"summary\":\"Pilot metrics are the set of measurements you define before a pilot starts that determine whether the pilot succeeded and whether to expand.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Platform team owns AI tooling: a central proxy or agent registry supplying identity, cost tracking, sandboxing, observability and evals by default - declare an agent once, get production-readiness in minutes rather than weeks\",\"summary\":\"A central proxy or agent registry hands out identity, cost tracking, sandboxing, observability and evals by default, so declaring an agent takes minutes rather than weeks.\"},{\"text\":\"Internal Developer Platform with AI layer, plus enablement run as a named programme with attendance you can count (a standing guild, guided hackathons, hands-on labs) rather than a launch email\",\"summary\":\"The self-service platform teams already deploy through gains an AI layer, and the enablement beside it runs as a named programme with attendance you can count rather than a launch email.\"},{\"text\":\"Standardized agent setup per team, but no mandated tool - the platform is standard, the choice is free, and outcomes are what get measured; a default-model policy set centrally (Ramp: firms restricting frontier use cut spend per employee 9.7% while token prices fell 41%); \\\"bad day protocol\\\" for model and harness regressions, plus a vendor-exit plan\",\"summary\":\"Every team starts from the same agent baseline - context injection, permissions, monitoring - but no tool is mandated: the platform is standard, the choice is free, and outcomes are what get measured.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"AI-assisted work is the default path rather than an initiative, and the org advances by removing its next bottleneck rather than by buying tokens\",\"summary\":\"An AI-first development culture is one where agents are the default approach to development tasks, not an option that some developers use sometimes.\"},{\"text\":\"Agent fleet management as discipline\",\"summary\":\"Once a developer runs several agents at once, start one and check back stops working: agents need scheduling, monitoring, failure handling and quotas.\"},{\"text\":\"Developer supervises agents rather than authoring most changes\",\"summary\":\"In Steve Yegge's model of AI adoption stages, Stages 6 and 7 represent a fundamental shift in what a developer does.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Centralized agent orchestration: scheduling, placement and lifecycle handled by a platform, not per team\",\"summary\":\"A central orchestration layer schedules, scales and recovers agent workloads across the organization, the way Kubernetes does for containers.\"},{\"text\":\"Human-at-the-wheel, not human-in-the-loop\",\"summary\":\"\\\"Human-in-the-loop\\\" describes an approval model where humans review and approve individual agent actions before they execute.\"},{\"text\":\"Organization optimized for agent throughput, not human throughput\",\"summary\":\"Organizations are designed around assumptions about how work gets done.\"}]}]},{\"name\":\"Knowledge Management\",\"description\":\"How institutional knowledge is captured, shared, and made available to both humans and agents.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Processes passed on verbally, not documented\",\"summary\":\"Folk tradition knowledge is the undocumented institutional lore that accumulates around every long-lived codebase.\"},{\"text\":\"Little written documentation\",\"summary\":\"The documentation chicken-and-egg problem is one of the most persistent failure modes in software organizations.\"},{\"text\":\"Key knowledge held by senior staff\",\"summary\":\"The architecture, the history and the dangerous edge cases live in two or three people's heads. Everything looks fine until one of them is unavailable.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Docs refresh initiative\",\"summary\":\"A docs refresh initiative is the structured, time-boxed effort to bring existing documentation back into alignment with reality.\"},{\"text\":\"Architecture Decision Records (ADRs)\",\"summary\":\"Short records of why a technical decision was made: the context, the options considered and the consequences accepted, not just the choice itself.\"},{\"text\":\"Written onboarding paths\",\"summary\":\"A written onboarding path is a structured, step-by-step guide that takes a new engineer from zero access to productive contribution on a specific codebase or team.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Documentation = infrastructure (not an HR problem)\",\"summary\":\"Most engineering organizations treat documentation as a people problem: if engineers wrote better docs, if seniors were more generous with their knowledge, if new hires asked more questions.\"},{\"text\":\"Lint rules \u003e docs (enforced \u003e suggested)\",\"summary\":\"Documentation that says \\\"we use camelCase for variable names\\\" is a suggestion.\"},{\"text\":\"A queryable map of the codebase (structure, ownership, change history); agentic search and plain-text memory over vector databases\",\"summary\":\"The repository becomes a queryable graph of calls, dependencies, owners and tests, so \\\"what breaks if I change this interface?\\\" has a real answer.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Context Fabric: MCP servers feed agents automatically, discoverable and assessable from published metadata before a client ever connects\",\"summary\":\"The MCP servers an organization runs to feed agents context automatically, discoverable and assessable from published metadata before a client ever connects.\"},{\"text\":\"Skills and MCP servers packaged once and installed across clients as vendor-neutral plugins, then treated as maintained assets with a review cadence and an eviction rule - installing a useful skill and keeping it forever are separate decisions\",\"summary\":\"The agent turns a vague ticket into a structured spec, and the skills and MCP servers it runs on are packaged once as vendor-neutral plugins and kept as maintained assets with a review cadence and an eviction rule.\"},{\"text\":\"Docs auto-updated by agents on code change\",\"summary\":\"A code change triggers the doc change: the signature moves and the API reference moves with it, rather than waiting for someone to remember.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Self-evolving knowledge base\",\"summary\":\"A self-evolving knowledge base is a knowledge infrastructure that improves itself without requiring humans to initiate updates.\"},{\"text\":\"The written record is kept current by agents: drift is detected, corrected and validated\",\"summary\":\"The organisation's written record is kept current by agents rather than by anyone's good intentions: drift from reality is detected, corrected, and the correction is validated before it is trusted.\"},{\"text\":\"Organizational memory = Git-backed, agent-readable, always current\",\"summary\":\"The knowledge that used to live in senior engineers' heads becomes Git-backed, agent-readable and current enough that agents can act on it.\"}]}]},{\"name\":\"Team Structure \u0026 Roles\",\"description\":\"How teams are organized and what roles exist to support AI-augmented engineering.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Traditional roles: dev, QA, PM\",\"summary\":\"Dev writes, QA tests, PM specifies. Clear scopes and clean handoffs, and the model that predicts exactly where AI adoption will create pressure.\"},{\"text\":\"Seniors review and fix AI-generated code; human-skill preservation (reject code you can't understand even if it works)\",\"summary\":\"The anti-pattern where juniors generate code faster than they can verify it, and seniors spend their days debugging the results instead of building.\"},{\"text\":\"AI being evaluated in the team's stack\",\"summary\":\"\\\"AI doesn't work in our environment\\\" is the most common organizational statement that blocks progress at L1.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"A named AI champion per team, with time actually allocated to the role\",\"summary\":\"Each team has one named AI champion with time actually allocated to the role, rather than an unfunded expectation resting on whoever cared most.\"},{\"text\":\"Context engineer role (initial)\",\"summary\":\"Making the codebase legible to agents, through CLAUDE.md files, MCP integrations and machine-readable docs, becomes somebody's actual job.\"},{\"text\":\"Training: how to instruct confidently and then verify confidently, which is not the same as reading every line\",\"summary\":\"Training people to instruct confidently and then verify confidently, which is not the same skill as reading every line the agent wrote.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Review shifts up the lifecycle: judgment relocates rather than disappears - problem selection, architecture, the quality bar, which signals to trust, and shipping authority stay human even when authorship does not\",\"summary\":\"When most code in a PR is agent-generated, judgment does not disappear - it relocates to problem selection, architecture, the quality bar, which signals to trust, and shipping authority.\"},{\"text\":\"Platform Engineer (AI tooling); Harness Engineer as the consolidated named skill - the harness, not the model, is the asset that survives a vendor swap\",\"summary\":\"The Platform Engineer specializing in AI tooling owns the harness - the asset that survives a vendor swap, where the model does not.\"},{\"text\":\"Context Engineer = full role (now mainstream - dedicated job postings across industry)\",\"summary\":\"At L3, context engineering graduates from \\\"something the champion does in their spare time\\\" to a full-time engineering role with its own scope, career path, and organizational standing.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Span of control = how many agents you can effectively supervise; the binding limit is the orchestrator's context, not tokens - batches capped at 2-4, status polling restricted, overlapping file ownership read as a signal to consolidate\",\"summary\":\"How many agents one developer can actually supervise at once. The binding limit is the orchestrator's context, not the token bill, which caps a batch at two to four.\"},{\"text\":\"Developer = manager of agent fleet (now a product default: Cursor Run Mode, Claude agent view, Antigravity, MultiDevin)\",\"summary\":\"At L4, the developer's primary job is not to write code - it's to manage a fleet of AI agents that write code.\"},{\"text\":\"Each running agent has a visible health state, so a stalled or drifting one is noticed without being hunted for\",\"summary\":\"Steve Yegge's \\\"keep your Tamagotchi alive\\\" framing captures a crucial insight about working with AI agents at L4: they are not fire-and-forget automations.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Agentic Engineer: orchestration + supervision + architecture\",\"summary\":\"The Agentic Engineer is the L5 role that emerges when AI agents become the primary development modality and the human's job is to architect, orchestrate, and supervise rather than implement.\"},{\"text\":\"PEV loop: Plan â Execute â Verify\",\"summary\":\"The PEV loop - Plan, Execute, Verify - is the fundamental operating model for working with AI agents at high maturity.\"},{\"text\":\"Non-coder contributors via agent interfaces\",\"summary\":\"Product managers, designers and domain experts direct agents to change the software directly, without writing code or waiting on an engineer.\"}]}]},{\"name\":\"Tech Debt \u0026 Modernization\",\"description\":\"How AI accelerates paying down tech debt and modernizing legacy systems.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Tech debt accumulates\",\"summary\":\"Debt grows is the default state of software engineering in organizations that have not made debt management a deliberate practice.\"},{\"text\":\"Legacy code left untouched\",\"summary\":\"\\\"Legacy = don't touch it\\\" is the fear-based approach to old or poorly understood code that characterizes L1 organizations.\"},{\"text\":\"Multi-year migration backlog\",\"summary\":\"Known modernization work has piled up to multiple years at current velocity, which in practice means it will never be finished by human effort alone.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Debt categorized and prioritized; separate Disposable Software (deliberate throwaway) from durable systems - and track comprehension debt, the debt that accumulates silently where machines verify machines\",\"summary\":\"Debt categorized and prioritized is the L2 state where an organization has moved from informal awareness of technical debt to a structured system for tracking, classifying, and ordering debt items.\"},{\"text\":\"Manual migration attempts\",\"summary\":\"A developer works through the new API by hand, fixing compilation errors one at a time. Real progress over L1 paralysis, and far too slow to clear the backlog.\"},{\"text\":\"OpenRewrite basic recipes\",\"summary\":\"Structured, tested refactorings for JVM code: a recipe replaces a deprecated API or applies a security fix across every usage, deterministically.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Continuous Modernization: agent pays off debt in the background, and the payoff is now measurable in tokens - one refactor cut the input cost of every future change to that code by 83%\",\"summary\":\"Tech debt stops being a project that competes with features: an agent works the inventory continuously, and the payoff is now measurable in tokens rather than argued for on principle.\"},{\"text\":\"Library bumps, version upgrades auto\",\"summary\":\"Dependency upgrades are delegated to a system that watches upstream releases, tests the bump and opens the PR, or merges it when confidence is high.\"},{\"text\":\"OpenRewrite + agent = systematic refactoring\",\"summary\":\"OpenRewrite supplies the safe transformations, an agent chooses and sequences the recipes, reads the test failures and handles what the recipes miss.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"\\\"Dead project too expensive to modernize\\\" â agent modernizes for pennies (sqlite-utils 4.0 for $149; reverse engineering that never penciled out now does)\",\"summary\":\"The software an organization wrote off as unmaintainable becomes economic again once the modernization cost is agent time rather than engineer months.\"},{\"text\":\"Cross-repo migration agents: the permanently-deferred migration is now a two-week job, so the backlog is a choice rather than a constraint\",\"summary\":\"One consistent migration applied across dozens or hundreds of repositories at once - the permanently-deferred migration is now a two-week job, so the backlog is a choice rather than a constraint.\"},{\"text\":\"Java 8 â 21, Angular.js â Angular 17 via agents; the Bun model for full rewrites (conformance test suite first, then 64 concurrent agents)\",\"summary\":\"Java 8 to Java 21 and AngularJS to Angular 17 are two of the most common large-scale migration challenges in enterprise softw
102are engineering.\"}]},{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Tech debt = near-zero steady state\",\"summary\":\"Tech debt near-zero steady state is the L5 condition where technical debt does not accumulate over time because the rate of debt remediation equals or exceeds the rate of debt creation.\"},{\"text\":\"Agent fleet maintains, upgrades, patches 24/7\",\"summary\":\"An agent fleet that maintains, upgrades, and patches 24/7 is the L5 state where codebase maintenance is operationalized as infrastructure rather than treated as engineering work.\"},{\"text\":\"CVE remediation: detect â fix â test â ship autonomous\",\"summary\":\"A vulnerability announcement triggers the whole pipeline: find the affected repos, bump to the patched release, run the tests, ship, with no human involved.\"}]}]}]},{\"slug\":\"infrastructure\",\"title\":\"Infrastructure\",\"areas\":[{\"name\":\"Agent Runtime \u0026 Sandboxing\",\"description\":\"Where and how AI agents execute code - isolation, security, and resource management.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Agent in developer's IDE\",\"summary\":\"At the earliest stage of AI-assisted development, the agent lives inside the developer's IDE - literally running as an extension or plugin within VS Code, Cursor, or JetBrains.\"},{\"text\":\"Agent runs in the developer's local environment\",\"summary\":\"The L1 default: the agent runs straight in the developer's environment, with the same filesystem, environment variables, network and credentials.\"},{\"text\":\"Agent access is coarse-grained (all or none)\",\"summary\":\"The L1 permission model, and it has only two settings: the agent can reach everything the developer can, or so little that it cannot do useful work.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Dedicated dev environments\",\"summary\":\"Dedicated dev environments move agent execution off the developer's laptop and into isolated cloud-hosted workspaces.\"},{\"text\":\"Basic sandboxing (Docker, bubblewrap, eBPF directory confinement), and untrusted repositories opened with auto-run hooks disabled - repo-supplied agent, editor and git config (`core.fsmonitor`, GitSpawn) executes on folder open, before any install step\",\"summary\":\"Basic Docker sandboxing wraps the agent's execution environment in a container that is isolated from the host system, and untrusted repositories are opened with auto-run hooks disabled, because repo-supplied agent, editor and git config executes on folder open.\"},{\"text\":\"Agent credentials scoped per project and short-lived, never a personal PAT or a shared long-lived key; spend caps and baseline alerts on every provider account - the first mass agent-run campaign found Claude, Cursor and Gemini tokens on 5,871 machines and burned $600k of credits through one dashboard\",\"summary\":\"Agents stop borrowing the developer's personal tokens and get their own short-lived per-project credentials, scoped to the one repository or bucket they need, with a spend cap and a baseline alert on every provider account.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Isolated agent environments (devbox model); credentials injected at run time from a vault or broker, never in prompts, files or eval sandboxes - an injected evaluation sandbox handed over production keys for several providers\",\"summary\":\"The devbox model is the architectural pattern where each agent task gets its own isolated environment, created at task start and destroyed at task end.\"},{\"text\":\"Pre-warmed containers with codebase\",\"summary\":\"Pre-warmed containers are agent environments that have been prepared in advance and are waiting in a ready state before any task is assigned to them.\"},{\"text\":\"Network isolation with egress denied by default and destinations allowlisted; the evaluation and test environment is inside the security boundary, not outside it; audit everything the agent WRITES, because escapes work by planting files that trusted host tools later read\",\"summary\":\"Egress is denied by default and destinations are allowlisted, so the agent reaches GitHub, registries and staging but not production - and the evaluation environment sits inside that boundary, not outside it.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Ephemeral devboxes spin up fast enough that the agent never waits on the environment\",\"summary\":\"The 10-second devbox spin-up is the performance target that Stripe's agent infrastru
102cture team set as the benchmark for production-grade agent environments.\"},{\"text\":\"Pre-loaded services, code, MCP tools\",\"summary\":\"The devbox starts with everything already running: dependent services, MCP servers, observability and test infrastructure, not just the codebase.\"},{\"text\":\"MicroVM, hardware-isolated execution as default (kubernetes-sigs agent-sandbox standard, AWS Lambda MicroVMs, Docker Cloud Sandboxes, Microsoft MXC); assume escape, including over DNS: workload identity per agent (SPIFFE, WIF, Google Agent Identity) with task-scoped tokens that expire in minutes + cryptographic run provenance\",\"summary\":\"seccomp, AppArmor and eBPF decide what an agent process may do at the syscall level, whatever the agent or its container configuration believes.\"}]},{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Agent fleet on dedicated compute, on a harness whose parts are replaceable - model adapter, tool registry, session log and the agent loop itself swappable without a rewrite\",\"summary\":\"Agent workloads move off laptops and CI runners onto a compute layer of their own, running a harness whose model adapter, tool registry, session log and agent loop can each be swapped without a rewrite.\"},{\"text\":\"Agent execution environments scale with demand, independently of CI runner capacity\",\"summary\":\"The environments agents actually run in scale up and down with demand on their own signals, independently of CI runner capacity, without manual intervention.\"},{\"text\":\"Each agent = isolated machine or managed-agent-on-your-hardware (Devin Outposts model); sender-constrained tokens (DPoP, mTLS) and every session of one agent revocable in seconds\",\"summary\":\"The fleet architecture choice at scale: one machine per agent for strong isolation, or shared machines with resource management for density and cost.\"}]}]},{\"name\":\"MCP \u0026 Tool Integration\",\"description\":\"How agents connect to external tools, APIs, and internal systems via MCP (now universal standard) and plugins.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Agent uses built-in tools only\",\"summary\":\"Zero MCP is the baseline state: your AI agent has no programmatic connection to any tool, system, or data source outside its training data.\"},{\"text\":\"Agent relies on public / general knowledge\",\"summary\":\"The agent answers from its training data, so it knows what the Stripe API looks like in general but nothing about how your integration actually works.\"},{\"text\":\"Integrations done by copy-paste\",\"summary\":\"The universal first way of giving an agent context: paste the error, the schema, the ticket into the chat. When the conversation ends, so does the context.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"1-3 basic MCP servers (Git, Jira, docs)\",\"summary\":\"The first practical MCP deployment replaces the most expensive copy-paste operations with programmatic connections.\"},{\"text\":\"Manual MCP setup per developer\",\"summary\":\"Manual MCP setup per developer is the phase where MCP servers exist and work, but each developer is responsible for installing and configuring them independently.\"},{\"text\":\"Basic tool authorization; the stateless MCP core is shipping, so servers drop sticky sessions and run serverless - and Roots, Sampling and Logging are on a 12-month clock\",\"summary\":\"Basic tool authorization is the first deliberate access control layer on MCP tool usage: a version-controlled ruleset saying which agents can call which tools, under what conditions, written down rather than clicked through.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"MCP platform: centralized server management\",\"summary\":\"A centralized MCP platform moves server configuration, deployment, and credential management from individual developer machines to organization-managed infrastructure.\"},{\"text\":\"Servers for architecture, ownership and SLA data are run as products: versioned, owned and monitored\",\"summary\":\"The servers exposing architecture, ownership and SLA data are operated as products rather than as scripts: versioned, owned by a named team, and monitored like anything else in production.\"},{\"text\":\"RBAC per MCP tool through an MCP gateway that shows each caller only the tools it may use (Kong MCP bundling); clients registered via Client ID Metadata Documents, tokens bound to one server; lazy tool-loading cuts tokens and live attack surface\",\"summary\":\"Role-Based Access Control per MCP tool means defining precisely which agents can call which tools, based on the agent's role, the task it's performing and the data it's operating on - enforced through an MCP gateway that shows each caller only the tools it may use, with clients registered via Client ID Metadata Documents and tokens bound to one server.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"The organisation's tool surface reachable through one governed MCP gateway, with access granted centrally by the IdP (Okta Cross App Access / ID-JAG, MCP Enterprise-Managed Authorization) instead of per-user consent sprawl, and a kill switch that revokes an agent's live tokens at the gateway\",\"summary\":\"The Toolshed model, pioneered by Stripe, consolidates hundreds of distinct tools behind a single MCP endpoint.\"},{\"text\":\"Agent discovery: agent knows what tools are available\",\"summary\":\"Agent discovery is the capability for an agent to dynamically enumerate what tools are available in its current environment and adapt its behavior accordingly.\"},{\"text\":\"MCP governance: every server on a lifecycle from intake to deprecation with a per-server pause switch; tool metadata watched for change at runtime, not only at install (Deadbugz rewrote its tool descriptions after three calls); plugins pinned by full commit SHA, never a branch name (Plugin4Shell); injection resistance tested in the IDE configuration you actually ship\",\"summary\":\"MCP servers stop being configuration and become production services: owners, changelogs, versioned APIs, a lifecycle from intake to deprecation with a per-server pause switch, tool metadata watched for change at runtime, plugins pinned by full commit SHA, and audit logs.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"MCP as nervous system: bidirectional context flow\",\"summary\":\"At L5, MCP is no longer just a protocol for giving agents access to tools - it is the real-time information backbone that connects every part of the software delivery system.\"},{\"text\":\"Production â MCP â Agent â Code â Deploy â Production\",\"summary\":\"The closed loop: production detects a condition, MCP carries it to an agent, the agent changes the code, CI/CD ships it and production verifies the effect.\"},{\"text\":\"Agent-to-Agent Protocol (A2A) + MCP combined\",\"summary\":\"MCP connects agents to tools; A2A connects agents to each other. Together they are the infrastructure layer a multi-agent system runs on.\"}]}]},{\"name\":\"Build System\",\"description\":\"Build tooling optimized for agent-scale throughput - caching, incrementality, and speed.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Maven/Gradle default config\",\"summary\":\"Maven and Gradle ship with sensible defaults for single-developer, sequential workflows.\"},{\"text\":\"Full rebuild on every change\",\"summary\":\"A full rebuild recompiles every source file and re-runs every build step from scratch on each invocation, regardless of what changed.\"},{\"text\":\"Each build recomputes the full graph from source, on every machine\",\"summary\":\"Builds share nothing: every developer's machine and every CI run recomputes the same artifacts from source, over and over, because each cache dies with the process that produced it.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Basic build caching; packages an agent installs from docs or llms.txt are verified against a registry allowlist first (237 of 8,565 llms.txt files pointed at dead, typo'd or unregistered packages)\",\"summary\":\"Basic build caching stores the outputs of build steps and reuses them when the inputs haven't changed, and packages an agent installs from docs or llms.txt are checked against a registry allowlist before they enter the cache.\"},{\"text\":\"Parallel build steps\",\"summary\":\"Parallel build steps execute independent stages of the build and test pipeline concurrently rather than sequentially.\"},{\"text\":\"The build cache is shared between developers and CI, not rebuilt per machine\",\"summary\":\"One build cache serves everybody: an artifact computed once - on a developer's laptop or on a CI runner - is reused by every later build that needs it, instead of being recomputed per machine.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Incremental builds: only changed targets\",\"summary\":\"Incremental builds with only changed targets rebuild exactly and only the build targets that depend on files that have changed since the last build.\"},{\"text\":\"Remote execution distributing build steps across machines\",\"summary\":\"Remote execution distributes build actions across a cluster of machines rather than running them locally.\"},{\"text\":\"A build tool with an explicit dependency graph and a shared cache (Bazel, Buck2, Pants, Nx, Turborepo)\",\"summary\":\"Bazel, Buck2, and Pants are hermetic build systems originally developed by Google, Meta, and Toolchain respectively to handle the scale and correctness requirements of massive monorepos.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Agent-specific build profiles; multi-root workspaces and worktree isolation per agent, now shipped as a default by agents themselves rather than assembled by hand\",\"summary\":\"Agent-specific build profiles are lightweight build configurations optimized for the agent iteration use case rather than the human pre-merge or release use case, and they now have to assume every agent is building in its own git worktree.\"},{\"text\":\"Build system aware of agent iteration patterns\",\"summary\":\"A build system aware of agent iteration patterns goes beyond passive responsiveness - it actively anticipates what agents will need to build next and prepares accordingly.\"},{\"text\":\"Sub-2min feedback on any change\",\"summary\":\"Sub-2-minute feedback means that for any change an agent or developer makes to the codebase, the signal \\\"this compiles and the relevant tests pass\\\" arrives within 120 seconds.\"}]}
102,{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Build = commodity (near-instant for agents)\",\"summary\":\"When build time is a commodity, it has ceased to be a meaningful variable in agent throughput calculations.\"},{\"text\":\"Compilation bottleneck eliminated via crate/module architecture (the Bun lesson: a conformance test suite makes even a runtime rewrite verifiable)\",\"summary\":\"Compilation bottleneck elimination through crate/module architecture means restructuring a codebase so that the unit of compilation is small, focused, and independently compilable.\"},{\"text\":\"Disk I/O optimized for concurrent agent workloads (Cursor lesson)\",\"summary\":\"Disk I/O is the hidden bottleneck when running hundreds of concurrent agents on a shared infrastructure.\"}]}]},{\"name\":\"Observability \u0026 Feedback Loop\",\"description\":\"Monitoring agent behavior, costs, and outcomes to close the improvement loop.\",\"levels\":[{\"level\":1,\"name\":\"Assisted\",\"items\":[{\"text\":\"Basic logging\",\"summary\":\"The most primitive form of production visibility: unstructured print statements scattered wherever somebody once needed to debug something.\"},{\"text\":\"Alerting on errors\",\"summary\":\"Alerting on errors is the practice of automatically notifying a human when something goes wrong in production - before a customer reports it.\"},{\"text\":\"Prod feedback and token-cost visibility not yet wired to dev\",\"summary\":\"\\\"No connection: prod to dev feedback\\\" describes the state where production incidents have no automatic path back to the developer or agent that caused them.\"}]},{\"level\":2,\"name\":\"Delegated\",\"items\":[{\"text\":\"Structured logging\",\"summary\":\"Structured logging replaces free-form text log output with machine-parseable records - typically JSON - where every field has a defined name and type.\"},{\"text\":\"OpenTelemetry basic\",\"summary\":\"OpenTelemetry (OTel) is the open standard for collecting and exporting telemetry data - traces, metrics, and logs - from distributed systems.\"},{\"text\":\"Post-deploy monitoring; per-session token cost as table stakes; input tokens (context) drive spend - track them, not output\",\"summary\":\"The window right after a deploy becomes its own monitoring phase, with tighter thresholds on error rates and latency, instead of ship and move on.\"}]},{\"level\":3,\"name\":\"Systematic\",\"items\":[{\"text\":\"Full observability stack (OTel + Grafana)\",\"summary\":\"A full observability stack means having all three telemetry pillars - metrics, traces, and logs - collected, correlated, and queryable in a unified system.\"},{\"text\":\"Production metrics â dashboards; agent telemetry through a governed gateway (self-hosted control plane: identity, policy, telemetry - Claude Apps Gateway model; catch shadow AI via proxy)\",\"summary\":\"Production metrics dashboards are the operational nerve center of a mature engineering team: real-time, continuously updated views into the health and behavior of every production service.\"},{\"text\":\"Incident data available for context; usage-truth reconciliation (client-reported tokens vs agent-claimed work); every tool call traced to the agent identity, the human it acts for and the token used\",\"summary\":\"Past incidents, runbooks and metric baselines are reachable programmatically, rather than buried in Confluence pages, Slack threads and people's memories.\"}]},{\"level\":4,\"name\":\"Governed\",\"items\":[{\"text\":\"Production anomaly â auto-ticket â agent investigation; incident rate and firefighting hours tracked against change volume, because the two move in opposite directions\",\"summary\":\"The production anomaly to auto-ticket to agent investigation pipeline automates the first phase of incident response, and tracks incident rate and firefighting hours against change volume, because the two move in opposite directions.\"},{\"text\":\"Self-healing basic: known patterns auto-fixed, with diagnosis kept human; anomalous agent behaviour (scope drift, credential reuse, unusual egress) revokes the agent's credentials automatically - a training-sandbox escape in September got past partial monitoring because the automatic shutdown failed\",\"summary\":\"Self-healing for known patterns means specific, well-understood failure conditions are remediated automatically, with diagnosis kept human, and anomalous agent behaviour revokes the agent's own credentials automatically.\"}
102,{\"text\":\"Infrastructure recommends code changes back to the team, and agent sessions are audited for quality regression over time\",\"summary\":\"Infrastructure stops merely running the code and starts reading it: it watches production behavior and hands back concrete code changes to make.\"}]},{\"level\":5,\"name\":\"Self-improving\",\"items\":[{\"text\":\"Full production â agent loop\",\"summary\":\"The full production-to-agent loop is the L5 realization of observability as an agent input channel.\"},{\"text\":\"Anomaly â investigate â fix â test â deploy autonomous\",\"summary\":\"End-to-end incident response with nobody in the loop: the anomaly fires, an agent finds the cause, writes the fix, tests it and ships it behind a canary.\"},{\"text\":\"Infrastructure self-drives: code defines infra, production informs code\",\"summary\":\"\\\"Infrastructure self-drives\\\" describes the fully realized bidirectional relationship between code and infrastructure at L5.\"}]}]}]}],\"gates\":{\"development\":{\"Coding Agent Usage\":{\"1\":{\"must\":[\"At least one AI coding assistant (Copilot, Cursor, Claude Code) is installed and active for at least one developer\",\"AI autocomplete or chat is used at least once per week by the team\"],\"should\":[\"Developers have access to AI chat in their IDE sidebar\",\"Team has experimented with AI-assisted code generation on non-critical tasks\"],\"prerequisites\":[],\"evidence\":[\"IDE plugin install count or license allocation records\",\"Git history showing AI-assisted commits (Copilot attribution tags or similar)\"]},\"2\":{\"must\":[\"Agents operate in multi-step agentic mode (edits without per-step approval)\",\"At least one agentic IDE (Cursor, Windsurf, or Claude Code) is used by 50%+ of the team\",\"CLAUDE.md, .cursorrules, or equivalent agent instruction file exists in 100% of active repositories\"],\"should\":[\"Developers use two or more AI tools in parallel (e.g., Copilot + Claude Code)\",\"Agent instruction files are reviewed and updated at least quarterly\",\"Agent autonomy is bounded by a written permission ruleset (deny/ask) committed to the repo, not by per-prompt clicking\"],\"prerequisites\":[],\"evidence\":[\"Agent instruction files committed in repository root\",\"IDE telemetry or license dashboard showing agentic mode usage\",\"PR descriptions referencing agent-assisted development\"]},\"3\":{\"must\":[\"Coding conventions are written as explicit, agent-parseable rules (not implicit tribal knowledge)\",\"Per-team or per-repo rules files exist and are maintained with code review\",\"CLI agents (Claude Code, Codex) are the primary coding interface for 50%+ of feature work\"],\"should\":[\"Agent usage is tracked per developer and per repository\",\"Agent instruction files follow a standardized template across the organization\"],\"prerequisites\":[{\"label\":\"Development L2 (Context Engineering) - agent instruction files must exist before rules-per-team layering is meaningful\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"CLI agent session logs or telemetry showing primary usage\",\"Rules files in repository with commit history showing regular updates\",\"Coding conventions document cross-referenced from agent instruction files\"]},\"4\":{\"must\":[\"Unattended agents execute tasks without developer presence\",\"Agents are invocable from at least two channels (Slack, CLI, Web, PagerDuty)\",\"Developers routinely run several agent sessions concurrently, against a documented span-of-control limit\"],\"should\":[\"Agent task completion rate without human intervention exceeds 60%\",\"Agent invocation produces a PR within a defined SLA (e.g., under 30 minutes for standard tasks)\"],\"prerequisites\":[{\"label\":\"Infrastructure L3 (Agent Runtime \u0026 Sandboxing) - isolated environments required for unattended execution\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L3 (CI/CD Pipeline) - CI under 5 minutes required for agent iteration loops\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Agent invocation logs from multiple channels with timestamps\",\"Dashboard showing parallel agent session counts per developer\",\"PR history showing agent-authored PRs merged without synchronous developer oversight\"]},\"5\":{\"must\":[\"Multi-agent orchestration system (planner-worker hierarchy) is in production\",\"The agent fleet scales past what a single team could supervise, bounded by compute and review capacity rather than by tooling\",\"Agent fleet produces 1,000+ commits per week without manual dispatch\"],\"should\":[\"Planner agents decompose epics into tasks and assign to worker agents autonomously\",\"Agent fleet self-recovers from failures without human escalation for 90%+ of error cases\"],\"prerequisites\":[{\"label\":\"Development L4 (Coding Agent Usage) - unattended agents must be proven before fleet orchestration\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}
102,{\"label\":\"Infrastructure L4 (Agent Runtime \u0026 Sandboxing) - ephemeral devboxes with sub-10s spin-up required\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Orchestration system dashboard showing planner-worker task flow\",\"Git history showing 1,000+ weekly commits attributed to agent fleet\",\"Agent fleet monitoring showing concurrent agent count and error recovery rate\"]}},\"Context Engineering\":{\"1\":{\"must\":[\"The agent can read the file(s) the developer is working on\",\"Developers can supply the agent with project context when needed\"],\"should\":[\"README.md exists (may be incomplete)\",\"Developers manually paste context into AI chat when needed\"],\"prerequisites\":[],\"evidence\":[\"Absence of agent instruction files in repository\",\"README.md with last-modified date older than 6 months\"]},\"2\":{\"must\":[\"CLAUDE.md or equivalent exists with project description, tech stack, and top conventions\",\"Written coding conventions document exists and is referenced from agent instruction files\",\"Agent instruction files are committed to the repository (not local-only)\"],\"should\":[\"CLAUDE.md includes explicit prohibitions (banned libraries, anti-patterns)\",\"Agent instruction files are reviewed as part of the standard PR process\"],\"prerequisites\":[],\"evidence\":[\"CLAUDE.md, .cursorrules, or .github/copilot-instructions.md in repository root\",\"Coding conventions document accessible from agent instruction files\",\"Commit history showing agent instruction file updates\"]},\"3\":{\"must\":[\"MCP servers provide structured context (architecture, ownership, SLAs) to agents\",\"Context is organized across at least 3 of the 5 levels: System, Code, Org, Historical, Operational\",\"Token budget management is implemented (agents receive context within defined token limits)\"],\"should\":[\"Context sources are versioned and tested for correctness\",\"Context budgeting policy defines priority order when token limits are reached\"],\"prerequisites\":[{\"label\":\"Infrastructure L2 (MCP \u0026 Tool Integration) - basic MCP servers must exist before structured context delivery\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"MCP server configuration files listing active context sources\",\"Token budget configuration in agent settings\",\"Context coverage audit showing 3+ context levels populated\"]},\"4\":{\"must\":[\"Organization pushes context to agents automatically (BYOC - Bring Your Own Context)\",\"A queryable map of code structure and ownership is integrated with the agent context pipeline\",\"Ticket-to-spec automation generates acceptance tests from requirements without manual writing\"],\"should\":[\"Context push triggers on repository events (commit, PR, deploy) without manual refresh\",\"Knowledge graph covers 80%+ of active repositories\"],\"prerequisites\":[{\"label\":\"Development L3 (Context Engineering) - MCP-based context must be operational before push-based delivery\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Organization L3 (Knowledge Management) - documentation-as-infrastructure must exist before knowledge graph integration\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"BYOC pipeline configuration showing automated context push triggers\",\"Knowledge graph dashboard showing repository coverage percentage\",\"Sample ticket-to-spec outputs with auto-generated acceptance tests\"]},\"5\":{\"must\":[\"Agents maintain persistent identity and memory across sessions (Beads/Git-backed)\",\"Production telemetry feeds back into agent context automatically (deploy, error, performance data)\",\"Context that has gone stale is detected and refreshed before an agent runs on it\"],\"should\":[\"Agent memory persists architectural decisions and their rationale across sessions\",\"Self-healing context updates are validated by automated tests before commit\"],\"prerequisites\":[{\"label\":\"Development L4 (Context Engineering) - automated context delivery must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}
102,{\"label\":\"Infrastructure L4 (Observability \u0026 Feedback Loop) - production telemetry pipeline required for context auto-update\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Agent memory store with session-spanning entries and timestamps\",\"Production telemetry-to-context pipeline configuration with update frequency\",\"Git history showing agent-authored documentation updates with passing CI\"]}},\"Code Review \u0026 Quality\":{\"1\":{\"must\":[\"All code is reviewed by a human before merge\",\"CI runs at least one automated check on each change\"],\"should\":[\"Code review turnaround is tracked (even if slow)\",\"Team is aware that AI-generated code has higher defect rates (1.7x issues, 2.74x security vulnerabilities)\"],\"prerequisites\":[],\"evidence\":[\"PR approval records showing human reviewer on every merged PR\",\"Average review turnaround time in PR analytics\"]},\"2\":{\"must\":[\"An AI reviewer comments on pull requests in the repositories the team works in (CodeRabbit, Qodo, or equivalent)\",\"Linter rules are configured and run in CI on every PR\",\"PRs clearly indicate whether code is AI-generated or AI-assisted (labels, tags, or commit metadata)\"],\"should\":[\"AI review suggestions are triaged (accepted/rejected) rather than ignored\",\"Linter configuration is committed to the repository and versioned\"],\"prerequisites\":[],\"evidence\":[\"AI review tool configuration in CI pipeline\",\"Linter configuration file in repository\",\"PR labels or commit metadata distinguishing AI-generated code\"]},\"3\":{\"must\":[\"AI review agent runs as a first-pass reviewer on every PR before human review\",\"Lint rules enforce architectural standards (not just style) - the \\\"Bug to Codify to Lint Rule\\\" pipeline is active\",\"At least 3 architectural guardrail rules have been created from past bugs or incidents\"],\"should\":[\"AI review agent findings are categorized by severity (info, warning, blocking)\",\"New lint rules are proposed automatically when recurring review comments are detected\"],\"prerequisites\":[{\"label\":\"Delivery L2 (CI/CD Pipeline) - CI must run in under 10 minutes for AI review to be practical as first pass\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"CI configuration showing AI review agent as required check\",\"Lint rule change history showing rules created from incident post-mortems\",\"AI review agent output logs with severity categories\"]},\"4\":{\"must\":[\"Automated Green/Yellow/Red classification runs on every PR\",\"Classification is deterministic: the same change lands in the same class on every run\",\"Auto-approve rate target of 60%+ Green PRs is tracked and reported\"],\"should\":[\"Yellow PRs receive expedited human review (within 1 hour)\",\"Classification model accuracy is validated monthly against human review outcomes\"],\"prerequisites\":[{\"label\":\"Development L3 (Code Review \u0026 Quality) - AI review agent and architectural guardrails must be in place\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L3 (Testing Strategy) - reliable test oracles required for Green classification to be trustworthy\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Dashboard showing Green/Yellow/Red distribution across PRs\",\"Auto-merge logs for Green PRs with zero post-merge reverts\",\"Monthly auto-approve rate report showing 60%+ Green target tracking\",\"Every agent-authored PR shows a named human owner in the merge record\"]},\"5\":{\"must\":[\"Agent fleet self-reviews code (error-fix-converge loop) before submitting for merge\",\"Human review is limited to Red-classified PRs (architectural decisions only)\",\"Continuous auto-refactoring runs in background without human initiation\"],\"should\":[\"Agent self-review catches 90%+ of issues that would be found by human review\",\"Auto-refactoring PRs are tracked separately and have their own quality metrics\"],\"prerequisites\":[{\"label\":\"Development L4 (Code Review \u0026 Quality) - automated classification and auto-merge must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L4 (Testing Strategy) - trustworthy test oracles required for self-review convergence\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Agent iteration logs showing error-fix-converge cycles before PR submission\",\"PR analytics showing human review only on Red-classified PRs\",\"Auto-refactoring PR history with associated quality metrics\"]}},\"Testing Strategy\":{\"1\":{\"must\":[\"An automated test suite exists and runs\",\"The team writes and maintains its own tests\"],\"should\":[\"Team is aware of flaky test impact (16% of dev time per Google data)\",\"AI-generated tests are reviewed for circular testing (testing what code does, not what it should do)\"],\"prerequisites\":[],\"evidence\":[\"Coverage report from the existing test suite\",\"Test authorship in git history (manual, no agent attribution)\"]},\"2\":{\"must\":[\"Agents generate unit tests; humans write acceptance tests\",\"Flaky test quarantine process is active (flaky tests are isolated, not deleted)\",\"Humans define the expected results for important paths (not just snapshotting current output)\"],\"should\":[\"Flaky test c
102ount is tracked and reported weekly\",\"Quarantined tests have a resolution SLA (e.g., fix or delete within 30 days)\"],\"prerequisites\":[],\"evidence\":[\"Test files with agent attribution alongside human-authored acceptance tests\",\"Quarantine list or label in test framework configuration\",\"Flaky test tracking dashboard or issue tracker labels\"]},\"3\":{\"must\":[\"Expected results are derived from requirements/specs (the requirement is the oracle, not the code)\",\"Acceptance tests are auto-generated from ticket requirements (Autonomous Requirements pipeline)\",\"Incremental test selection runs only tests affected by changed code paths\"],\"should\":[\"Oracle reliability is reviewed per service, not just overall\",\"Test generation from tickets includes edge cases, not just happy paths\"],\"prerequisites\":[{\"label\":\"Delivery L2 (CI/CD Pipeline) - CI under 10 minutes required for incremental test selection to matter\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L2 (Context Engineering) - written conventions required for meaningful test oracle stabilization\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Oracle-reliability dashboard (e.g., TORS) with per-service breakdown\",\"Ticket-to-test pipeline configuration with sample outputs\",\"CI configuration showing incremental test selection driven by the change set\"]},\"4\":{\"must\":[\"A failing test reliably indicates a real defect (oracle false-positives are rare)\",\"Agents iterate tests to green in isolated sandbox CI without blocking team CI queue\",\"Mutation testing validates that tests catch real defects (not just achieve coverage)\"],\"should\":[\"Sandbox CI iteration count per PR is tracked (ITS target: 1-3)\",\"Mutation testing kill rate exceeds 80%\"],\"prerequisites\":[{\"label\":\"Development L3 (Testing Strategy) - requirement-derived oracles and incremental test selection must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L3 (Agent Runtime \u0026 Sandboxing) - isolated environments required for sandbox iteration\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Oracle-reliability dashboard (e.g., TORS) with per-service breakdown\",\"Sandbox CI logs showing agent iteration cycles separate from team CI\",\"Mutation testing reports showing kill rate and surviving mutants\"]},\"5\":{\"must\":[\"Test suite is self-healing (agent detects broken tests, diagnoses root cause, fixes without human input)\",\"Production logs automatically generate regression tests for observed failures\",\"Agents detect edge cases, write tests, fix bugs, and ship - full autonomous loop\"],\"should\":[\"Self-healing test updates are validated by mutation testing before merge\",\"Production-to-test pipeline latency is under 1 hour (failure observed to regression test committed)\"],\"prerequisites\":[{\"label\":\"Development L4 (Testing Strategy) - trustworthy oracles and mutation testing must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L4 (Observability \u0026 Feedback Loop) - production telemetry pipeline required for log-to-test generation\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Self-healing test commit history showing agent-diagnosed and agent-fixed test failures\",\"Production log-to-test pipeline configuration with sample generated tests\",\"End-to-end autonomous bug fix PRs (edge case detected, test written, fix shipped)\"]}}},\"delivery\":{\"CI/CD Pipeline\":{\"1\":{\"must\":[\"A CI pipeline runs on pull requests\",\"CI results are reported after the pipeline completes\"],\"should\":[\"CI runs on every PR (not just on manual trigger)\",\"Shared runner queue exists even if slow\"],\"prerequisites\":[],\"evidence\":[\"CI pipeline configuration file in repository\",\"CI run duration logs showing median \u003e 15 minutes\"]},\"2\":{\"must\":[\"Pipeline definitions are versioned in the repository and changed through code review\",\"Dedicated CI runners are allocated per team (no shared queue across all teams)\",\"CI completes in under 10 minutes (median)\"],\"should\":[\"CI duration is tracked as a metric and reviewed monthly\",\"Cache hit rate exceeds 70%\"],\"prerequisites\":[],\"evidence\":[\"CI run duration dashboard showing median under 10 minutes\",\"Cache
102configuration in CI pipeline (e.g., actions/cache, Gradle build cache)\",\"Runner allocation configuration showing per-team resources\"]},\"3\":{\"must\":[\"Agent CI runs are isolated from untrusted input: no event data interpolated into shell steps, and agent passes run as separate jobs with separately scoped tokens\",\"Parallel agents get per-worktree pipelines rather than serialising on one runner\",\"CI completes in under 5 minutes (median)\"],\"should\":[\"P95 CI duration is under 8 minutes\",\"Build system supports hermetic builds (reproducible outputs regardless of machine)\"],\"prerequisites\":[{\"label\":\"Infrastructure L2 (Build System) - basic build caching must be in place before remote caching is meaningful\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"CI run duration dashboard showing median under 5 minutes\",\"Remote cache configuration and cache hit rate metrics\",\"Build configuration showing incremental/changed-only targeting\"]},\"4\":{\"must\":[\"Agent sandbox CI absorbs repeated agent iteration without blocking the team CI queue\",\"Ephemeral sandbox environments spin up in under 10 seconds for agent CI loops\",\"CI completes in under 2 minutes (median)\"],\"should\":[\"P95 CI duration is under 3 minutes\",\"CI feedback latency (from push to result) is tracked and reported\"],\"prerequisites\":[{\"label\":\"Delivery L3 (CI/CD Pipeline) - sub-5-minute CI and remote caching must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L4 (Agent Runtime \u0026 Sandboxing) - ephemeral devboxes with sub-10s spin-up required\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"CI run duration dashboard showing median under 2 minutes\",\"Sandbox spin-up time metrics showing sub-10-second P50\",\"Agent CI iteration logs showing sustained retry loops that never entered the team CI queue\"]},\"5\":{\"must\":[\"Production feedback loop auto-adjusts the CI test suite (adds tests for observed failures, removes redundant tests)\",\"CI auto-scales runner capacity based on agent load (no manual capacity planning)\",\"CI provides sub-minute feedback for standard changes\"],\"should\":[\"CI runner utilization stays between 50-80% (auto-scaling prevents both waste and queuing)\",\"Test suite evolution is auditable (each auto-added/removed test has a provenance record)\"],\"prerequisites\":[{\"label\":\"Delivery L4 (CI/CD Pipeline) - sub-2-minute CI and ephemeral sandboxes must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L4 (Observability \u0026 Feedback Loop) - production telemetry required for feedback-driven test suite adjustment\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"CI run duration dashboard showing sub-minute median for standard changes\",\"Auto-scaling configuration and runner utilization metrics\",\"Test suite change log showing production-feedback-driven additions and removals\"]}},\"Merge \u0026 Deploy\":{\"1\":{\"must\":[\"Merging is a manual step performed by a person on every change\",\"The team merges pull requests at least weekly\"],\"should\":[\"Basic CD pipeline exists (even if simple or manually triggered)\",\"Deploy frequency is at least weekly\"],\"prerequisites\":[],\"evidence\":[\"PR merge history showing manual approvals\",\"Deploy logs showing manual trigger or simple CD pipeline\"]},\"2\":{\"must\":[\"CD pipeline includes at least one gate (tests pass, security scan, approval)\",\"Merge queue is implemented (GitHub merge queue, Mergify, or equivalent)\",\"Auto-rebase is enabled for PRs targeting main branch\"],\"should\":[\"Merge conflicts are detected and flagged before review is requested\",\"Deploy frequency is at least daily\"],\"prerequisites\":[{\"label\":\"Delivery L1 (CI/CD Pipeline) - CI pipeline must exist for merge queue to function\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Merge queue configuration in repository settings or CI\",\"Auto-rebase configuration (branch protection rules, bot configuration)\",\"CD pipeline definition showing gate conditions\"]},\"3\":{\"must\":[\"Policy-based merge rules are enforced (OPA, branch protection, or equivalent)\",\"Deterministic merge ordering with conflict detection prevents concurrent merge failures\",\"PRs require a small, published maximum number of CI rounds before merge, and the team tracks it\"],\"should\":[\"Merge rules are versioned as code and reviewed when changed\",\"PRs exceeding 2 CI rounds are flagged for investigation\"],\"prerequisites\":[{\"label\":\"Delivery L2 (Merge \u0026 Deploy) - merge queue and auto-rebase must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L2 (Governance \u0026 Compliance) - official AI tool policy required for policy-based rules\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Policy-as-code configuration (OPA rules, branch protection API config)\",\"CI round count per PR metrics showing 2-round maximum adherence\",\"Merge ordering logs showing deterministic processing\"]},\"4\":{\"must\":[\"Green-classified PRs auto-merge and auto-deploy without human intervention\",\"Merge throughput has risen substantially against the pre-agent baseline, and the merge path is no longer the constraint on delivery\",\"Canary or progressive deployment is automated (no manual rollout decisions)\"],\"should\":[\"Auto-deploy includes automated rollback on error rate threshold breach\",\"Merge queue wait time is under 10 minutes\"],\"prerequisites\":[{\"label\":\"Development L4 (Code Review \u0026 Quality) - Green/Yellow/Red classification and auto-merge must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L3 (Merge \u0026 Deploy) - policy-based rules and deterministic ordering must be in place\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Auto-merge and auto-deploy logs for Green PRs\",\"PR throughput dashboard showing sustained growth against the pre-agent baseline\",\"Canary deployment c
102onfiguration with automated promotion/rollback rules\"]},\"5\":{\"must\":[\"Full autonomous pipeline: agent produces PR, CI passes, merge, deploy, observe - no human in the loop\",\"Rollback is agent-driven (agent detects regression, reverts, and opens fix PR)\",\"Merge throughput is limited by product decisions rather than by the merge path itself\"],\"should\":[\"Mean time to rollback is under 5 minutes from anomaly detection\",\"Agent-driven rollbacks succeed without human intervention 95%+ of the time\"],\"prerequisites\":[{\"label\":\"Delivery L4 (Merge \u0026 Deploy) - auto-merge, auto-deploy, and canary deployment must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L4 (Observability \u0026 Feedback Loop) - production anomaly detection required for agent-driven rollback\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Merge throughput dashboard showing 1,000+ per week\",\"End-to-end autonomous pipeline logs (PR to production with no human steps)\",\"Agent-driven rollback logs with timestamps and success rate\"]}},\"Metrics\":{\"1\":{\"must\":[\"The team can say how often it ships and how often a change fails\",\"The team looks at these metrics at least once a month\"],\"should\":[\"Team acknowledges that its existing delivery metrics do not capture AI-assisted work\",\"Basic deployment frequency is at least known (even if not dashboarded)\"],\"prerequisites\":[],\"evidence\":[\"Absence of metrics dashboard or inconsistent/manual tracking\",\"No AI-specific fields in existing metrics systems\"]},\"2\":{\"must\":[\"A delivery-performance baseline (throughput, lead time, change failure rate, restore time - DORA, SPACE or an equivalent set) is on a dashboard the team can open\",\"AI tool license count vs. active usage rate is measured\",\"PR throughput per developer is tracked\"],\"should\":[\"Cost per merged PR is measured per tool (acceptance rate is not used as a quality signal)\",\"Metrics are reviewed in team retrospectives at least monthly\"],\"prerequisites\":[],\"evidence\":[\"Delivery-performance dashboard with current data\",\"License utilization report (licenses purchased vs. active users)\",\"PR throughput chart showing per-developer breakdown\"]},\"3\":{\"must\":[\"The team measures the cost of one agent iteration (cost per iteration, CPI). The cost includes model tokens and CI compute.\",\"The team counts the iterations that one change needs to pass CI (iterations to success, ITS).\",\"The team records the median time from a push to the CI result, for each reporting period.\"],\"should\":[\"The team sets a limit for the cost per iteration and for the iteration count.\",\"The team reports each measurement per team and per repository.\",\"The team attributes cost to a delivered change, not only to a pull request.\"],\"prerequisites\":[{\"label\":\"Delivery L2 (Metrics) - a delivery-performance baseline must be operational before the AI-specific metrics layer\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L2 (CI/CD Pipeline) - CI must be fast enough for the iteration count to be meaningful\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"A cost record for each iteration. The record separates the model component from the CI component.\",\"An iteration count for each change that entered CI, with the distribution across changes.\",\"A chart of the push-to-result time. The chart shows the median and the 95th percentile.\"]},\"4\":{\"must\":[\"Test-oracle reliability is measured and tracked on a dashboard\",\"Every agent run is recorded with a terminal state (success / flawed / blocked / manual), and the distribution is reviewed\",\"Each non-success state has a named owner and routes to a different remedy\"],\"should\":[\"Agent Autonomy Score (% of tasks completed without human intervention) is measured and broken down by task type\",\"Metrics trigger automated alerts when thresholds are breached (e.g., test-oracle reliability drops)\"],\"prerequisites\":[{\"label\":\"Delivery L3 (Metrics) - ITS, CPI, and CI feedback latency must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L4 (Code Review \u0026 Quality) - auto-approve workflow must exist for auto-approve rate tracking\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Oracle-reliability dashboard (e.g., TORS) with per-service breakdown\",\"Run-state distribution report showing how the non-success classes moved between periods\",\"Merge queue wait time chart showing sub-10-minute target\"]},\"5\":{\"must\":[\"Cost-per-feature is tracked (not cost-per-PR) - aggregating all agent, CI, and review costs per delivered feature\",\"Business value throughput is the primary metric (features delivered per week, not PRs merged per week)\"],\"should\":[\"Metrics system auto-detects vanity metrics (high activity, low value delivery) and flags them\",\"Cost-per-feature trend is declining quarter-over-quarter\"],\"prerequisites\":[{\"label\":\"Delivery L4 (Metrics) - all L4 metrics must be operational before business-value metrics are meaningful\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Cost-per-feature dashboard with feature-level cost attribution\",\"Business value throughput chart correlated with product delivery milestones\",\"Quarter-over-quarter cost-per-feature trend report\"]}},\"Governance \u0026 Compliance\":{\"1\":{\"must\":[\"The team knows which AI tools are in use\",\"AI-generated code follows the normal review and merge process\"],\"should\":[\"Team is aware of shadow AI usage (developers using private subscriptions)\",\"Organization has moved past \\\"ban AI\\\" as a policy position\"],\"prerequisites\":[],\"evidence\":[\"Absence of written AI tool policy\",\"No AI-related fields in commit metadata or PR templates\"]},\"2\":{\"must\":[\"Official AI tool policy exists and is communicated to all developers\",\"The organization can list which developers use which AI tools\",\"The regulatory obligations that apply to this organisation's jurisdiction and sector are written down, with a named owner\"],\"should\":[\"AI tool policy is reviewed at least annually\",\"Approved tool list is maintained and accessible\",\"Every agent in use has a named human owner\",\"New vend
102or AI features are enabled by a deliberate decision, not by vendor default\"],\"prerequisites\":[],\"evidence\":[\"Published AI tool policy document with distribution records\",\"AI tool usage tracking dashboard or report\"]},\"3\":{\"must\":[\"Minimum viable audit trail is captured per AI-assisted change: model identifier, timestamp, context description, human approver\",\"Policy-as-code enforces compliance rules in CI (OPA or equivalent)\",\"Compliance gates run on every PR to in-scope repositories\"],\"should\":[\"Audit trail fields are validated by CI (missing fields fail the build)\",\"Policy exceptions are logged and require follow-up within 48 hours\",\"Model traffic goes through an AI gateway with a model allowlist and exact-version pinning\",\"Each agent is registered as a distinct identity in the IdP, not acting on a developer's token\"],\"prerequisites\":[{\"label\":\"Delivery L2 (Governance \u0026 Compliance) - official AI tool policy must exist before codifying it\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L2 (CI/CD Pipeline) - CI must be fast enough for compliance gates to not block development\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Sample commit or PR metadata showing model, timestamp, context, approver fields\",\"OPA policy configuration in CI pipeline\",\"Compliance gate pass/fail logs\",\"AI gateway configuration showing the model allowlist, pinned versions and denied models\"]},\"4\":{\"must\":[\"Full provenance tracking per change: model version, prompt context, agent session ID, iteration count\",\"Automated compliance checks run without manual intervention on every merge\",\"AI-generated code is distinguishable from human-written code in version control (metadata, labels, or attribution)\"],\"should\":[\"Provenance data is queryable (e.g., \\\"show all changes made by model X in the last 30 days\\\")\",\"Compliance check results are aggregated into a governance dashboard\",\"Prompt and audit logs are retained beyond vendor defaults, in a store the organisation controls\",\"The AI gateway has a patch SLA and is monitored for CVEs like any other critical infrastructure\"],\"prerequisites\":[{\"label\":\"Delivery L3 (Governance \u0026 Compliance) - audit trail and policy-as-code must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L3 (Metrics) - metrics infrastructure required for compliance dashboarding\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Provenance metadata on commits/PRs showing full attribution chain\",\"Automated compliance check configuration with zero manual steps\",\"VCS query showing AI-vs-human code distinction\"]},\"5\":{\"must\":[\"Continuous compliance: agent monitors regulatory changes (EU AI Act updates, SOC2 changes) and proposes policy updates\",\"Audit trail is self-documenting (agent decisions include reasoning, not just outcomes)\",\"Enterprise-grade RBAC is enforced per agent (each agent has scoped permissions for specific tools and repositories)\"],\"should\":[\"Policy update proposals from compliance agent are auto-tested against existing codebase before rollout\",\"Agent RBAC permissions are audited automatically for least-privilege compliance\"],\"prerequisites\":[{\"label\":\"Delivery L4 (Governance \u0026 Compliance) - full provenance tracking and automated compliance must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L4 (MCP \u0026 Tool Integration) - a unified, governed tool gateway required for per-agent RBAC\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Compliance agent logs showing regulatory monitoring and policy update proposals\",\"Self-documenting audit trail entries with agent reasoning chains\",\"Agent RBAC configuration showing per-agent tool and repository permissions\"]}}},\"infrastructure\":{\"Agent Runtime \u0026 Sandboxing\":{\"1\":{\"must\":[\"Agents can run in the developer's local environment\",\"Agents have file-system and shell access in their run environment\"],\"should\":[\"Developers are aware of the security implications of agents with full local access\",\"Agent access scope (file system, network) is understood even if not restricted\"],\"prerequisites\":[],\"evidence\":[\"Agent runs as IDE plugin with no containerization or isolation\",\"No sandboxing configuration exists\"]},\"2\":{\"must\":[\"Dedicated development environments exist for agent execution (separate from developer's primary workspace)\",\"Agents run inside a container, not directly on a developer machine\",\"Agent credentials are scoped per project (not a single org-wide key)\"],\"should\":[\"Container images for agent environments are versioned and reproducible\",\"Credential rotation schedule exists for agent-scoped keys\",\"Agents never run on a developer's personal access token or a shared long-lived key\"],\"prerequisites\":[],\"evidence\":[\"Docker or container configuration files for agent environments\",\"Credential management configuration showing per-project scoping\",\"Environment provisioning documentation or scripts\"]},\"3\":{\"must\":[\"Isolated agent environments (devbox model) prevent agents from accessing other projects\",\"Pre-warmed containers with codebase at HEAD and dependencies installed are available\",\"Network isolation prevents agents from reaching production systems\"],\"should\":[\"Container warm pool size matches team's agent usage patterns\",\"Network isolation rules are tested and audited quarterly\",\"Credentials are injected at run time from a vault or broker, never stored in prompts, files or evaluation sandboxes\"],\"prerequisites\":[{\"label\":\"Infrastru
102cture L2 (Agent Runtime \u0026 Sandboxing) - basic sandboxing must be in place before isolation layering\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L2 (Build System) - build caching required for pre-warmed container efficiency\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Devbox configuration showing per-project isolation boundaries\",\"Pre-warmed container pool metrics (pool size, warm hit rate, cold start rate)\",\"Network policy configuration (Kubernetes NetworkPolicy, firewall rules) blocking production access\"]},\"4\":{\"must\":[\"Ephemeral devboxes spin up fast enough that an agent does not wait on the environment (single-digit seconds)\",\"Devboxes come pre-loaded with codebase, dependencies, and MCP tools\",\"Kernel-level policy enforcement restricts agent actions (syscall filtering, resource limits)\"],\"should\":[\"Devbox spin-up P99 latency is under 30 seconds\",\"Firecracker microVMs or equivalent provide VM-level isolation with container-level startup speed\"],\"prerequisites\":[{\"label\":\"Infrastructure L3 (Agent Runtime \u0026 Sandboxing) - isolated environments and pre-warmed containers must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L3 (MCP \u0026 Tool Integration) - MCP platform required for pre-loaded tool configuration\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Devbox spin-up latency dashboard showing P50 under 10 seconds\",\"Devbox snapshot configuration showing pre-loaded codebase, deps, and MCP tools\",\"Kernel policy configuration (seccomp profiles, cgroup limits)\"]}
102,\"5\":{\"must\":[\"Dedicated compute infrastructure exists for agent fleet (not shared with developer workstations or production)\",\"Agent fleet auto-scales with load (agents scale up during business hours, scale down off-hours)\",\"Each agent runs in a fully isolated environment (one machine per agent, or equivalent resource isolation)\"],\"should\":[\"Cost per agent-hour is tracked and optimized\",\"Fleet scaling responds to demand within 60 seconds\"],\"prerequisites\":[{\"label\":\"Infrastructure L4 (Agent Runtime \u0026 Sandboxing) - ephemeral devboxes and kernel-level policy must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Infrastructure allocation showing dedicated agent compute (separate from dev and prod)\",\"Auto-scaling configuration and scaling event logs\",\"Agent fleet dashboard showing per-agent isolation and resource utilization\"]}},\"MCP \u0026 Tool Integration\":{\"1\":{\"must\":[\"Agents use their built-in tools\",\"Agents draw on their general / built-in knowledge\"],\"should\":[\"Team is aware of MCP as a standard for agent-tool integration\",\"Integrations, if any, are manual (copy-paste between tools)\"],\"prerequisites\":[],\"evidence\":[\"No MCP configuration files in repository or developer environment\",\"Absence of tool integration beyond IDE built-ins\"]},\"2\":{\"must\":[\"1-3 MCP servers are configured (e.g., Git, Jira, documentation)\",\"MCP setup is documented but configured manually per developer\",\"Agents authenticate before they can use a tool server\"],\"should\":[\"MCP server configurations are shared via repository (not local-only)\",\"At least one MCP server provides internal documentation or codebase context\"],\"prerequisites\":[],\"evidence\":[\"MCP server configuration files (mcp.json or equivalent)\",\"Setup documentation for MCP server installation per developer\",\"MCP server authentication configuration\"]},\"3\":{\"must\":[\"Centralized MCP platform manages server provisioning, configuration, and lifecycle\",\"Domain-specific MCP servers exist (Architecture MCP, Ownership MCP, SLA MCP)\",\"RBAC controls which agents can access which MCP tools\"],\"should\":[\"MCP server health is monitored with alerting on downtime\",\"New MCP servers go through a standardized review and onboarding process\",\"Each caller sees only the MCP tools it is authorised to use (per-caller tool filtering at the gateway)\"],\"prerequisites\":[{\"label\":\"Infrastructure L2 (MCP \u0026 Tool Integration) - basic MCP servers must exist before centralization\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Organization L3 (AI Adoption Model) - platform team must own AI tooling for centralized MCP management\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"MCP platform configuration showing centralized server management\",\"RBAC policy configuration for MCP tool access\",\"MCP server inventory listing domain-specific servers with owners\"]},\"4\":{\"must\":[\"The organisation's tool surface is reachable through a unified, governed MCP gateway rather than per-team wiring\",\"Agent discovery: agents can query available tools and their capabilities at runtime\",\"MCP governance covers lifecycle management, versioning, and audit logging\"],\"should\":[\"MCP tool usage analytics track which tools are used, by which agents, how often\",\"MCP server versioning allows rollback to previous versions without downtime\",\"Agent access to tools is granted centrally through the IdP rather than by per-user consent\"],\"prerequisites\":[{\"label\":\"Infrastructure L3 (MCP \u0026 Tool Integration) - centralized MCP platform and RBAC must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L3 (Governance \u0026 Compliance) - audit trail infrastructure required for MCP audit logging\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"MCP gateway configuration showing the organisation's tool surface registered behind one gateway\",\"Agent discovery API or protocol documentation with runtime tool listing\",\"MCP governance logs showing lifecycle events (deploy, version, deprecate, audit)\"]}
102,\"5\":{\"must\":[\"MCP operates as a bidirectional nervous system: production data flows to agents, agent actions flow to production\",\"Full production loop: Production -\u003e MCP -\u003e Agent -\u003e Code -\u003e Deploy -\u003e Production\",\"Agent-to-Agent Protocol (A2A) and MCP are combined for multi-agent coordination\"],\"should\":[\"MCP latency for context delivery is under 500ms P95\",\"A2A protocol enables agents to discover and delegate to other agents without human configuration\"],\"prerequisites\":[{\"label\":\"Infrastructure L4 (MCP \u0026 Tool Integration) - a unified tool gateway and MCP governance must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L4 (Observability \u0026 Feedback Loop) - production telemetry pipeline required for bidirectional flow\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"MCP configuration showing bidirectional data flow (production to agent, agent to production)\",\"End-to-end production loop traces (anomaly detected, agent invoked, fix deployed)\",\"A2A protocol configuration showing agent-to-agent communication channels\"]}},\"Build System\":{\"1\":{\"must\":[\"A build system is in place (default configuration is fine)\",\"Builds run on each change\"],\"should\":[\"Build completes (even if slowly)\",\"CI runs builds on a shared queue (even if everyone waits)\"],\"prerequisites\":[],\"evidence\":[\"Build configuration file with default/untuned settings\",\"CI logs showing full rebuild on every PR\"]},\"2\":{\"must\":[\"Build caching is implemented (dependency cache, compilation cache)\",\"Parallel build steps are configured (test and lint run concurrently)\",\"The build cache is shared between developers and CI rather than rebuilt per machine\"],\"should\":[\"Cache hit rate exceeds 60%\",\"Build time has improved by at least 30% compared to uncached baseline\"],\"prerequisites\":[],\"evidence\":[\"Build cache configuration (Gradle build cache, npm cache, Docker layer cache)\",\"CI pipeline configuration showing parallel step execution\",\"Dedicated runner or resource pool configuration\"]},\"3\":{\"must\":[\"Incremental builds run only changed targets (not full rebuild)\",\"Remote execution distributes build steps across multiple machines\",\"The main codebase uses a build tool with an explicit dependency graph and a shared cache (Bazel, Buck2, Pants, Nx, Turborepo or equivalent)\"],\"should\":[\"BUILD file maintenance is assigned to specific team members or automated\",\"Remote cache hit rate exceeds 80%\"],\"prerequisites\":[{\"label\":\"Infrastructure L2 (Build System) - basic caching and parallelization must be in place before advanced build system adoption\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Build definitions in the repository showing an explicit dependency graph\",\"Remote execution configuration\",\"Build log showing incremental target selection\"]},\"4\":{\"must\":[\"Agent-specific build profiles exist (optimized for agent iteration patterns - fast feedback over comprehensive build)\",\"Build system understands agent iteration patterns and pre-caches likely next builds\",\"Any change gets build feedback in under 2 minutes\"],\"should\":[\"Build profiles are auto-selected based on invoker (agent vs. human vs. CI)\",\"Pre-caching hit rate exceeds 70% for agent iterations\"],\"prerequisites\":[{\"label\":\"Infrastructure L3 (Build System) - a dependency-graph build with shared cache and remote execution must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L3 (CI/CD Pipeline) - CI under 5 minutes required as baseline before sub-2-minute build targeting\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Build duration dashboard showing sub-2-minute feedback for all change types\",\"Agent-specific build profile configuration\",\"Pre-cache hit rate metrics for agent iteration patterns\"]},\"5\":{\"must\":[\"Build is a commodity: near-instant feedback for agents regardless of codebase size\",\"Codebase is structured into self-contained modules to eliminate the compilation bottleneck\",\"Disk I/O is optimized for concurrent agent workloads (parallel reads/writes across modules)\"],\"should\":[\"Build latency is under 30 seconds for 90%+ of changes\",\"Module dependency graph is automatically maintained and optimized\"],\"prerequisites\":[{\"label\":\"Infrastru
102cture L4 (Build System) - sub-2-minute builds and agent-specific profiles must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Build duration dashboard showing near-instant feedback for standard changes\",\"Codebase architecture showing modular structure (crate/module boundaries)\",\"Disk I/O benchmarks for concurrent agent build workloads\"]}},\"Observability \u0026 Feedback Loop\":{\"1\":{\"must\":[\"The application writes logs a developer can read\",\"Alerting fires on application errors\"],\"should\":[\"Logs are searchable (centralized logging, not just local files)\",\"Production issues do not yet feed back into dev priorities\"],\"prerequisites\":[],\"evidence\":[\"Logging configuration in application code\",\"Alert configuration (PagerDuty, Opsgenie, or equivalent)\"]},\"2\":{\"must\":[\"Structured logging is implemented (JSON logs with consistent fields)\",\"The application emits traces and metrics, not only logs\",\"Post-deploy monitoring checks run after each deployment\"],\"should\":[\"Traces are correlated across services\",\"Post-deploy checks include automated smoke tests\"],\"prerequisites\":[],\"evidence\":[\"Structured logging configuration showing JSON format with standard fields\",\"OpenTelemetry SDK configuration in application code\",\"Post-deploy monitoring job configuration in CD pipeline\"]},\"3\":{\"must\":[\"Traces, metrics and logs from production are queryable in one place\",\"Production metrics feed into dashboards accessible to all developers\",\"Incident data (post-mortems, error patterns) is available as agent context\"],\"should\":[\"SLOs are defined and tracked for key services\",\"Incident data is structured for machine consumption (not just human-readable post-mortem docs)\"],\"prerequisites\":[{\"label\":\"Infrastructure L2 (Observability \u0026 Feedback Loop) - structured logging and OTel must be in place\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L2 (Context Engineering) - agent instruction files must exist for incident context to be usable\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Observability stack configuration (OTel collector, Grafana dashboards)\",\"Production metrics dashboards with developer access\",\"Incident data accessible via MCP or structured API\"]},\"4\":{\"must\":[\"Production anomaly detection auto-creates tickets and triggers agent investigation\",\"Self-healing for known patterns: agent detects known error pattern, applies known fix, deploys, and verifies\",\"Infrastructure recommends code changes based on production data (Vercel SDI model)\"],\"should\":[\"Auto-created tickets include full context (traces, logs, affected users, similar past incidents)\",\"Self-healing success rate is tracked (% of auto-fixes that resolve the issue without human intervention)\"],\"prerequisites\":[{\"label\":\"Infrastructure L3 (Observability \u0026 Feedback Loop) - full observability stack and incident context must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L3 (Coding Agent Usage) - CLI agents must be available for automated agent investigation\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Auto-ticket creation logs triggered by production anomalies\",\"Self-healing event logs showing detection, fix, deploy, and verification steps\",\"Infrastructure recommendation pipeline configuration (production data to code change suggestions)\"]},\"5\":{\"must\":[\"Full production-to-agent loop operates autonomously: anomaly detected, investigated, fixed, tested, deployed\",\"Infrastructure self-drives: code defines infrastructure, production performance informs code changes\",\"Anomaly-to-deploy cycle completes without human intervention for 80%+ of known issue categories\"],\"should\":[\"Novel anomalies (not matching known patterns) are escalated to humans with full investigation context\",\"Mean time from anomaly detection to autonomous fix deployment is under 15 minutes\"],\"prerequisites\":[{\"label\":\"Infrastru
102cture L4 (Observability \u0026 Feedback Loop) - anomaly detection, self-healing, and SDI model must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L4 (Merge \u0026 Deploy) - auto-merge and auto-deploy must be operational for autonomous fix deployment\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"End-to-end autonomous fix traces (anomaly to deployed fix with no human steps)\",\"Infrastructure-as-code showing production-informed code changes\",\"Autonomous resolution rate dashboard showing 80%+ for known issue categories\"]}}},\"organization\":{\"AI Adoption Model\":{\"1\":{\"must\":[\"AI tools have been adopted (licenses acquired)\",\"Adoption is tracked informally\"],\"should\":[\"At least some developers are experimenting with AI tools\",\"Organization has not banned AI tool usage outright\"],\"prerequisites\":[],\"evidence\":[\"License purchase records without associated rollout plan\",\"No adoption tracking dashboard or reports\"]},\"2\":{\"must\":[\"2-3 pilot teams are designated with explicit AI adoption goals\",\"Adoption spreads through peer advocacy rather than a mandate, and the org can show where it spread\",\"Pilot metrics are defined and tracked (adoption rate, usage frequency, developer satisfaction)\"],\"should\":[\"Pilot results are shared with the broader organization\",\"Champion has direct access to leadership for escalation\"],\"prerequisites\":[],\"evidence\":[\"Pilot team designation document with goals and success criteria\",\"Champion role assignment with time allocation\",\"Pilot metrics dashboard showing tracked KPIs\"]},\"3\":{\"must\":[\"Platform team formally owns AI tooling (selection, provisioning, security, baseline configuration)\",\"Internal Developer Platform includes an AI layer (standardized agent setup, self-service provisioning)\",\"Standardized agent setup exists per team (every team has a working AI environment by default)\"],\"should\":[\"New developer onboarding includes AI tool setup that completes in under 30 minutes\",\"Platform team tracks adoption breadth (% of developers with active AI setup)\"],\"prerequisites\":[{\"label\":\"Organization L2 (AI Adoption Model) - pilot teams and champion must have validated AI tools before org-wide rollout\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Platform team charter or responsibility matrix including AI tooling ownership\",\"IDP configuration showing AI tool provisioning layer\",\"Standardized agent setup scripts or templates per team\"]},\"4\":{\"must\":[\"AI-assisted work is the default path, not an initiative: a large and stable majority of developers use agents daily\",\"Agent fleet management is a recognized discipline with defined practices\",\"Developer role has shifted toward supervising agents rather than authoring most changes\"],\"should\":[\"\\\"Span of control\\\" metric is tracked (how many agents a developer can effectively supervise)\",\"Organization benchmarks its adoption against external reference data rather than against its own launch week\"],\"prerequisites\":[{\"label\":\"Organization L3 (AI Adoption Model) - platform team ownership and standardized setup must be in place\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Organization L3 (Team Structure \u0026 Roles) - context engineer and platform engineer roles must exist\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Daily active usage rate showing a large, stable majority of developers\",\"Agent fleet management practices documentation\",\"Developer role descriptions reflecting agent supervision responsibilities\"]},\"5\":{\"must\":[\"Centralized agent orchestration system exists (\\\"Kubernetes for agents\\\")\",\"Developer role is \\\"human-at-the-wheel\\\" (strategic direction, not task-level involvement)\",\"Organization is optimized for agent throughput, not human throughput (meetings, processes, tooling all agent-aware)\"],\"should\":[\"Agent orchestration system handles scheduling, resource allocation, and failure recovery\",\"Organization measures agent utilization as a key infrastru
102cture metric\"],\"prerequisites\":[{\"label\":\"Organization L4 (AI Adoption Model) - AI-first culture and fleet management must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L5 (Agent Runtime \u0026 Sandboxing) - dedicated agent compute with auto-scaling required\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Agent orchestration system dashboard showing scheduling and resource management\",\"Organizational process documentation reflecting agent-first design\",\"Agent utilization metrics dashboard\"]}},\"Knowledge Management\":{\"1\":{\"must\":[\"The team has working knowledge of its systems\",\"Onboarding includes a README or equivalent starting point\"],\"should\":[\"Team acknowledges that tribal knowledge is a risk\",\"Some informal knowledge sharing exists (Slack threads, meeting notes)\"],\"prerequisites\":[],\"evidence\":[\"Documentation audit showing outdated or missing docs for key systems\",\"Onboarding feedback citing reliance on \\\"ask someone\\\" for critical information\"]},\"2\":{\"must\":[\"Documentation refresh initiative is active with measurable progress\",\"The team writes a decision record when it chooses an architecture or a major dependency\",\"Written onboarding path exists (new developer can self-serve key setup steps)\"],\"should\":[\"ADRs are indexed and searchable\",\"Onboarding path has been validated by at least one new hire completing it solo\"],\"prerequisites\":[],\"evidence\":[\"Documentation refresh tracking (issues, PRs, completion percentage)\",\"ADR directory in repository with recent entries\",\"Written onboarding guide with step-by-step instructions\"]},\"3\":{\"must\":[\"Documentation is treated as infrastructure (owned by engineering, not HR or PMO)\",\"Lint rules enforce conventions rather than relying on documentation alone (enforced \u003e suggested)\",\"A queryable map of the codebase (structure, ownership, change history) is operational\"],\"should\":[\"Documentation freshness is tracked (pages older than 90 days are flagged for review)\",\"Knowledge graph is integrated with agent context pipeline (agents query it at runtime)\"],\"prerequisites\":[{\"label\":\"Organization L2 (Knowledge Management) - ADRs and documentation refresh must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L2 (Context Engineering) - agent instruction files must exist for knowledge graph integration\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Documentation ownership in engineering team's responsibility matrix\",\"Lint rules enforcing conventions with corresponding documentation references\",\"Knowledge graph dashboard showing codebase coverage\"]},\"4\":{\"must\":[\"Context Fabric: MCP servers automatically feed institutional knowledge to agents\",\"Autonomous Requirements pipeline: unclear tickets are auto-expanded into specs with acceptance criteria\",\"Agents auto-update documentation when code changes (no manual doc maintenance)\"],\"should\":[\"Context Fabric covers 80%+ of active repositories\",\"Doc auto-update PRs are reviewed and merged within 24 hours\"],\"prerequisites\":[{\"label\":\"Organization L3 (Knowledge Management) - documentation-as-infrastructure and knowledge graph must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L3 (MCP \u0026 Tool Integration) - centralized MCP platform required for Context Fabric\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"MCP server configuration showing automated knowledge delivery to agents\",\"Autonomous Requirements pipeline with sample ticket-to-spec outputs\",\"Agent-authored documentation update PRs in git history\"]},\"5\":{\"must\":[\"Knowledge base is self-evolving (agents add, update, and validate knowledge entr
102ies continuously)\",\"Agent detects stale context, updates it, and validates the update - without human initiation\",\"Organizational memory is Git-backed, agent-readable, and provably current\"],\"should\":[\"Knowledge base freshness score exceeds 95% (% of entries updated within their defined freshness window)\",\"Self-evolving updates are validated against codebase to prevent knowledge drift\"],\"prerequisites\":[{\"label\":\"Organization L4 (Knowledge Management) - Context Fabric and autonomous docs must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L5 (Context Engineering) - self-healing context pipeline required for autonomous knowledge management\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Knowledge base with agent-authored entries and update timestamps\",\"Stale context detection and auto-update logs\",\"Git-backed knowledge store with provenance tracking\"]}},\"Team Structure \u0026 Roles\":{\"1\":{\"must\":[\"The team has standard engineering roles (developer, QA, PM)\",\"Senior developers review and fix AI-generated code\"],\"should\":[\"Team is open to experimenting with AI-assisted workflows\",\"At least one person informally champions AI tool usage\"],\"prerequisites\":[],\"evidence\":[\"Job descriptions showing traditional role definitions\",\"Code review comments showing seniors correcting AI-generated patterns\"]},\"2\":{\"must\":[\"AI champion is designated per team with allocated time (not just informal interest)\",\"Developers have been trained on how to give an agent a task\",\"Context engineer role exists (initial, possibly part-time) for maintaining agent instruction files\"],\"should\":[\"Champion has a regular cadence for sharing learnings across the team\",\"Training materials are documented and available for new hires\"],\"prerequisites\":[{\"label\":\"Organization L2 (AI Adoption Model) - internal champion must be identified at org level\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Team roster showing designated AI champion with time allocation\",\"Context engineer role assignment (even if combined with other duties)\",\"Training session records or materials\"]},\"3\":{\"must\":[\"Team's primary activity has shifted from writing code to evaluating and reviewing AI-generated code\",\"Platform Engineer role with AI tooling responsibility exists on the platform team\",\"Context Engineer is a full dedicated role (not part-time, not combined with other duties)\"],\"should\":[\"Role definitions are updated to reflect AI-augmented responsibilities\",\"Hiring criteria include AI tool proficiency\"],\"prerequisites\":[{\"label\":\"Organization L3 (AI Adoption Model) - platform team must own AI tooling for Platform Engineer role to exist\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Organization L2 (Team Structure \u0026 Roles) - champion and initial context engineer must be in place\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Platform Engineer job description including AI tooling responsibilities\",\"Context Engineer role as a dedicated position (headcount or full-time allocation)\",\"Time tracking showing majority of developer time on review/evaluation vs. writing\"]},\"4\":{\"must\":[\"Span of control is measured: how many parallel agents each developer effectively supervises\",\"Performance evaluation includes agent supervision effectiveness (not just personal code output)\",\"Developer role is formally defined as \\\"manager of agent fleet\\\"\"],\"should\":[\"A span-of-control limit is defined per role and derived from what the orchestrator can actually keep in context\",\"Agent supervision training is part of standard developer onboarding\"],\"prerequisites\":[{\"label\":\"Organization L3 (Team Structure \u0026 Roles) - context engineer and platform engineer roles must be established\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L4 (Coding Agent Usage) - parallel agents per developer must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Updated role descriptions defining developer as agent supervisor\",\"Span of control metrics dashboard\",\"Performance review criteria including agent supervision effectiveness\"]},\"5\":{\"must\":[\"PEV (Plan, Execute, Verify) loop is the standard workflow for all engineering tasks\",\"Non-coder contributors can produce software changes via agent interfaces\",\"Agentic Engineer role combines orchestration, supervision, and architecture responsibilities\"],\"should\":[\"Agentic Engineer career ladder exists with defined progression criteria\",\"Non-coder contribution rate is tracked as an organizational capability metric\"],\"prerequisites\":[{\"label\":\"Organization L4 (Team Structure \u0026 Roles) - developer-as-agent-manager role must be established\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Development L5 (Coding Agent Usage) - multi-agent orchestration must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Agentic Engineer role description with orchestration and supervision responsibilities\",\"PEV loop documentation and adoption evidence in team workflows\",\"Non-coder contributor logs showing software changes via agent interfaces\"]}}
102,\"Tech Debt \u0026 Modernization\":{\"1\":{\"must\":[\"The team is aware of its main tech-debt areas\",\"Legacy systems are kept stable\"],\"should\":[\"Team acknowledges tech debt exists and can enumerate major items\",\"Migration backlog exists (even if years long with no progress)\"],\"prerequisites\":[],\"evidence\":[\"Tech debt backlog with items older than 12 months and no progress\",\"Legacy system documentation (or lack thereof) showing avoidance patterns\"]},\"2\":{\"must\":[\"Tech debt is categorized and prioritized (severity, impact, effort)\",\"At least one manual migration attempt has been completed or is in progress\",\"OpenRewrite or equivalent automated refactoring tool has been evaluated or adopted for basic recipes\"],\"should\":[\"Tech debt reduction is allocated time in sprint planning (even if small)\",\"Migration attempts are documented with lessons learned\"],\"prerequisites\":[],\"evidence\":[\"Categorized tech debt backlog with priority ratings\",\"Completed or in-progress migration project documentation\",\"OpenRewrite configuration or evaluation report\"]},\"3\":{\"must\":[\"Continuous modernization: agents work on tech debt reduction in background (non-blocking to feature work)\",\"Library version bumps and dependency upgrades are automated via agent PRs\",\"OpenRewrite + agent combination is used for systematic refactoring campaigns\"],\"should\":[\"Agent tech debt PRs follow the same review process as feature PRs\",\"Depen
102dency freshness score is tracked (% of dependencies within N versions of latest)\"],\"prerequisites\":[{\"label\":\"Development L2 (Coding Agent Usage) - agents must be operational for agent-driven modernization\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L2 (CI/CD Pipeline) - CI must be fast enough for agent refactoring iteration\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Agent-authored tech debt reduction PRs in git history\",\"Automated dependency upgrade configuration (Renovate + agent, Dependabot + agent)\",\"OpenRewrite recipe configuration with agent integration\"]},\"4\":{\"must\":[\"Projects previously deemed \\\"too expensive to modernize\\\" are being modernized by agents at low cost\",\"Cross-repository migration agents operate across multiple codebases simultaneously\",\"Major version migrations (e.g., Java 8 to 21, Angular.js to Angular 17) are agent-driven\"],\"should\":[\"Cost-per-migration-PR is tracked and decreasing\",\"Cross-repo migrations complete within defined SLAs (e.g., 100 repos migrated in 30 days)\"],\"prerequisites\":[{\"label\":\"Organization L3 (Tech Debt \u0026 Modernization) - agent-driven modernization and OpenRewrite must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Infrastructure L3 (Agent Runtime \u0026 Sandboxing) - isolated environments required for cross-repo agent work\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Previously-stalled migration projects now in progress or completed with agent assistance\",\"Cross-repo migration agent logs showing multi-repository operation\",\"Major version migration PRs authored by agents with passing CI\"]},\"5\":{\"must\":[\"Tech debt is at near-zero steady state (new debt is paid down within the same sprint it is created)\",\"Agent fleet maintains, upgrades, and patches codebases 24/7 without human scheduling\",\"CVE remediation is autonomous: detect vulnerability, generate fix, test, and ship\"],\"should\":[\"Mean time from CVE disclosure to deployed fix is under 24 hours for critical vulnerabilities\",\"Tech debt score (measured by static analysis) has been stable or improving for 6+ months\"],\"prerequisites\":[{\"label\":\"Organization L4 (Tech Debt \u0026 Modernization) - cross-repo migration agents must be operational\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0},{\"label\":\"Delivery L4 (Merge \u0026 Deploy) - auto-merge and auto-deploy required for autonomous CVE remediation\",\"perspectiveSlug\":\"\",\"area\":\"\",\"level\":0}],\"evidence\":[\"Tech debt trend dashboard showing near-zero steady state\",\"Agent fleet activity logs showing 24/7 maintenance operations\",\"CVE remediation traces: detection to deployed fix with timestamps\"]}}}},\"reportData\":{\"development\":{\"Coding Agent Usage\":{\"1\":{\"guideSlug\":\"copilot-autocomplete\",\"guideTitle\":\"Copilot autocomplete\",\"targetItems\":[{\"text\":\"Copilot autocomplete\",\"guideSlug\":\"copilot-autocomplete\"},{\"text\":\"Chat in sidebar, ad-hoc questions\",\"guideSlug\":\"chat-in-sidebar-ad-hoc-questions\"},{\"text\":\"Agent runs without codebase context\",\"guideSlug\":\"no-integration-with-codebase-context\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team just got Copilot licenses. Everyone's excited, but after a month the buzz has faded. Usage is uneven - some developers love it, others turned it off after getting bad suggestions in their legacy Java codebase. Bob can't tell if the investment is paying off because there are no metrics.\\n\\n**What Bob should do:** Don't try to measure ROI yet at L1. Instead, focus on adoption: ensure every developer has the plugin installed, knows the keystrokes, and has used it for at least 2 weeks. The real measurement comes at L2, when standards (CLAUDE.md) make suggestions consistently useful. Bob's quick win: pick one team, help them write a CLAUDE.md file, and compare their experience to teams without one. That's the business case for L2.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah approved the Copilot budget but can't demonstrate value. Developers s
102ay \\\"it's nice\\\" but she has no data. Her stakeholders want numbers: time saved, code quality impact, developer satisfaction scores.\\n\\n**What Sarah should do:** At L1, the honest answer is that autocomplete provides convenience, not transformation. The transformation starts at L2-L3 when AI tools understand the codebase context. Sarah should use Copilot's dashboard for the one thing it answers honestly - how many seats are active, and how often - and stop short of the acceptance-rate figure on the same screen, which counts suggestions taken rather than code kept. Her real business case comes from moving to L3 where she can measure ITS (Iterations-to-Success) and CPI (Cost-per-Iteration).\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been using Copilot since beta. He's already frustrated with its limitations - it keeps suggesting patterns that violate the team's architecture decisions. He wants to fix this but doesn't know where to start.\\n\\n**What Victor should do:** Victor is the natural champion to push from L1 to L2. His next step is writing a `.cursorrules` or `CLAUDE.md` file that encodes the team's conventions: preferred patterns, forbidden anti-patterns, architecture decisions. Once that file exists in the repo, every developer's autocomplete improves immediately. Victor should present the before/after to Bob as evidence for broader AI tooling investment.\"}],\"gettingStarted\":[\"Install the plugin\",\"Learn the keystrokes\",\"Start in familiar territory\"],\"links\":[{\"title\":\"Quickstart for GitHub Copilot\",\"url\":\"https://docs.github.com/en/copilot/get-started\",\"domain\":\"docs.github.com\"},{\"title\":\"Best Practices for Using GitHub Copilot\",\"url\":\"https://docs.github.com/en/copilot/get-started/best-practices\",\"domain\":\"docs.github.com\"},{\"title\":\"Cursor Editor - Quickstart\",\"url\":\"https://cursor.com/docs/get-started/quickstart\",\"domain\":\"cursor.com\"},{\"title\":\"GitHub Copilot vs Cursor vs Codeium: Comparison 2026\",\"url\":\"https://www.digitalocean.com/resources/articles/github-copilot-vs-cursor\",\"domain\":\"digitalocean.com\"},{\"title\":\"From Autocomplete to Context: AI Code Completion in 2025\",\"url\":\"https://pieces.app/blog/ai-code-completion-tools\",\"domain\":\"pieces.app\"},{\"title\":\"METR Study: Measuring AI Impact on Developer Productivity\",\"url\":\"https://metr.org/blog/2025-07-10-early-2025-ai-experienced-os-dev-study/\",\"domain\":\"metr.org\"}]},\"2\":{\"guideSlug\":\"agent-in-ide-with-yolo-mode\",\"guideTitle\":\"Agent in IDE; autonomy set by a written rule, not a per-prompt click - and increasingly by the organisation rather than the developer (GitHub's enterprise-managed Copilot agent permissions for shell, files and network domains cannot be overridden by users)\",\"targetItems\":[{\"text\":\"Agent in IDE; autonomy set by a written rule, not a per-prompt click - and increasingly by the organisation rather than the developer (GitHub's enterprise-managed Copilot agent permissions for shell, files and network domains cannot be overridden by users)\",\"guideSlug\":\"agent-in-ide-with-yolo-mode\"},{\"text\":\"An agent instruction file ships with every active repository\",\"guideSlug\":\"claude-md-cursorrules-in-repo\"},{\"text\":\"Copilot + Claude Code in parallel\",\"guideSlug\":\"copilot-claude-code-in-parallel\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$13\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is starting to see real velocity numbers. Developers running agents on whole tasks are completing features faster, and she wants to measure this rigorously. But she's worried about quality - are they shipping more bugs along with the faster features?\\n\\n**What Sarah should do:** Track two metrics in parallel: PR throughput (up, because delegated agents accelerate implementation) and post-merge bug rate (should be stable or improving, because agents are also writing more tests). If bug rate increases alongside throughput, that's a sign the review gate is being skipped. Sarah should make the post-merge bug rate a standing metric in team reviews - not as a punitive measure, but as the quality signal that validates the throughput improvement. The goal is to show that delegated agent work improves speed without degrading quality, which is the business case for scaling it to more teams.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$14\"}],\"gettingStarted\":[\"Start on a clean branch\",\"Write a CLAUDE.md first\",\"Write the permission rules before you turn autonomy up\"],\"links\":[{\"title\":\"Claude Code - Permissions and
102Settings\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/settings\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Cursor Composer Documentation\",\"url\":\"https://cursor.com/docs/agent/overview\",\"domain\":\"cursor.com\"},{\"title\":\"Aider - AI Pair Programming in the Terminal\",\"url\":\"https://aider.chat/docs/usage/modes.html\",\"domain\":\"aider.chat\"},{\"title\":\"GitHub Copilot Workspace Overview\",\"url\":\"https://githubnext.com/projects/copilot-workspace\",\"domain\":\"githubnext.com\"},{\"title\":\"Claude Code Best Practices\",\"url\":\"https://www.anthropic.com/engineering/claude-code-best-practices\",\"domain\":\"anthropic.com\"},{\"title\":\"Cursor 3 - Agent-First IDE\",\"url\":\"https://www.cursor.com/blog/cursor-3\",\"domain\":\"cursor.com\"},{\"title\":\"Claude Code puts auto mode in the driver's seat - The Register\",\"url\":\"https://www.theregister.com/ai-and-ml/2026/08/10/claude-code-puts-auto-mode-in-the-drivers-seat/5285326\",\"domain\":\"theregister.com\"},{\"title\":\"Agentic Code Quality - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/agentic-code-quality\",\"domain\":\"addyo.substack.com\"}]},\"3\":{\"guideSlug\":\"agent-aware-coding-conventions-explicit-implicit\",\"guideTitle\":\"Code is written to be read by agents: explicit over implicit, obvious over clever\",\"targetItems\":[{\"text\":\"Code is written to be read by agents: explicit over implicit, obvious over clever\",\"guideSlug\":\"agent-aware-coding-conventions-explicit-implicit\"},{\"text\":\"Rules files per-team/per-repo\",\"guideSlug\":\"rules-files-per-team-per-repo\"},{\"text\":\"CLI agents as primary (Claude Code with Opus 5.5 or Sonnet 5.5, Codex on GPT-6 Sol, Cursor, Gemini CLI) with cheap and open-weight workers (GPT-6 Luna, DeepSeek V4.1 Flash, MiMo-V2.6, GLM-5.3, Kimi K3) for execution\",\"guideSlug\":\"cli-agents-claude-code-codex-as-primary\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$15\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$16\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$17\"}],\"gettingStarted\":[\"Find where the code hides its behaviour\",\"Categorise your existing conventions\",\"Rewrite for precision\"],\"links\":[{\"title\":\"Claude Code - Writing Effective CLAUDE.md Files\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Cursor Rules - Best Practices\",\"url\":\"https://cursor.com/docs/context/rules\",\"domain\":\"cursor.com\"},{\"title\":\"Anthropic - Context Engineering for Agents\",\"url\":\"https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"Writing Machine-Readable Documentation\",\"url\":\"https://github.blog/developer-skills/github-education/how-to-write-better-prompts-for-github-copilot/\",\"domain\":\"github.blog\"},{\"title\":\"Claude Code Best Practices\",\"url\":\"https://www.anthropic.com/engineering/claude-code-best-practices\",\"domain\":\"anthropic.com\"}]},\"4\":{\"guideSlug\":\"one-shot-unattended-agents-stripe-minions-model\",\"guideTitle\":\"Scheduled / unattended agents + three-tier routing: a System One decision model decides (TypeSafe Jev - routing, compaction, safety gates at milliseconds), a cheap model executes, the frontier plans; re-costed monthly per task, not per token (Copilot Auto tiers, Uber's cheap subagents, Stripe Minions)\",\"targetItems\":[{\"text\":\"Scheduled / unattended agents + three-tier routing: a System One decision model decides (TypeSafe Jev - routing, compaction, safety gates at milliseconds), a cheap model executes, the frontier plans; re-costed monthly per task, not per token (Copilot Auto tiers, Uber's cheap subagents, Stripe Minions)\",\"guideSlug\":\"one-shot-unattended-agents-stripe-minions-model\"},{\"text\":\"Slack/CLI/Web/PagerDuty invocation â PR\",\"guideSlug\":\"slack-cli-web-invocation-pr\"},{\"text\":\"3-5 parallel agents per developer + merge queues for agent fleets; the ceiling is orchestrator context pollution, not token cost (cap batches at 2-4, no concurrent repo-wide git ops)\",\"guideSlug\":\"3-5-parallel-agents-per-developer\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob has been hearing about \\\"autonomous agents\\\" for months and is excited about the concept but nervous about the execution. His team is at L3 - they use CLI agents well, have mature CLAUDE.md files, and good test coverage. He wants to pilot unattended agents but doesn't know
102where to start safely.\\n\\n**What Bob should do:** The safest starting point is dependency upgrades. Pick one repository with good test coverage, identify a dependency that's two major versions behind, and write the task spec: \\\"Upgrade library X from version Y to version Z. Fix all type errors and test failures that result. Do not change business logic. PR title format: `chore: upgrade X to Z`.\\\" Run 3 agents across 3 different dependencies. Review the resulting PRs. Bob will discover the real failure modes - not catastrophic, because the task is scoped and the tests are the safety net. This pilot produces concrete learnings about task spec quality and sandbox configuration that generalize to more ambitious tasks.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah can see that unattended agents have significant ROI potential - if an agent can complete a task while a developer is doing something else, that's pure throughput multiplication. But she needs to measure this to justify the infrastructure investment (sandboxes, orchestration tooling).\\n\\n**What Sarah should do:** The key metric for unattended agents is \\\"agent success rate\\\" - the percentage of agent runs that produce a mergeable PR without human intervention. Track this alongside task type, CLAUDE.md version, and test coverage. A high success rate on a task type means that type is ready for production-scale automation. Sarah should also track \\\"developer time per PR\\\" for unattended-agent vs. manually-implemented PRs: the comparison is total calendar time from task initiation to merge, not active developer time. Unattended agents will show dramatically shorter calendar time for suitable tasks, which is the business case for the infrastructure investment.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has already run unattended agents manually (launching Claude Code in YOLO mode and walking away). He's seen it work well for test generation and dependency upgrades. He wants to build the automation layer that makes this scalable and repeatable.\\n\\n**What Victor should do:** Victor should build the \\\"minion launcher\\\" - a simple CLI tool that takes a task specification template, spins up a git worktree, runs Claude Code with the appropriate flags and context, and creates a PR when complete. Start simple: a bash script wrapping Claude Code and `gh pr create`. The key design decisions are: how to pass task context to the agent, how to detect success vs. failure, and how to structure the PR for easy review. Victor should also define the company's standard task specification format - a YAML or Markdown template that captures task scope, success criteria, and out-of-scope constraints. This format becomes the interface between humans (who write task specs) and agents (who execute them).\"}],\"gettingStarted\":[\"Identify suitable task categories\",\"Build the sandbox\",\"Write task specifications as structured prompts\"],\"links\":[{\"title\":\"Claude Code - Running Agents Autonomously\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/settings\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Stripe Engineering - Agentic AI at Scale\",\"url\":\"https://stripe.dev/blog/minions-stripes-one-shot-end-to-end-coding-agents\",\"domain\":\"stripe.dev\"},{\"title\":\"Claude Code - GitHub Actions Integration\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/github-actions\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Git Worktrees - Running Parallel Workspaces\",\"url\":\"https://git-scm.com/docs/git-worktree\",\"domain\":\"git-scm.com\"},{\"title\":\"Claude Code Best Practices - Anthropic Engineering\",\"url\":\"https://www.anthropic.com/engineering/claude-code-best-practices\",\"domain\":\"anthropic.com\"},{\"title\":\"Ephemeral Sandboxes for AI Agents\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/security\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Building a software factory for AI SDK - Vercel\",\"url\":\"https://vercel.com/blog/building-a-software-factory-for-ai-sdk\",\"domain\":\"vercel.com\"},{\"title\":\"Practical Loop Engineering - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/practical-loop-engineering\",\"domain\":\"addyo.substack.com\"},{\"title\":\"Incident report: unsanctioned agent behaviour during cyber testing - UK AISI\",\"url\":\"https://www.aisi.gov.uk/blog/incident-report-unsanctioned-agent-behaviour-during-cyber-testing\",\"domain\":\"aisi.gov.uk\"}]},\"5\":{\"guideSlug\":\"multi-agent-orchestration-gas-town-custom\",\"guideTitle\":\"Multi-agent orchestration (Claude Code dynamic workflows - the Bun-in-Rust model, Gas Town / custom)\",\"targetItems\":[{\"text\":\"Multi-agent orchestration (Claude Code dynamic workflows - the Bun-in-Rust model, Gas Town / custom)\",\"guideSlug\":\"multi-agent-orchestration-gas-town-custom\"},{\"text\":\"Planner â Worker hierarchy\",\"guideSlug\":\"planner-worker-hierarchy\"},{\"text\":\"Fleet size bounded by compute and review capacity, not by tooling\",\"guideSlug\":\"hundreds-of-agents-on-codebase-1000-commits-h\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$18\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is watching the L5 trend in the industry and wants to understand the ROI before committing to the engineering investment. Multi-agent orchestration sounds transformative but also sounds expensive to build.\\n\\n**What Sarah should do:** Sarah's analysis should compare two costs: (1) the engineering investment to build the orchestration system (one-time, 4-8 weeks of engineering time), and (2) the ongoing human coordination cost it replaces (the time senior developers spend orchestrating parallel agents manually at L4). If senior developers are spending 2+ hours per day on agent coordination, an orchestration system that reduces this to 30 minutes of review has a payback period measured in weeks. Sarah should also look at task types: which tasks require the most agent coordination? Those are the candidates for the first orchestration pipeline. A focused first pipeline targeting a high-frequency task type will demonstrate ROI faster than a general-purpose orchestration system.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$19\"}],\"gettingStarted\":[\"Study existing orchestration frameworks\",\"Define your agent roles\",\"Build the communication layer\"],\"links\":[{\"title\":\"Anthropic - Building Effe
102ctive Agents\",\"url\":\"https://docs.anthropic.com/en/docs/build-with-claude/agents\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Claude Code - Multi-Agent Networks\",\"url\":\"https://code.claude.com/docs/en/common-workflows\",\"domain\":\"code.claude.com\"},{\"title\":\"Gas Town - Open Source Agent Orchestration\",\"url\":\"https://github.com/anthropics/claude-code\",\"domain\":\"github.com\"},{\"title\":\"Anthropic Research - Multi-Agent Frameworks\",\"url\":\"https://www.anthropic.com/research/building-effective-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"Stripe Engineering - Internal AI Agent Systems\",\"url\":\"https://stripe.dev/blog/minions-stripes-one-shot-end-to-end-coding-agents\",\"domain\":\"stripe.dev\"},{\"title\":\"LangGraph - Stateful Multi-Agent Orchestration\",\"url\":\"https://langchain-ai.github.io/langgraph/\",\"domain\":\"langchain-ai.github.io\"},{\"title\":\"Code w/ Claude 2026 - fleets, Outcomes and Dreaming\",\"url\":\"https://simonwillison.net/2026/May/6/code-w-claude-2026/\",\"domain\":\"simonwillison.net\"}]}},\"Context Engineering\":{\"1\":{\"guideSlug\":\"zero-context-agent-sees-only-the-open-file\",\"guideTitle\":\"Agent works from the currently open file\",\"targetItems\":[{\"text\":\"Agent works from the currently open file\",\"guideSlug\":\"zero-context-agent-sees-only-the-open-file\"},{\"text\":\"Project knowledge lives with people, not docs\",\"guideSlug\":\"tribal-knowledge-in-people-s-heads\"},{\"text\":\"Onboarding leans on the existing README and people\",\"guideSlug\":\"readme-outdated-for-6-months\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has had Copilot for three months. He keeps hearing that AI tools are \\\"hit or miss.\\\" A few developers love it; others say it generates garbage. When he investigates, the pattern is clear: the developers who love it are using it for greenfield scripts and isolated utility functions. The developers who hate it are using it in the core domain model, where suggestions consistently violate the team's carefully evolved patterns.\\n\\n**What Bob should do:** Bob doesn't need to restrict AI usage - he needs to help his team understand where L1 tools break down and why. The developers getting good results are unknowingly working around the zero-context problem by choosing self-contained tasks. Bob's action: run a workshop where teams map their codebase into \\\"high-coupling\\\" and \\\"low-coupling\\\" zones, and help them understand that L1 tools work best in the low-coupling areas. Then make the case for L2 investment: a CLAUDE.md file that encodes the domain model's conventions.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking AI tool usage and sees that developers in the platform team use autocomplete constantly, while developers in the core services team barely use it. The productivity delta is significant, but she can't explain why to her stakeholders. The tools are the same; the usage is not.\\n\\n**What Sarah should do:** The difference is context complexity. Platform code tends to be more self-contained; core service code is deeply coupled and convention-driven. Sarah should frame this for stakeholders not as \\\"some teams adopted AI\\\" but as \\\"the teams with the most to gain from AI are blocked by a context engineering problem.\\\" The investment case for L2 (CLAUDE.md, written conventions) is that it unlocks AI productivity in the teams where productivity gains are most valuable.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been dealing with this problem for months. He's built his own workflow: before using an agent on anything complex, he writes a \\\"context preamble\\\" - a few paragraphs describing the module's responsibilities, the patterns it uses, and what it must not do. It works, but it's manual, per-session, and none of his colleagues know about it.\\n\\n**What Victor should do:** Victor already has the content for a CLAUDE.md file - it's in those context preambles he keeps writing. His action is to take the three most common preambles, formalize them, and commit them as CLAUDE.md (or module-level `AGENT.md` files) in the repository. Then he should demonstrate the before/after to Bob: here's the suggestion quality without the file, here's the quality with it. That's the concrete evidence Bob needs to fund L2 investment across the team.\"}],\"gettingStarted\":[\"Recognize the zero-context boundary\",\"Compensate manually\",\"Keep prompts concrete\"],\"links\":[{\"title\":\"Anthropic CLAUDE.md Documentation\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"GitHub Copilot Context Limitations\",\"url\":\"https://docs.github.com/en/copilot/using-github-copilot/getting-code-suggestions-in-your-ide-with-github-copilot\",\"domain\":\"docs.github.com\"},{\"title\":\"How LLMs Use Context Windows - OpenAI\",\"url\":\"https://platform.openai.com/docs/guides/text-generation\",\"domain\":\"platform.openai.com\"}
102,{\"title\":\"Building Effective AI Agents - Anthropic\",\"url\":\"https://www.anthropic.com/research/building-effective-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"Cursor - Understanding Context\",\"url\":\"https://cursor.com/docs/context/codebase-indexing\",\"domain\":\"cursor.com\"}]},\"2\":{\"guideSlug\":\"claude-md-with-basic-project-info\",\"guideTitle\":\"CLAUDE.md pruned to repo-specific gotchas and committed to the repo (repos without committed agent config saw twice the cognitive-complexity growth: +53% vs +27%)\",\"targetItems\":[{\"text\":\"CLAUDE.md pruned to repo-specific gotchas and committed to the repo (repos without committed agent config saw twice the cognitive-complexity growth: +53% vs +27%)\",\"guideSlug\":\"claude-md-with-basic-project-info\"},{\"text\":\"Written coding conventions\",\"guideSlug\":\"written-coding-conventions\"},{\"text\":\"Agent instruction files + Skills as the unit of reuse (SKILL.md alongside CLAUDE.md, AGENTS.md, llms.txt; cross-agent registries)\",\"guideSlug\":\"agent-instruction-files-in-repo-60k-repos-on-github\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has been complaining that AI suggestions are inconsistent - they work great for some developers and not others. He suspects it has to do with the senior engineers' habit of providing detailed context in their prompts, while junior engineers just describe the immediate task. He's right, but he doesn't know how to systematize it.\\n\\n**What Bob should do:** Bob should sponsor a \\\"CLAUDE.md sprint\\\" - one dedicated afternoon where one senior engineer per team produces the first version of a CLAUDE.md for their primary repository. Bob's goal for this sprint is not perfection; it's existence. A basic CLAUDE.md committed to every main repository within the quarter. He should then compare the teams that complete the sprint against those that haven't on signals that survive review - how often an AI-assisted change comes back for a correction pass, and the static-analysis warning count on new code - which gives him the ROI data he needs to continue investing in context engineering.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been trying to understand why AI adoption is so uneven across teams. She runs a survey and finds that developers in teams without shared context files (CLAUDE.md, .cursorrules) rate AI tools much lower than developers in teams that have them. The pattern is clear, but she hasn't been able to make the business case for investing time in these files.\\n\\n**What Sarah should do:** Sarah should calculate the \\\"context tax\\\" - the time each developer spends per week manually providing project context in prompts before the agent can help effectively. Even if the estimate is rough (say, 15 minutes per developer per day in a team without CLAUDE.md), the math quickly justifies the one-time investment of writing the file. A 10-person team spending 15 minutes/day = 2.5 engineer-hours per day = 600+ engineer-hours per year spent on a problem a 200-line file could fix.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been the unofficial \\\"AI whisperer\\\" on his team - developers come to him when they're getting bad suggestions, and he shows them the right context to provide. He's been doing this for months and realizes he's essentially teaching the same 10 things over and over.\\n\\n**What Victor should do:** Victor's teaching material is the content of a CLAUDE.md file. He should take the 10 things he consistently explains, write them up as a structured context file, commit it to the main repositories, and point developers to it instead of repeating himself. Then he should go further: write a brief guide for the team on how to interpret and extend the CLAUDE.md file, and propose a quarterly review process where the team discusses what should be added based on recent agent errors. Victor is the right person to drive this from technical judgment; Bob should provide the organizational mandate.\"}],\"gettingStarted\":[\"Create the file\",\"Write the project overview section\",\"Document the tech stack\"],\"links\":[{\"title\":\"CLAUDE.md Documentation - Anthropic Claude Code\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"The New Rules of Context Engineering for Claude 5 Generation Models - Anthropic\",\"url\":\"https://claude.com/blog/the-new-rules-of-context-engineering-for-claude-5-generation-models\",\"domain\":\"claude.com\"},{\"title\":\"GitHub Copilot Custom Instructions\",\"url\":\"https://docs.github.com/en/copilot/customizing-copilot/adding-custom-instructions-for-github-copilot\",\"domain\":\"docs.github.com\"},{\"title\":\"Cursor Rules Documentation\",\"url\":\"https://cursor.com/docs/context/rules\",\"domain\":\"cursor.com\"},{\"title\":\"Awesome CLAUDE.md - Community Examples\",\"url\":\"https://github.com/josix/awesome-claude-md\",\"domain\":\"github.com\"},{\"title\":\"How to Write Effective AI Context Files - Sourcegraph Blog\",\"url\":\"https://sourcegraph.com/blog/anatomy-of-a-coding-assistant\",\"domain\":\"sourcegraph.com\"},{\"title\":\"Committed agent configuration and code quality across 441 repos (arXiv 2608.25241)\",\"url\":\"https://arxiv.org/abs/2608.25241\",\"domain\":\"arxiv.org\"},{\"title\":\"Audit your Agent files - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/audit-your-agent-files\",\"domain\":\"addyo.substack.com\"},{\"title\":\"ChainDrop: when opening a repository becomes execution - Pillar Security\",\"url\":\"https://www.pillar.security/blog/chaindrop-when-opening-a-repository-becomes-execution\",\"domain\":\"pillar.security\"}]},\"3\":{\"guideSlug\":\"mcp-servers-architecture-ownership-sla\",\"guideTitle\":\"Architecture, ownership and operational context reach the agent through governed servers rather than pasted text\",\"targetItems\":[{\"text\":\"Architecture, ownership and operational context reach the agent through governed servers rather than pasted text\",\"guideSlug\":\"mcp-servers-architecture-ownership-sla\"},{\"text\":\"Retrieval is deterministic and cheap: the agent searches and extracts on demand instead of being fed a pre-built index\",\"guideSlug\":\"5-level-context-system-code-org-historical-operational\"},{\"text\":\"Context budgeting with a standing eviction rule and a hard cap (Uber: 400K tokens with auto-compaction); agent files hand-written and audited, not generated - machine-generated context files did worse than none at 20%+ more cost; and any security rule in CLAUDE.md is backed by a technical control (only 4.4% are)\",\"guideSlug\":\"context-budgeting-token-economy\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$1a\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking agent suggestion quality and sees a pattern: suggestions about the current state of the system (who owns what, what's deployed, what the current schema looks like) are consistently worse than suggestions about how to write code. The coding quality has improved with CLAUDE.md; the operational knowledge quality hasn't.\\n\\n**What Sarah should do:** Sarah should frame the MCP server investment in terms of two ROI streams: (1) reduced time agents spend making mistakes about current organizational state, and (2) reduced time developers spend looking up that state manually. Both are measurable. She should also note the compounding nature: as agents take on more complex multi-step tasks (L4), the quality of their operational context becomes increasingly critical. MCP server investment at L3 is
102infrastructure that enables L4 and L5 capabilities.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been manually providing operational context in every agent session: the current schema migration state, which services are currently degraded, what the team's sprint focus is. He's essentially acting as a human MCP server. He knows this doesn't scale and has been researching the MCP protocol to understand what building a real MCP server would take.\\n\\n**What Victor should do:** Victor should build the first internal MCP server as a proof of concept, then use it to demonstrate the productivity delta to Bob. A minimal viable MCP server - say, one that exposes the current database schema and service health status - can be built in a day using the official MCP SDK. Victor should instrument it (log queries, measure latency), deploy it to the team's dev environment, and measure how much time it saves compared to his current manual-context approach. That measurement is the business case for operationalizing MCP infrastructure.\"}],\"gettingStarted\":[\"Identify your highest-value context sources\",\"Stand up a simple MCP server\",\"Define ownership explicitly\"],\"links\":[{\"title\":\"Model Context Protocol - Anthropic Documentation\",\"url\":\"https://modelcontextprotocol.io/introduction\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"MCP TypeScript SDK - GitHub\",\"url\":\"https://github.com/modelcontextprotocol/typescript-sdk\",\"domain\":\"github.com\"},{\"title\":\"MCP Python SDK - GitHub\",\"url\":\"https://github.com/modelcontextprotocol/python-sdk\",\"domain\":\"github.com\"},{\"title\":\"Stripe Agent Toolkit - MCP Integration\",\"url\":\"https://github.com/stripe/agent-toolkit\",\"domain\":\"github.com\"},{\"title\":\"Awesome MCP Servers - Community List\",\"url\":\"https://github.com/punkpeye/awesome-mcp-servers\",\"domain\":\"github.com\"},{\"title\":\"Building Production MCP Servers - Anthropic Blog\",\"url\":\"https://www.anthropic.com/news/model-context-protocol\",\"domain\":\"anthropic.com\"}]},\"4\":{\"guideSlug\":\"byoc-org-pushes-context-to-the-agent\",\"guideTitle\":\"BYOC: org PUSHES context to the agent\",\"targetItems\":[{\"text\":\"BYOC: org PUSHES context to the agent\",\"guideSlug\":\"byoc-org-pushes-context-to-the-agent\"},{\"text\":\"A queryable map of code structure, ownership and change history\",\"guideSlug\":\"knowledge-graph-graph-buddy-codetale\"},{\"text\":\"Spec-Driven Development: AGENTS.md as shared spec interface (Spec Kit constitution.md, Andrew Ng + JetBrains, Thoughtworks)\",\"guideSlug\":\"autonomous-requirements-ticket-spec-acceptance-tests-auto\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has MCP servers providing organizational context (service ownership, deployment status, sprint data) and developers are getting good results at L3. But he notices that agent sessions still have a significant \\\"warmup\\\" phase - the first few exchanges in every session are the agent asking clarifying questions about the project context. This warmup is eating into the time savings that agents are supposed to provide.\\n\\n**What Bob should do:** Bob should sponsor the development of a task-type-aware context assembly pipeline. The target: when a developer assigns a Jira ticket to an agent, the pipeline automatically assembles a context package for that ticket type and injects it at session start. The success metric is eliminating the warmup phase: agents should start producing actionable suggestions in the first exchange, not the third or fourth. Bob should prioritize the 2-3 most common ticket types first - the ones where the warmup overhead is most expensive - and measure the before/after.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking ITS (Iterations-to-Success) across agent workflows and sees high variance. Some tasks complete in 2-3 iterations; others take 8-10. When she investigates the high-iteration cases, the pattern is almost always the same: the agent spent several early iterations establishing context before it could start working effectively. The context assembly was manual and inconsistent.\\n\\n**What Sarah should do:** Sarah should present BYOC as an ITS reduction initiative. If she can show that 40% of agent iterations are \\\"context establishment\\\" rather than \\\"task execution,\\\" and that automated context assembly could eliminate most of those, she has a clear ROI calculation: ITS reduction times average iteration cost = direct productivity gain. She should work with engineering to build a proof of concept on the highest-volume agent task type, measure the ITS improvement, and use that to fund the full BYOC pipeline build-out.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$1b\"}],\"gettingStarted\":[\"Catalog your agent task types\",\"Map context needs per task type\",\"Build the assembly pipeline\"],\"links\":[{\"title\":\"Building Effe
102ctive Agents - Anthropic\",\"url\":\"https://www.anthropic.com/research/building-effective-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"Model Context Protocol - Anthropic\",\"url\":\"https://modelcontextprotocol.io/introduction\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"Claude Code: Sub-agents and Context Management\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/overview\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Prompt Caching - Anthropic API\",\"url\":\"https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"LangChain: Context Management in Agent Workflows\",\"url\":\"https://python.langchain.com/docs/concepts/\",\"domain\":\"python.langchain.com\"}]},\"5\":{\"guideSlug\":\"persistent-agent-identity-memory-beads-git\",\"guideTitle\":\"Persistent agent identity + memory (Beads/Git, Open Memory Protocol; treat auto-memory as an exfiltration surface - \\\"Memory Heist\\\")\",\"targetItems\":[{\"text\":\"Persistent agent identity + memory (Beads/Git, Open Memory Protocol; treat auto-memory as an exfiltration surface - \\\"Memory Heist\\\")\",\"guideSlug\":\"persistent-agent-identity-memory-beads-git\"},{\"text\":\"Production telemetry â context auto-update\",\"guideSlug\":\"production-telemetry-context-auto-update\"},{\"text\":\"Stale context is detected and refreshed before an agent runs on it\",\"guideSlug\":\"self-healing-context-agent-detects-stale-docs-updates\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$1c\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$1d\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been writing context preambles for his agent sessions for months. He notices that he's writing the same five things every time: the payment retry logic, the user repository index constraint, the circular dependency issue, the event ordering guarantee, and the module ownership boundary rule. He could almost recite them from memory. He's frustrated that he has to keep repeating himself to the agent.\\n\\n**What Victor should do:** Victor should convert his mental model of \\\"things I always have to explain\\\" into a Beads memory file. Write each of those five items as a structured memory record, commit them to the repository, and configure his agent sessions to load those records at start. Then stop writing them in his context preambles - let the memory do the work. After a month, review the memory records: have new ones accumulated? Are old ones still accurate? This exercise will give Victor concrete experience with memory maintenance, which he can then systematize for the rest of the team.\"}],\"gettingStarted\":[\"Define your memory schema\",\"Create the memory directory structure\",\"Teach the agent to write memory\"],\"links\":[{\"title\":\"CLAUDE.md Memory Documentation - Anthropic\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Mem0: Memory Layer for AI Agents\",\"url\":\"https://mem0.ai/\",\"domain\":\"mem0.ai\"},{\"title\":\"MemGPT: LLMs as Operating Systems with External Memory\",\"url\":\"https://memgpt.ai/\",\"domain\":\"memgpt.ai\"},{\"title\":\"Zep: Long-Term Memory for AI Assistants\",\"url\":\"https://www.getzep.com/\",\"domain\":\"getzep.com\"},{\"title\":\"Building Long-Term Memory for AI Agents - LangChain Blog\",\"url\":\"https://blog.langchain.com/memory-for-agents/\",\"domain\":\"blog.langchain.com\"}]}},\"Code Review \u0026 Quality\":{\"1\":{\"guideSlug\":\"manual-review-of-100-code\",\"guideTitle\":\"Every PR gets human review\",\"targetItems\":[{\"text\":\"Every PR gets human review\",\"guideSlug\":\"manual-review-of-100-code\"},{\"text\":\"Review turnaround measured in hours; 78.9% of agentic PRs pass through a single reviewer\",\"guideSlug\":\"review-bottleneck-2h-waiting-for-feedback\"},{\"text\":\"AI and human code share one review path\",\"guideSlug\":\"no-distinction-between-ai-vs-human-code\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team of 50 engineers has a review SLA of \\\"24 hours,\\\" but in practice PRs often wait 2-3 days. Senior engineers complain about review load. Junior developers complain about slow feedback. Deployment frequency is limited because code sits in review queues rather than flowing to production.\\n\\n**What Bob should do:** Bob needs to recognize that 100% manual review is a structural problem, not a process discipline problem. Telling engineers to \\\"review faster\\\" or \\\"submit smaller PRs\\\" is a band-aid. Bob's real move is to invest in tools that reduce review burden: a shared linter configuration (this week, free), then an AI review tool like CodeRabbit or GitHub Copilot Reviews (this month). The goal is not to eliminate human review - it's to make human reviewers faster and more effective by handling the routine checks automatically. Bob should present this as \\\"enabling senior engineers to do higher-value work\\\" rather than \\\"replacing review,\\\" because that's accurate.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is tracking PR cycle time as part of her developer productivity metrics. The data shows that the median PR spends 18 hours in review before merging. Her dashboard shows this is the single biggest contributor to long cycle times. Her stakeholders want to know what she's going to do about it.\\n\\n**What Sarah should do:** The 18-hour review time is a solvable problem, but not by adding more reviewers. The solution is reducing the review surface area through automation. Sarah should propose a two-step investment: (1) deploy a shared linter configuration with CI enforcement, which eliminates style-related review comments immediately, and (2) trial an AI review tool (CodeRabbit has a free tier) on one team for 30 days. The expected outcome is a 30-50% reduction in review time for that team. If that experiment succeeds, Sarah has the business case to roll out to all teams.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been the primary reviewer for 60-70% of the team's PRs because he's the one who knows the codebase best. He's spending 3 hours a day reviewing code and is frustrated that he can't make progress on his own architectural work. He's seen AI review tools mentioned in conference talks but doesn't know
102which ones are worth the time to evaluate.\\n\\n**What Victor should do:** Victor should immediately evaluate CodeRabbit or GitHub Copilot Reviews by enabling one of them on the repository. The specific tool matters less than getting the data: does an AI first pass catch the kinds of issues Victor is currently catching? If the AI handles 60% of what Victor would have commented on, that's 60% less time in review. Victor should also recognize that being the only reviewer for 70% of PRs is a bus-factor problem - the solution isn't to review faster, it's to document the standards he's applying so that other reviewers (human and AI) can apply them consistently.\"}],\"gettingStarted\":[\"Measure your baseline\",\"Size your PRs deliberately\",\"Separate trivial from substantive changes\"],\"links\":[{\"title\":\"Google's Engineering Practices: What to Look For in a Code Review\",\"url\":\"https://google.github.io/eng-practices/review/reviewer/looking-for.html\",\"domain\":\"google.github.io\"},{\"title\":\"Accelerate: The Science of Lean Software and DevOps (book)\",\"url\":\"https://itrevolution.com/product/accelerate/\",\"domain\":\"itrevolution.com\"},{\"title\":\"LinearB: Engineering Metrics - PR Cycle Time\",\"url\":\"https://linearb.io/blog/pr-cycle-time\",\"domain\":\"linearb.io\"},{\"title\":\"The Code Review Bottleneck (DX Research)\",\"url\":\"https://getdx.com/research/measuring-developer-productivity-with-the-dx-core-4/\",\"domain\":\"getdx.com\"},{\"title\":\"GitHub Copilot Code Review\",\"url\":\"https://docs.github.com/en/copilot/using-github-copilot/code-review/using-copilot-code-review\",\"domain\":\"docs.github.com\"}]},\"2\":{\"guideSlug\":\"ai-assisted-review-suggestions\",\"guideTitle\":\"AI-assisted review suggestions (CodeRabbit, Qodo, Claude Security beta)\",\"targetItems\":[{\"text\":\"AI-assisted review suggestions (CodeRabbit, Qodo, Claude Security beta)\",\"guideSlug\":\"ai-assisted-review-suggestions\"},{\"text\":\"Basic linter rules\",\"guideSlug\":\"basic-linter-rules\"},{\"text\":\"Diff awareness - reviewer knows it's AI code; reject code the human can't understand even if CI is green; humans write the PR description (the why)\",\"guideSlug\":\"diff-awareness-reviewer-knows-it-s-ai-code\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team receives an average of 45 PRs per day across 12 repositories. His senior engineers are spending 2-3 hours daily in code review, time he'd rather they spend on design and architecture. He's heard about AI review tools but hasn't acted because he's worried about setup complexity and false positives annoying his team.\\n\\n**What Bob should do:** Bob should approve a 30-day trial of CodeRabbit on the team's two highest-traffic repositories. The setup is under an hour, the cost is low, and the trial gives concrete data: are AI comments useful? Do they reduce human review time? Bob should ask his tech leads to report on this after 30 days with two metrics: PR cycle time (did it decrease?) and reviewer time spent (did senior engineers get hours back?). If both numbers improve, the case for expanding and investing in configuration is easy to make.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tasked with reducing PR cycle time from 22 hours to 12 hours within a quarter. She knows human reviewers are the bottleneck but doesn't want to mandate faster review - she wants to reduce the work required per review. She's looking for a tool-based solution.\\n\\n**What Sarah should do:** AI-assisted review is the highest-leverage intervention for reducing PR cycle time at L2. Sarah should structure a controlled experiment: deploy an AI reviewer on team A (control: team B keeps current process), measure both teams' PR cycle time weekly for 6 weeks. If the AI reviewer reduces time-to-first-review (by providing instant first-pass comments) and reduces revision cycles (by catching issues before human review), Sarah has the data she needs for both the business case and the rollout. The metric she should track: median hours from PR creation to \\\"all comments resolved and approved.\\\"\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has started manually pasting PR diffs into Claude and asking for a review before submitting. He's found it catches 30-40% of issues that would otherwise appear as reviewer comments, saving him embarrassment and revision cycles. He wants to automate this and share it with the team.\\n\\n**What Victor should do:** Victor should take his manual Claude review process and replicate it with a formal AI review tool. He can use Claude's API to build a lightweight GitHub Action that runs on PR creation, sends the diff to Claude with a prompt that includes the team's conventions, and posts the response as a PR comment. This can be done in a day and would give the whole team the same review quality Victor is getting manually. Victor should then evaluate CodeRabbit or Copilot Reviews alongside his custom solution - they offer features (inline comments on specific lines, configuration persistence, review memory) that a simple API call doesn't.\"}],\"gettingStarted\":[\"Choose a tool and enable it\",\"Configure with your conventions\",\"Review the first week of AI comments as a team\"],\"links\":[{\"title\":\"CodeRabbit: AI Code Reviews\",\"url\":\"https://coderabbit.ai/docs\",\"domain\":\"coderabbit.ai\"},{\"title\":\"GitHub Copilot Code Review Documentation\",\"url\":\"https://docs.github.com/en/copilot/using-github-copilot/code-review/using-copilot-code-review\",\"domain\":\"docs.github.com\"},{\"title\":\"Sourcery: AI Code Reviews for Python\",\"url\":\"https://docs.sourcery.ai/Guides/Getting-Started/GitHub/\",\"domain\":\"docs.sourcery.ai\"},{\"title\":\"Amazon CodeGuru Reviewer\",\"url\":\"https://docs.aws.amazon.com/codeguru/latest/reviewer-ug/welcome.html\",\"domain\":\"docs.aws.amazon.com\"},{\"title\":\"Google Research: Automating Code Review Activities\",\"url\":\"https://research.google/pubs/resolving-code-review-comments-with-machine-learning/\",\"domain\":\"research.google\"}]},\"3\":{\"guideSlug\":\"lint-as-architecture-standards-enforced-rules\",\"guideTitle\":\"Lint-as-architecture (standards = enforced rules; Vercel Konsistent for agents and humans)\",\"targetItems\":[{\"text\":\"Lint-as-architecture (standards = enforced rules; Vercel Konsistent for agents and humans)\",\"guideSlug\":\"lint-as-architecture-standards-enforced-rules\"},{\"text\":\"AI review agent as first pass (self-verification; adversarial verification pass on whole-repo scans)\",\"guideSlug\":\"ai-review-agent-as-first-pass\"},{\"text\":\"Architecture guardrails: Bug â Codify â Lint Rule\",\"guideSlug\":\"architecture-guardrails-bug-codify-lint-rule\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$1e\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is tracking post-merge bugs (bugs that were introduced in a PR that passed review). She's noticed that many of them fall into patterns: the same architectural violations appearing repeatedly across different PRs and different developers. Her team is fixing the same class of bugs every quarter.\\n\\n**What Sarah should do:** Sarah should frame lint-as-architecture as a recurring bug elimination strategy. She can calculate the cost of the repeated architectural violations (incident time, on-call response, remediation) and compare it to the one-time cost of writing a lint rule (half a day of engineering time). The ROI for each rule is the expected incident cost eliminated, which is usually immediately obvious. Sarah should ask engineering leads to track \\\"classes of bugs eliminated by lint rules\\\" as a quality metric - a growing number indicates the team is progressively hardening its quality gate.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor keeps writing the same review comment: \\\"Don't call the database from the HTTP handler layer.\\\" He's written it on 12 PRs in the past 3 months. He knows exactly what the bad pattern looks like and what the correct pattern is. He's confident he can write a lint rule for it.\\n\\n**What Victor should do:** Victor should write the lint rule this sprint. It's a direct conversion of something he's doing manually (reviewing code for a pattern) into something automated (a lint rule that checks for the pattern). He should document the rationale in the rule, add unit tests for it, and submit it as a PR to the linter configuration repository. Once merged, Victor is no longer the enforcer of this constraint - C
102I is. He should count how many PRs over the next three months the lint rule blocks. That number is hours of review time recovered (his own and others'). Victor should make this visible to Bob and Sarah as evidence that lint-as-architecture is paying off.\"}],\"gettingStarted\":[\"Identify your most repeated review comments\",\"Start with one rule that enforces a clear boundary\",\"Choose the right tool for your language\"],\"links\":[{\"title\":\"ESLint: Custom Rules Documentation\",\"url\":\"https://eslint.org/docs/latest/extend/custom-rules\",\"domain\":\"eslint.org\"},{\"title\":\"ArchUnit: Unit Testing Java Architecture\",\"url\":\"https://www.archunit.org/userguide/html/000_Index.html\",\"domain\":\"archunit.org\"},{\"title\":\"golangci-lint: Custom Linters\",\"url\":\"https://golangci-lint.run/contributing/new-linters/\",\"domain\":\"golangci-lint.run\"},{\"title\":\"Pylint: Writing Custom Checkers\",\"url\":\"https://pylint.readthedocs.io/en/latest/how_tos/custom_checkers.html\",\"domain\":\"pylint.readthedocs.io\"},{\"title\":\"Netflix Tech Blog: Detecting Architecture Violations with Custom Linting\",\"url\":\"https://netflixtechblog.com/java-in-flames-e763b3d32166\",\"domain\":\"netflixtechblog.com\"}]},\"4\":{\"guideSlug\":\"green-yellow-red-auto-evaluation\",\"guideTitle\":\"Green/Yellow/Red auto-evaluation\",\"targetItems\":[{\"text\":\"Green/Yellow/Red auto-evaluation\",\"guideSlug\":\"green-yellow-red-auto-evaluation\"},{\"text\":\"Classification is fully algorithmic: the same change always lands in the same class\",\"guideSlug\":\"green-auto-merge-fully-algorithmic\"},{\"text\":\"Policy-based auto-approval driven by a risk classifier trained on your own incident history (Zalando: 33% of PRs auto-approved as low-risk, 20-40% lead-time reduction) - and every agent PR carries a named human owner, or it does not merge (agentic PRs merge at 79% elite vs 37% fair, and the gap is ownership)\",\"guideSlug\":\"policy-based-auto-approval-60-green-target\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$1f\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah wants to demonstrate to her engineering leadership that the team's quality investments are paying off. She has data on PR cycle time, post-merge bugs, and AI adoption. But she doesn't have a metric that shows the quality of the overall development process improving over time.\\n\\n**What Sarah should do:** The Green rate is exactly the metric Sarah needs. A rising Green rate over time means the team's code is increasingly meeting a consistent quality bar before review. It combines test coverage, lint compliance, and AI review quality into a single indicator. Sarah should propose implementing the traffic-light evaluation primarily for its measurement value, even before auto-merge is enabled. Tracking the Green rate weekly gives her a quality trend line she can report to leadership. When the rate rises consistently (as quality investments at L2-L3 compound), she has concrete evidence of ROI.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor is skeptical of the traffic-light system. He's worried that Green will be defined too loosely, that auto-merge will create incidents, and that the team will lose the \\\"second pair of eyes\\\" that review provides for catching unexpected issues. He's not wrong to be cautious.\\n\\n**What Victor should do:** Victor's caution is appropriate and should inform the Green criteria. He should be the one who defines what Green means - starting from \\\"what would make me comfortable not reviewing this PR at all?\\\" That's a high bar, and it should be. Victor can propose running the system in observation mode for 60 days before any auto-merge happens, with himself reviewing all PRs that would have been auto-merged to validate that they would have been safe. If that 60-day validation shows his veto rate is near zero, he has the evidence he needs to feel comfortable enabling auto-merge - and the team has the evidence it needs to trust Victor's judgment.\"}],\"gettingStarted\":[\"Define your initial Green criteria explicitly\",\"Define your Yellow criteria\",\"Define your Red criteria\"],\"links\":[{\"title\":\"GitHub: Required Status Checks\",\"url\":\"https://docs.github.com/en/repositories/configuring-branches-and-merges-in-your-repository/managing-protected-branches/about-protected-branches#require-status-checks-before-merging\",\"domain\":\"docs.github.com\"},{\"title\":\"Google: How Code Review Works at Google (Site Reliability Engineering)\",\"url\":\"https://sre.google/sre-book/foreword/\",\"domain\":\"sre.google\"},{\"title\":\"DORA: Key Metrics for DevOps Performance\",\"url\":\"https://dora.dev/guides/dora-metrics-four-keys/\",\"domain\":\"dora.dev\"},{\"title\":\"Trunk-Based Development: Keeping the Mainline Green\",\"url\":\"https://trunkbaseddevelopment.com/\",\"domain\":\"trunkbaseddevelopment.com\"},{\"title\":\"Atlassian: Automating Code Quality Checks\",\"url\":\"https://www.atlassian.com/continuous-delivery/principles/continuous-integration-vs-delivery-vs-deployment\",\"domain\":\"atlassian.com\"}]},\"5\":{\"guideSlug\":\"agent-fleet-self-reviews-cursor-model-error-fix-converge\",\"guideTitle\":\"Agent fleet self-reviews (error â fix â converge), bounded: a drafting agent never approves its own work, and LLM defect detection degrades across successive review rounds\",\"targetItems\":[{\"text\":\"Agent fleet self-reviews (error â fix â converge), bounded: a drafting agent never approves its own work, and LLM defect detection degrades across successive review rounds\",\"guideSlug\":\"agent-fleet-self-reviews-cursor-model-error-fix-converge\"},{\"text\":\"Human review only for Red (architectural)\",\"guideSlug\":\"human-review-only-for-red-architectural\"},{\"text\":\"Continuous auto-refactoring in background\",\"guideSlug\":\"continuous-auto-refactoring-in-background\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$20\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah's throughput metrics are strong: the fleet is producing 300+ PRs per week with 70% auto-merging. But she's starting to see noise in her post-merge defect data: a small but growing proportion of auto-merged PRs are being reverted within 24 hours. She wants to understand whether the agent fleet is outpacing the quality gate.\\n\\n**What Sarah should do:** Sarah should correlate the reverted PRs with their convergence history. Were reverted PRs ones that converged quickly (1-2 iterations) or ones that took many iterations? Were they produced by agents working in well-tested areas or areas with lower TORS? Her hypothesis should be: PRs converging in areas with TORS \u003c 90% are slipping through the quality gate. If confirmed, the fix is targeted - improve test coverage in the specific areas where reverts are occurring, not fleet-wide changes. Sarah should present this as: \\\"We've identified the specific quality gaps that are causing reverts. Here's the investment needed to close them.\\\"\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$21\"}
102],\"gettingStarted\":[\"Ensure your quality gate produces actionable error messages\",\"Configure agents to run tests locally before submitting PRs\",\"Instrument convergence metrics\"],\"links\":[{\"title\":\"Cursor: Background Agents and Self-Healing Code\",\"url\":\"https://www.cursor.com/blog/shadow-workspace\",\"domain\":\"cursor.com\"},{\"title\":\"Anthropic: Claude Code - Agentic Coding\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/overview\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Google DeepMind: AlphaCode 2 - Code Generation at Scale\",\"url\":\"https://deepmind.google/discover/blog/competitive-programming-with-alphacode/\",\"domain\":\"deepmind.google\"},{\"title\":\"SWE-bench: Evaluating Language Models on Real Software Engineering Tasks\",\"url\":\"https://www.swebench.com/\",\"domain\":\"swebench.com\"},{\"title\":\"Cognition AI: Devin - AI Software Engineer\",\"url\":\"https://cognition.ai/blog/introducing-devin\",\"domain\":\"cognition.ai\"},{\"title\":\"AI-to-AI code reviews of GitHub pull requests (arXiv 2608.21311)\",\"url\":\"https://arxiv.org/abs/2608.21311\",\"domain\":\"arxiv.org\"},{\"title\":\"MCR-Bench: multi-round LLM code review (arXiv 2608.27442)\",\"url\":\"https://arxiv.org/abs/2608.27442\",\"domain\":\"arxiv.org\"},{\"title\":\"Practical Loop Engineering - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/practical-loop-engineering\",\"domain\":\"addyo.substack.com\"},{\"title\":\"More than just code review - Simon Willison\",\"url\":\"https://simonwillison.net/2026/Aug/22/more-than-just-code-review/\",\"domain\":\"simonwillison.net\"}]}},\"Testing Strategy\":{\"1\":{\"guideSlug\":\"tests-written-manually-coverage-40\",\"guideTitle\":\"Tests written by hand\",\"targetItems\":[{\"text\":\"Tests written by hand\",\"guideSlug\":\"tests-written-manually-coverage-40\"},{\"text\":\"Flaky tests are a recurring cost\",\"guideSlug\":\"flaky-tests-16-of-dev-time-google-data\"},{\"text\":\"AI tests treat current output as \\\"correct\\\" - no independent oracle for the intended result\",\"guideSlug\":\"ai-tests-circularly-test-what-code-does-not-what-it-should-d\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has been shipping for two years without a systematic testing conversation. Coverage is estimated at around 35% - but it's never been formally measured. A recent production incident (a refactor broke an integration that had no tests) cost the team a full day of incident response and a difficult conversation with a customer.\\n\\n**What Bob should do:** Use the incident as a catalyst, not a blame exercise. The first action is measurement: run coverage reporting across all services and make the results visible to the whole team. Then establish a simple policy: coverage cannot decrease on any PR. This doesn't fix the existing debt, but it stops the accumulation. Bob should also start the conversation about AI-assisted test generation as a way to climb out of the debt hole without burning developer time - it's the core proposition for moving to L2.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is trying to make the case for expanding AI tooling to include test generation, but her stakeholders keep asking: \\\"If developers aren't writing tests now, why will AI tools fix that? Won't they just generate bad tests?\\\" She doesn't have a great answer yet.\\n\\n**What Sarah should do:** The stakeholder concern is legitimate and deserves a direct answer. The reason manual test writing stays below 40% isn't laziness - it's friction. Writing a test manually requires context-switching, boilerplate, and cognitive overhead that developers deprioritize under pressure. AI test generation removes the friction: you write the implementation, the agent generates the test scaffolding, and the developer's job becomes reviewing rather than authoring. Sarah should frame the investment not as \\\"AI writes our tests\\\" but as \\\"AI removes the friction that causes us to skip tests.\\\"\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has 85% coverage on his
102services because he practices TDD religiously. He's frustrated watching the rest of the team ship undertested code that he eventually gets paged about at 2am. He's suggested writing tests multiple times in code review and been told \\\"we'll add them later.\\\"\\n\\n**What Victor should do:** Victor's TDD practice is proof that high coverage is achievable - but preaching TDD to a team under deadline pressure is ineffective. His leverage is in tooling, not culture. Victor should set up AI-assisted test generation for the team's most common test patterns, lower the friction to near-zero, and demonstrate on one PR what it looks like when the AI generates the test scaffolding. Making test generation fast is more effective than making test writing mandatory.\"}],\"gettingStarted\":[\"Establish a coverage baseline\",\"Identify the highest-risk coverage gaps\",\"Add coverage reporting to CI\"],\"links\":[{\"title\":\"Google Testing Blog: Code Coverage Best Practices\",\"url\":\"https://testing.googleblog.com/2020/08/code-coverage-best-practices.html\",\"domain\":\"testing.googleblog.com\"},{\"title\":\"Martin Fowler: Test Coverage\",\"url\":\"https://martinfowler.com/bliki/TestCoverage.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"The Practical Test Pyramid - Ham Vocke\",\"url\":\"https://martinfowler.com/articles/practical-test-pyramid.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"Istanbul/nyc - JavaScript Coverage Tool\",\"url\":\"https://github.com/istanbuljs/nyc\",\"domain\":\"github.com\"},{\"title\":\"pytest-cov - Python Coverage Plugin\",\"url\":\"https://pytest-cov.readthedocs.io/en/latest/\",\"domain\":\"pytest-cov.readthedocs.io\"},{\"title\":\"JaCoCo - Java Code Coverage Library\",\"url\":\"https://www.jacoco.org/jacoco/\",\"domain\":\"jacoco.org\"}]},\"2\":{\"guideSlug\":\"agent-generated-unit-tests-human-acceptance-tests\",\"guideTitle\":\"Agent-generated unit tests + human acceptance tests\",\"targetItems\":[{\"text\":\"Agent-generated unit tests + human acceptance tests\",\"guideSlug\":\"agent-generated-unit-tests-human-acceptance-tests\"},{\"text\":\"Flaky test quarantine\",\"guideSlug\":\"flaky-test-quarantine\"},{\"text\":\"Humans define expected results for key paths (acceptance tests are the oracle)\",\"guideSlug\":\"test-oracle-stabilization\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob wants to move the team from L1 to L2 testing practices and has been pitched the hybrid model. He's worried about the implementation cost: configuring AI tools, training the team on the new process, and maintaining the boundary between test types over time. He's not sure it's worth the setup cost given the team's existing backlog.\\n\\n**What Bob should do:** Bob should pilot the hybrid model on one team for one sprint before rolling it out broadly. The key metric to track: total test coverage before and after, time spent on test-writing tasks, and any review comments about test quality. If one team's coverage climbs from 40% to 65% in a sprint without significant time investment, that's the business case. The ongoing cost of maintaining the boundary is lower than the ongoing cost of L1 testing debt and the inevitable production incidents it produces.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah needs to demonstrate that the hybrid model provides better value than the L1 baseline. Coverage numbers are one metric, but she wants to show that the tests are actually catching bugs - not just providing coverage.\\n\\n**What Sarah should do:** The metric Sarah needs is bug escape rate - the number of bugs that reach production vs. bugs caught in testing. As acceptance tests grow from requirements, they should catch more pre-production bugs. Track this alongside coverage: if coverage goes up and bug escape rate stays flat, the tests may be too superficial. If coverage goes up and bug escape rate drops, the hybrid model is working. Sarah should also track the ratio of acceptance tests to features shipped - ensuring that the human testing layer is growing proportionately with product scope.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has already adopted a version of the hybrid model informally: he writes unit tests quickly (sometimes using AI) and carefully authors scenario-based tests for complex business logic. He wants to formalize this into team-wide practice but isn't sure how to enforce the boundary without creating bureaucratic friction.\\n\\n**What Victor should do:** Victor should write the team's test guidelines document (not a heavy process doc - a two-page reference that explains what each test type is, when to use each, and what the review expectation is). Then he should make enforcement lightweight: a simple review checklist item \\\"Does this PR have at least one acceptance test for each user-facing behavior?\\\" is enough. The most important thing Victor can do is model the pattern visibly - when his PRs consistently follow the hybrid model, other engineers learn by example.\"}],\"gettingStarted\":[\"Define the boundary\",\"Set up AI-assisted unit test generation\",\"Create an acceptance test template\"],\"links\":[{\"title\":\"The Practical Test Pyramid - Ham Vocke\",\"url\":\"https://martinfowler.com/articles/practical-test-pyramid.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"Acceptance Test-Driven Development - Elisabeth Hen
102drickson\",\"url\":\"https://curiousduck.io/posts/collections/2024-06-27-atdd/\",\"domain\":\"curiousduck.io\"},{\"title\":\"Growing Object-Oriented Software, Guided by Tests - Freeman \u0026 Pryce\",\"url\":\"http://www.growing-object-oriented-software.com/\",\"domain\":\"growing-object-oriented-software.com\"},{\"title\":\"Google Testing Blog: Test Sizes\",\"url\":\"https://testing.googleblog.com/2010/12/test-sizes.html\",\"domain\":\"testing.googleblog.com\"},{\"title\":\"GitHub Copilot for Test Generation\",\"url\":\"https://docs.github.com/en/copilot/tutorials/write-tests\",\"domain\":\"docs.github.com\"},{\"title\":\"Martin Fowler: Unit Test\",\"url\":\"https://martinfowler.com/bliki/UnitTest.html\",\"domain\":\"martinfowler.com\"}]},\"3\":{\"guideSlug\":\"tors-90-test-oracle-reliability-score\",\"guideTitle\":\"Expected results come from requirements (tickets/specs are the oracle, not the code); test *process* is not mandated to agents - outcomes are measured instead\",\"targetItems\":[{\"text\":\"Expected results come from requirements (tickets/specs are the oracle, not the code); test *process* is not mandated to agents - outcomes are measured instead\",\"guideSlug\":\"tors-90-test-oracle-reliability-score\"},{\"text\":\"Acceptance tests from tickets (Autonomous Requirements)\",\"guideSlug\":\"acceptance-tests-from-tickets-autonomous-requirements\"},{\"text\":\"Incremental test selection (only changed paths)\",\"guideSlug\":\"incremental-test-selection-only-changed-paths\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob has been tracking build reliability (percentage of CI runs completing without retry) as his proxy metric. It's at 89%. He wants to move to L3 automated quality gates but isn't sure if 89% build reliability translates to a reliable enough foundation.\\n\\n**What Bob should do:** Build reliability and TORS are related but not the same. Build reliability measures run-level reliability (did the whole CI run succeed?), while TORS measures test-level reliability (do individual test failures mean something?). Bob needs to start tracking TORS explicitly to understand the quality of the test signal, not just the build surface. If TORS is below 90% even when build reliability is 89%, the team isn't ready for automated quality gates. The specific metric Bob should add to his dashboard is TORS by service - it will reveal which services are ready for L4 automation and which need more stabilization work.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah wants to put a dollar value on reaching 90% TORS as a business case for the engineering investment. She has the Google 16% flaky test data but needs to translate it into a number for their team.\\n\\n**What Sarah should do:** TORS provides a direct productivity calculation. At 75% TORS, 25% of test failure investigations are wasted - the developer investigates, finds nothing wrong, re-runs, and moves on. For a team spending an average of 30 minutes per test failure investigation, 25% waste rate means roughly one full day per developer per month lost to false positives, at scale. Calculate this for the team size and present it as the cost of staying below 90% TORS. The engineering investment in oracle stabilization and flakiness elimination pays back quickly at that rate.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been informally tracking which tests are worth investigating - he mentally marks tests as \\\"probably flaky\\\" and ignores them. His personal TORS intuition is good, but it's not shared with the team and it's not encoded anywhere.\\n\\n**What Victor should do:** Victor's mental model needs to become infrastru
102cture. He should work with the DevOps team to instrument TORS measurement into the CI pipeline and publish the score on the engineering dashboard. His next step is to formalize his intuition: when does he dismiss a failure vs. investigate? The answer is the TORS measurement protocol. Once it's documented and automated, the rest of the team benefits from Victor's judgment without needing to consult him individually on every CI failure.\"}],\"gettingStarted\":[\"Instrument CI for TORS measurement\",\"Establish the baseline TORS\",\"Set the 90% target and timeline\"],\"links\":[{\"title\":\"Gradle Enterprise Test Analytics: Flakiness Detection\",\"url\":\"https://gradle.com/gradle-enterprise-solutions/test-distribution/\",\"domain\":\"gradle.com\"},{\"title\":\"Google Testing Blog: Flaky Tests at Google and How We Mitigate Them\",\"url\":\"https://testing.googleblog.com/2016/05/flaky-tests-at-google-and-how-we.html\",\"domain\":\"testing.googleblog.com\"},{\"title\":\"BuildKite Test Analytics\",\"url\":\"https://buildkite.com/test-analytics\",\"domain\":\"buildkite.com\"},{\"title\":\"DataDog CI Visibility - Test Flakiness Tracking\",\"url\":\"https://docs.datadoghq.com/continuous_integration/guides/flaky_test_management/\",\"domain\":\"docs.datadoghq.com\"},{\"title\":\"Martin Fowler: Continuous Integration\",\"url\":\"https://martinfowler.com/articles/continuousIntegration.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"TDD inside the agent loop - Birgitta Böckeler, martinfowler.com\",\"url\":\"https://martinfowler.com/articles/exploring-gen-ai/tdd-in-the-agent-loop.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"DECODE: what developers do with accepted AI completions (arXiv 2607.25130)\",\"url\":\"https://arxiv.org/abs/2607.25130\",\"domain\":\"arxiv.org\"},{\"title\":\"SecTDD: security tests supplied up front (arXiv 2608.09740)\",\"url\":\"https://arxiv.org/abs/2608.09740\",\"domain\":\"arxiv.org\"}]},\"4\":{\"guideSlug\":\"tors-95\",\"guideTitle\":\"Held-out oracles the agent never sees gate releases (\\\"Building to the Test\\\": with oracle access agents ship dead code passing all 222 tests); property-based testing + fuzzing over LLM-written tests\",\"targetItems\":[{\"text\":\"Held-out oracles the agent never sees gate releases (\\\"Building to the Test\\\": with oracle access agents ship dead code passing all 222 tests); property-based testing + fuzzing over LLM-written tests\",\"guideSlug\":\"tors-95\"},{\"text\":\"Agent iterates tests to green in sandbox (doesn't block team CI)\",\"guideSlug\":\"agent-iterates-tests-to-green-in-sandbox-doesn-t-block-team-\"},{\"text\":\"Mutation testing on high-risk paths as the real coverage signal (one component: 100% line coverage, 61% mutation strength)\",\"guideSlug\":\"mutation-testing-agent-validation\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has been at 91% TORS for two months. They tried to enable automated merge but developers complained that it was blocking good code too often. Bob disabled automated merge to stop the friction, but now he's not sure how to get the team to 95% without it being an open-ended quality initiative.\\n\\n**What Bob should do:** The problem Bob experienced is exactly what the 95% threshold exists to prevent. At 91%, automated merge is not reliable enough and generates friction. Bob should reframe the work: before re-enabling automated merge, bring TORS to 95% as a prerequisite, not a nice-to-have. The concrete initiative: run a TORS audit to find the concentrated sources of the remaining false positives, scope the fix as a time-boxed sprint (not an open-ended project), and measure TORS daily during the sprint. When 95% is stable for two weeks, re-enable automated merge for one service as a controlled pilot.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah's team is preparing to present the L4 automation investment to the board. She needs to quantify the value of reaching 95% TORS specifically, not just \\\"better test quality.\\\"\\n\\n**What Sarah should do:** The value of 95% TORS is the value of automated merge decisions. For a team of 50 engineers merging 100 PRs per week, automated merge with 95% TORS eliminates the human review bottleneck for the majority of PRs. If each PR currently requires 2 hours of review time (reviewer availability, context loading, review itself), and automated merge handles 60% of PRs, the time savings are 60 PRs/week x 2 hours = 120 hours/week freed from routine review. That's 3 full-time engineers worth of time redirected to higher-value work. Sarah should model this for her team's actual numbers.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been the person who unblocks developers when automated merge incorrectly rejects their PRs. He's the de facto human override for the system, which is not sustainable. He needs the system to be reliable enough that overrides are rare exceptions.\\n\\n**What Victor should do:** Victor is in the best position to fix the remaining TORS gap because he's seen every false positive firsthand. He should maintain a log of every override he's performed in the last month and categorize the root causes. Almost certainly, three or four specific test categories account for most of the overrides. Victor should prioritize fixing those categories rather than the aggregate TORS number. When the four worst offenders are fixed, TORS will likely jump from 91% to 95%+ and the override rate will drop to near zero.\"}],\"gettingStarted\":[\"Audit the remaining false positive sources\",\"Invest in test environment determinism\",\"Implement test retry intelligence\"],\"links\":[{\"title\":\"Google Engineering: Continuous Integration Done Right\",\"url\":\"https://testing.googleblog.com/2011/10/google-test-analytics-now-in-open.html\",\"domain\":\"testing.googleblog.com\"},{\"title\":\"Gradle Enterprise: Test Distribution and Predictive Test Selection\",\"url\":\"https://gradle.com/gradle-enterprise-solutions/test-distribution/\",\"domain\":\"gradle.com\"},{\"title\":\"Testcontainers - Hermetic Test Environments\",\"url\":\"https://testcontainers.com/\",\"domain\":\"testcontainers.com\"},{\"title\":\"BuildKite Test Analytics: Reliability Tracking\",\"url\":\"https://buildkite.com/test-analytics\",\"domain\":\"buildkite.com\"},{\"title\":\"Martin Fowler: Feature Toggle for Controlled Rollout\",\"url\":\"https://martinfowler.com/articles/feature-toggles.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"Building to the Test: Agents Optimize Visible Oracles (arXiv 2606.28430)\",\"url\":\"https://arxiv.org/abs/2606.28430\",\"domain\":\"arxiv.org\"}
102,{\"title\":\"Dan Luu: AI Coding with Proper Testing Infrastructure\",\"url\":\"https://danluu.com/ai-coding/\",\"domain\":\"danluu.com\"}]},\"5\":{\"guideSlug\":\"self-healing-test-suite\",\"guideTitle\":\"Self-healing test suite\",\"targetItems\":[{\"text\":\"Self-healing test suite\",\"guideSlug\":\"self-healing-test-suite\"},{\"text\":\"Production logs â auto-generated regression tests\",\"guideSlug\":\"production-logs-auto-generated-regression-tests\"},{\"text\":\"Agent detects edge case â writes test â fixes bug â ships\",\"guideSlug\":\"agent-detects-edge-case-writes-test-fixes-bug-ships\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has been managing the test suite manually for three years. Coverage has drifted down, flakiness has accumulated, and two engineers spend a significant portion of their time on test maintenance rather than feature development. He knows something needs to change but the shift to autonomous maintenance feels like a big jump.\\n\\n**What Bob should do:** Bob doesn't need to jump to full self-healing immediately. The progression is: L3 quarantine + oracle stabilization (done), L4 agent sandbox iteration (in progress), and L5 self-healing as the next step. The practical entry point for self-healing is the simplest healing mode: flaky test remediation. Start by automating the quarantine-to-fix workflow for the most common oracle failure patterns (timing, ordering). Once that works reliably, add refactor-induced repair. The full self-healing suite is a series of incremental automations, not a single big-bang deployment.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah wants to quantify the ROI of self-healing test suites. She knows two engineers spend significant time on test maintenance, but she's not sure how to project the value of automating that work.\\n\\n**What Sarah should do:** The calculation is straightforward: measure current test maintenance time per engineer per week across the team. If two engineers spend 30% of their time on test maintenance, that's 12 engineer-hours per week - roughly $120k/year in a mid-market engineering org. Self-healing doesn't eliminate all maintenance (humans still handle escalations), but it should reduce it by 70-80%. The reduction in maintenance cost, combined with the quality improvement from continuous monitoring, is the ROI case. Sarah should also model the opportunity cost: what those two engineers would produce if 70% of their maintenance time were redirected to feature work.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor is technically the person best positioned to build the self-healing infrastructure. He's excited about the concept but overwhelmed by the scope. He doesn't know where to start without building an enormous system.\\n\\n**What Victor should do:** Victor should start with the smallest possible version of one healing behavior: automatic detection and fix for timing-sensitive flaky tests. These are the most common, the most well-understood, and the easiest to fix autonomously. Build the pipeline: detect (CI instrumentation flags timing-based failures), diagnose (agent reads the stack trace and identifies the non-deterministic time call), fix (agent injects a clock interface and replaces the time call with a controlled value), validate (sandbox CI confirms the test no longer flakes), and submit (automatic PR for brief human review). Once this works end-to-end for one failure category, the pattern generalizes to other categories.\"}],\"gettingStarted\":[\"Establish the prerequisite maturity\",\"Define the healing policy\",\"Build the flaky test detection pipeline\"],\"links\":[{\"title\":\"Google Testing Blog: Test Flakiness - One of the Main Challenges of Automated Testing\",\"url\":\"https://testing.googleblog.com/2020/12/test-flakiness-one-of-main-challenges.html\",\"domain\":\"testing.googleblog.com\"},{\"title\":\"Stryker Mutator: Mutation Testing for Continuous Quality Improvement\",\"url\":\"https://stryker-mutator.io/\",\"domain\":\"stryker-mutator.io\"},{\"title\":\"Martin Fowler: Refactoring - Improving the Design of Existing Code\",\"url\":\"https://martinfowler.com/books/refactoring.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"Testcontainers: Deterministic Test Environments\",\"url\":\"https://testcontainers.com/\",\"domain\":\"testcontainers.com\"},{\"title\":\"Anthropic Claude API: Building Agentic Systems\",\"url\":\"https://docs.anthropic.com/en/docs/build-with-claude/agents\",\"domain\":\"docs.anthropic.com\"}]}}},\"delivery\":{\"CI/CD Pipeline\":{\"1\":{\"guideSlug\":\"ci-15-minutes\",\"guideTitle\":\"CI runs on every change\",\"targetItems\":[{\"text\":\"CI runs on every change\",\"guideSlug\":\"ci-15-minutes\"},{\"text\":\"Agent waits for CI feedback\",\"guideSlug\":\"agent-is-blind-waits-for-feedback\"},{\"text\":\"Shared runner, queue\",\"guideSlug\":\"shared-runner-queue\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$22\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$23\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor runs 3-4 parallel agents and feels the pain of 18-minute CI acutely. He's already hacked around it: he has a local script that runs just the affected tests before pushing, and he only pushes to CI when he's reasonably confident. But he knows this is a local workaround that doesn't scale to the team. The real fix is in the pipeline, not in individual developer habits.\\n\\nVictor should volunteer to own the CI speed project and approach it as a systems problem. His first move: instrument every stage of the pipeline with explicit timing and publish the data to the team's dashboard. His second move: implement the fast-path / slow-path split so agents get feedback in under 5 minutes even before the full suite is optimized. His third move: document the impact - before and after CI times, and the resulting agent iteration rate improvement. Victor should position this work not as \\\"CI cleanup\\\" but as \\\"agent infrastru
102cture\\\" - the framing that gets engineering leadership to prioritize it.\"}],\"gettingStarted\":[\"Measure before optimizing\",\"Identify the top three slow stages\",\"Separate fast feedback from slow validation\"],\"links\":[{\"title\":\"GitHub Actions - Caching dependencies\",\"url\":\"https://docs.github.com/en/actions/writing-workflows/choosing-what-your-workflow-does/caching-dependencies-to-speed-up-workflows\",\"domain\":\"docs.github.com\"},{\"title\":\"CircleCI - Optimizing your build\",\"url\":\"https://circleci.com/docs/guides/optimize/optimizations/\",\"domain\":\"circleci.com\"},{\"title\":\"BuildKite - Pipeline optimization\",\"url\":\"https://buildkite.com/docs/pipelines/optimize-pipelines\",\"domain\":\"buildkite.com\"},{\"title\":\"Google Engineering Practices - CI best practices\",\"url\":\"https://abseil.io/resources/swe-book/html/ch23.html\",\"domain\":\"abseil.io\"},{\"title\":\"DORA Research - CI/CD metrics\",\"url\":\"https://dora.dev/research/\",\"domain\":\"dora.dev\"}]},\"2\":{\"guideSlug\":\"basic-caching\",\"guideTitle\":\"Pipeline definitions live in the repo and are reviewed like application code\",\"targetItems\":[{\"text\":\"Pipeline definitions live in the repo and are reviewed like application code\",\"guideSlug\":\"basic-caching\"},{\"text\":\"Dedicated runners per team\",\"guideSlug\":\"dedicated-runners-per-team\"},{\"text\":\"CI \u003c 10 minutes\",\"guideSlug\":\"ci-10-minutes\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$24\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$25\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$26\"}],\"gettingStarted\":[\"Inventory what is currently defined outside the repo\",\"Export one pipeline to a file and make it authoritative\",\"Move secrets out of the definition but keep their declaration in it\"],\"links\":[{\"title\":\"GitHub Actions - Workflow syntax\",\"url\":\"https://docs.github.com/en/actions/writing-workflows/workflow-syntax-for-github-actions\",\"domain\":\"docs.github.com\"},{\"title\":\"GitHub - Reusable workflows\",\"url\":\"https://docs.github.com/en/actions/sharing-automations/reusing-workflows\",\"domain\":\"docs.github.com\"},{\"title\":\"GitLab CI - .gitlab-ci.yml reference\",\"url\":\"https://docs.gitlab.com/ee/ci/yaml/\",\"domain\":\"docs.gitlab.com\"},{\"title\":\"actionlint - Static checker for GitHub Actions workflows\",\"url\":\"https://github.com/rhysd/actionlint\",\"domain\":\"github.com\"},{\"title\":\"GitHub - About code owners\",\"url\":\"https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/about-code-owners\",\"domain\":\"docs.github.com\"}]},\"3\":{\"guideSlug\":\"incremental-builds-only-changed-fragments\",\"guideTitle\":\"Agent CI treated as internet-facing: no `${{ github.event.* }}` interpolated into `run:`, agent passes split into separate jobs with per-job token scope\",\"targetItems\":[{\"text\":\"Agent CI treated as internet-facing: no `${{ github.event.* }}` interpolated into `run:`, agent passes split into separate jobs with per-job token scope\",\"guideSlug\":\"incremental-builds-only-changed-fragments\"},{\"text\":\"Per-worktree pipelines, so parallel agents do not serialise on one runner\",\"guideSlug\":\"bazel-remote-caching-engflow\"},{\"text\":\"CI \u003c 5 minutes\",\"guideSlug\":\"ci-5-minutes\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team runs a Java monorepo with 15 services. CI is at 9 minutes, and the breakdown shows that 6 minutes is Java compilation. The team has already enabled Gradle build cache locally but not in CI. When Bob hears this, he realizes that the incremental build state being computed by developers locally is not being shared with CI - CI is doing full rebuilds while developers' local builds are already fast.\\n\\nBob should ask the CI owner to enable Gradle remote build cache with a simple self-hosted cache node (Gradle provides a Docker image for this). The CI configuration change is minimal - add `--build-cache` to the Gradle command and point to the remote cache URL. The expected result: CI compilation drops from 6 minutes to under 1 minute for typical branch builds that share compilation state with recent developer or CI runs. Bob should set a t
102wo-week measure period after the change and report the results to the team as a concrete example of infrastructure investment paying off quickly.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah's CI timing analysis shows that 68% of CI runs for the JavaScript monorepo are building all 12 packages even when only 1-2 packages changed. She knows Nx has an `affected` command that would run builds and tests only for the packages affected by the current commit, but it hasn't been connected to CI.\\n\\nSarah should quantify the waste: if 68% of runs build 12 packages when they should build 2-3 packages, the team is doing 4-6x more work than necessary on most CI runs. At 9-minute CI, correctly scoped CI would take approximately 2-3 minutes for the typical change. Sarah should present this calculation to Bob alongside the Nx `affected` configuration - a two-day engineering task to implement - and frame it as \\\"we are currently paying for and waiting for 6x more work than we need on every typical commit.\\\" The data makes the priority self-evident.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$27\"}],\"gettingStarted\":[\"Identify your build system's incremental build support\",\"Preserve the incremental build cache across CI runs\",\"For monorepos: implement affected package detection\"],\"links\":[{\"title\":\"Turborepo - Incremental builds and remote caching\",\"url\":\"https://turbo.build/repo/docs/core-concepts/remote-caching\",\"domain\":\"turbo.build\"},{\"title\":\"Nx - Affected commands\",\"url\":\"https://nx.dev/ci/features/affected\",\"domain\":\"nx.dev\"},{\"title\":\"Gradle - Build cache documentation\",\"url\":\"https://docs.gradle.org/current/userguide/build_cache.html\",\"domain\":\"docs.gradle.org\"},{\"title\":\"TypeScript - Project references and incremental compilation\",\"url\":\"https://www.typescriptlang.org/docs/handbook/project-references.html\",\"domain\":\"typescriptlang.org\"},{\"title\":\"Pants - Python and JVM build system\",\"url\":\"https://www.pantsbuild.org/\",\"domain\":\"pantsbuild.org\"},{\"title\":\"The Hacker News - Claude Code and Gemini CLI flaws let a public issue reach CI secrets\",\"url\":\"https://thehackernews.com/2026/08/claude-code-and-gemini-cli-flaws-let.html\",\"domain\":\"thehackernews.com\"}]},\"4\":{\"guideSlug\":\"ci-as-sandbox-50-attempts-in-5-min-without-blocking-team\",\"guideTitle\":\"CI as Sandbox: 50 attempts in 5 min without blocking team; merge queues for parallel agent fleets (auto-merge only what builds and passes tests); scheduled / async agents land PRs overnight\",\"targetItems\":[{\"text\":\"CI as Sandbox: 50 attempts in 5 min without blocking team; merge queues for parallel agent fleets (auto-merge only what builds and passes tests); scheduled / async agents land PRs overnight\",\"guideSlug\":\"ci-as-sandbox-50-attempts-in-5-min-without-blocking-team\"},{\"text\":\"Every CI run for an agent gets its own disposable environment, so a poisoned run cannot reach the next one\",\"guideSlug\":\"ephemeral-sandboxes-agent-has-own-environment-10s-spin-up\"},{\"text\":\"CI \u003c 2 minutes\",\"guideSlug\":\"ci-2-minutes\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$28\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$29\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$2a\"}],\"gettingStarted\":[\"Provision a dedicated agent sandbox runner pool\",\"Create a lightweight \\\"agent sandbox\\\" CI pipeline\",\"Implement a concurrency limit for sandbox pipelines\"],\"links\":[{\"title\":\"GitHub Actions - Repository dispatch for triggering workflows\",\"url\":\"https://docs.github.com/en/actions/writing-workflows/choosing-when-your-workflow-runs/events-that-trigger-workflows#repository_dispatch\",\"domain\":\"docs.github.com\"},{\"title\":\"GitHub CLI - Managing workflow runs\",\"url\":\"https://cli.github.com/manual/gh_run\",\"domain\":\"cli.github.com\"},{\"title\":\"GitHub Actions - Concurrency control\",\"url\":\"https://docs.github.com/en/actions/writing-workflows/choosing-what-your-workflow-does/using-concurrency\",\"domain\":\"docs.github.com\"},{\"title\":\"Stripe Engineering - Minions agent system\",\"url\":\"https://stripe.dev/blog/minions-stripes-one-shot-end-to-end-coding-agents\",\"domain\":\"stripe.dev\"},{\"title\":\"Anthropic - Claude Code MCP integration\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp\",\"domain\":\"docs.anthropic.com\"}]},\"5\":{\"guideSlug\":\"sub-minute-feedback\",\"guideTitle\":\"Sub-minute feedback\",\"targetItems\":[{\"text\":\"Sub-minute feedback\",\"guideSlug\":\"sub-minute-feedback\"},{\"text\":\"Runner capacity auto-scales with agent load, with no manual capacity planning\",\"guideSlug\":\"self-driving-ci-auto-scaling-per-agent-load\"},{\"text\":\"Production feedback â CI auto-adjusts test suite; every production failure becomes a permanent regression test; verification moves before the PR opens and CI checks the evidence instead of re-running it (main-branch success fell to a five-year low of 70.8% across 28M
102workflows)\",\"guideSlug\":\"production-feedback-ci-auto-adjusts-test-suite\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$2b\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$2c\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$2d\"}],\"gettingStarted\":[\"Build a test impact analysis system with sub-20 test selection\",\"Deploy pre-warmed runner containers with zero startup latency\",\"Implement remote build execution (RBE)\"],\"links\":[{\"title\":\"EngFlow - Remote Build Execution\",\"url\":\"https://www.engflow.com/\",\"domain\":\"engflow.com\"},{\"title\":\"BuildBuddy - Bazel remote caching and RBE\",\"url\":\"https://www.buildbuddy.io/\",\"domain\":\"buildbuddy.io\"},{\"title\":\"Actions Runner Controller - Kubernetes-based GitHub Actions runners\",\"url\":\"https://github.com/actions/actions-runner-controller\",\"domain\":\"github.com\"},{\"title\":\"Launchable - Function-level test impact analysis\",\"url\":\"https://www.launchableinc.com/\",\"domain\":\"launchableinc.com\"},{\"title\":\"Google Cloud - Remote Build Execution\",\"url\":\"https://cloud.google.com/build/docs/optimize-builds/speeding-up-builds\",\"domain\":\"cloud.google.com\"}]}},\"Merge \u0026 Deploy\":{\"1\":{\"guideSlug\":\"manual-pr-review-manual-merge\",\"guideTitle\":\"Merging is a manual act: someone clicks the button on every change\",\"targetItems\":[{\"text\":\"Merging is a manual act: someone clicks the button on every change\",\"guideSlug\":\"manual-pr-review-manual-merge\"},{\"text\":\"Merge capacity set by how fast humans can review\",\"guideSlug\":\"10-pr-day-capacity\"},{\"text\":\"Manual deploy or simple CD\",\"guideSlug\":\"manual-deploy-or-simple-cd\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has been using GitHub Copilot for three months and PR volume has quietly doubled. His team lead mentioned that the review queue is \\\"getting a bit long\\\" but hasn't escalated it as a formal problem. Bob's instinct is to add another senior reviewer to the rotation, which would address the symptom but not the cause.\\n\\n**What Bob should do:** Bob needs to look at the data before adding headcount to the review queue. He should pull 30-day PR cycle time broken down by stage. If the bottleneck is \\\"time from first review to merge,\\\" the problem is throughput, not quality, and the solution is automation, not more reviewers. Bob should use this analysis to frame the L2 investment: a merge queue and basic branch protection rules will do more for delivery velocity than another reviewer in the rotation. The data makes the case for automation credibly.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking developer satisfaction and sees friction appearing in standup notes: \\\"waiting on review,\\\" \\\"PR has been sitting for two days,\\\" \\\"merged stale code by mistake.\\\" These are symptoms of a manual process under load, but they're scattered across many retros and hard to aggregate into a single narrative.\\n\\n**What Sarah should do:** Sarah should set up a simple PR cycle time dashboard - GitHub's built-in Insights view or a free tier of LinearB will show average time-to-merge. Then she should present the trend line to the team: \\\"Our average PR takes X hours to merge and that number is going up.\\\" This creates shared visibility and a concrete improvement target. The next step is to identify the single biggest time sink (almost certainly \\\"time waiting for first review\\\") and design a lightweight intervention: a Slack notification when PRs are older than 4 hours with no review, or a daily digest of open PRs. These don't require tooling changes and can be operational within a week.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$2e\"}],\"gettingStarted\":[\"Map your current process explicitly\",\"Measure baseline PR cycle time\",\"Count PRs per day\"],\"links\":[{\"title\":\"GitHub Branch Protection Rules\",\"url\":\"https://docs.github.com/en/repositories/configuring-branches-and-merges-in-your-repository/managing-protected-branches/about-protected-branches\",\"domain\":\"docs.github.com\"},{\"title\":\"LinearB - Engineering Metrics\",\"url\":\"https://linearb.io/\",\"domain\":\"linearb.io\"},{\"title\":\"The SPACE Framework for Developer Productivity\",\"url\":\"https://queue.acm.org/detail.cfm?id=3454124\",\"domain\":\"queue.acm.org\"},{\"title\":\"Accelerate: The Science of Lean Software and DevOps\",\"url\":\"https://itrevolution.com/accelerate-book/\",\"domain\":\"itrevolution.com\"},{\"title\":\"GitHub Pull Request Review Best Practices\",\"url\":\"https://docs.github.com/en/pull-requests/collaborating-with-pull-requests/reviewing-changes-in-pull-requests/about-pull-request-reviews\",\"domain\":\"docs.github.com\"}]},\"2\":{\"guideSlug\":\"cd-pipeline-with-gates\",\"guideTitle\":\"CD pipeline with gates; agent commits come from a distinct bot identity, never a developer's account\",\"targetItems\":[{\"text\":\"CD pipeline with gates; agent commits come from a distinct bot identity, never a developer's account\",\"guideSlug\":\"cd-pipeline-with-gates\"},{\"text\":\"Basic merge queues\",\"guideSlug\":\"basic-merge-queues\"},{\"text\":\"Auto-rebase\",\"guideSlug\":\"auto-rebase\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$2f\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$30\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$31\"}],\"gettingStarted\":[\"Map your current deploy stages\",\"Implement a staging smoke test suite\",\"Add a post-deploy health check\"],\"links\":[{\"title\":\"GitHub Actions - Multi-Environment Deployments\",\"url\":\"https://docs.github.com/en/actions/deployment/targeting-different-environments/using-environments-for-deployment\",\"domain\":\"docs.github.com\"},{\"title\":\"ArgoCD - Progressive Delivery\",\"url\":\"https://argo-cd.readthedocs.io/en/stable/\",\"domain\":\"argo-cd.readthedocs.io\"},{\"title\":\"Google SRE - Canarying Releases\",\"url\":\"https://sre.google/workbook/canarying-releases/\",\"domain\":\"sre.google\"},{\"title\":\"Spinnaker - Automated Canary Analysis\",\"url\":\"https://spinnaker.io
102/docs/guides/user/canary/\",\"domain\":\"spinnaker.io\"},{\"title\":\"Flagger - Progressive Delivery Operator\",\"url\":\"https://flagger.app/\",\"domain\":\"flagger.app\"},{\"title\":\"VentureBeat - AI agents breached 395 organizations using credentials your IAM policy still treats as human\",\"url\":\"https://venturebeat.com/security/ai-agents-breached-395-organizations-using-credentials-your-iam-policy-still-treats-as-human\",\"domain\":\"venturebeat.com\"},{\"title\":\"Uber - Running a software factory efficiently at Uber scale\",\"url\":\"https://www.uber.com/blog/software-factory-uber-scale/\",\"domain\":\"uber.com\"}]},\"3\":{\"guideSlug\":\"policy-based-merge-rules\",\"guideTitle\":\"Policy-based merge rules; agents push with short-lived GitHub App or OIDC tokens and never hold deploy secrets; agent output lands as a stack of dependent, independently reviewable branches rather than one 1,000-line PR\",\"targetItems\":[{\"text\":\"Policy-based merge rules; agents push with short-lived GitHub App or OIDC tokens and never hold deploy secrets; agent output lands as a stack of dependent, independently reviewable branches rather than one 1,000-line PR\",\"guideSlug\":\"policy-based-merge-rules\"},{\"text\":\"Deterministic ordering + conflict detection (cross-vendor agent PR pairs conflict at 41.7% vs 19.8% intra-vendor - standardize the fleet or serialize the merges)\",\"guideSlug\":\"deterministic-ordering-conflict-detection\"},{\"text\":\"A published cap on CI rounds per PR, and the team tracks it\",\"guideSlug\":\"max-2-ci-rounds-per-pr-stripe-benchmark\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob has different standards for different change types but no way to enforce them consistently. Security-adjacent changes are supposed to require a security review, but sometimes they slip through without one because the reviewer didn't notice the change touched authentication code. Bob has no visibility into whether his team's merge policies are actually being followed.\\n\\n**What Bob should do:** Bob should start by auditing the last 90 days of merges for security-adjacent changes (any PR that touches `src/auth/`, `src/payments/`, or `config/security.yml`). How many were reviewed by someone with security expertise? How many were not? This audit converts \\\"we probably have a policy gap\\\" into \\\"we have a measurable gap in N% of security changes.\\\" From there, Bob should commission a CODEOWNERS implementation for high-risk paths and a Mergify rule that blocks merge on those paths without the required reviewer. A one-sprint investment in policy infrastructure delivers ongoing audit compliance.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah wants to understand the variance in PR cycle time across the team. She sees that some PRs merge in 2 hours and others take 3 days, and can't explain the difference. Her hypothesis is that the variance comes from inconsistent application of review criteria.\\n\\n**What Sarah should do:** Sarah should tag the last 90 days of PRs by change type (documentation, feature, infrastructure, security) and compare cycle time by category. If security PRs take 3x longer than feature PRs, the question is: is that appropriate (security review takes longer) or wasteful (security reviewers don't know they're needed until someone asks)? Policy-based rules solve the second problem: CODEOWNERS automatically routes security PRs to the right reviewers at PR open time, eliminating the 24-hour lag between \\\"PR opened\\\" and \\\"right reviewer notified.\\\" Sarah should quantify the notification lag and present it as the latency that policy automation eliminates.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$32\"}],\"gettingStarted\":[\"Document your current implicit merge criteria\",\"Implement CODEOWNERS\",\"Configure branch protection rules\"],\"links\":[{\"title\":\"Mergify - Documentation\",\"url\":\"https://docs.mergify.com/\",\"domain\":\"docs.mergify.com\"},{\"title\":\"GitHub CODEOWNERS - Documentation\",\"url\":\"https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/about-code-owners\",\"domain\":\"docs.github.com\"},{\"title\":\"GitHub Branch Protection Rules\",\"url\":\"https://docs.github.com/en/repositories/configuring-branches-and-merges-in-your-repository/managing-protected-branches/about-protected-branches\",\"domain\":\"docs.github.com\"},{\"title\":\"Prow - Kubernetes CI/CD System\",\"url\":\"https://docs.prow.k8s.io/docs/overview/\",\"domain\":\"docs.prow.k8s.io\"},{\"title\":\"Trunk.io - Merge Policy\",\"url\":\"https://docs.trunk.io/merge\",\"domain\":\"docs.trunk.io\"},{\"title\":\"GitHub - Turn one giant AI-generated pull request into a reviewable stack\",\"url\":\"https://github.blog/engineering/turn-one-giant-ai-generated-pull-request-to-a-reviewable-stack/\",\"domain\":\"github.blog\"},{\"title\":\"Zalando - Agentic engineering at Zalando: a snapshot\",\"url\":\"https://engineering.zalando.com/posts/2026/08/agentic-engineering-at-zalando-a-snapshot.html\",\"domain\":\"engineering.zalando.com\"},{\"title\":\"LinearB - Software factory 2026 AI benchmarks\",\"url\":\"https://linearb.io/blog/software-factory-2026-ai-benchmarks-code-review-roi\",\"domain\":\"linearb.io\"},{\"title\":\"Anthropic - Threat intelligence report, September 2026\",\"url\":\"https://www.anthropic.com/threat-intelligence-report-september-2026\",\"domain\":\"anthropic.com\"},{\"title\":\"CrowdStrike - Agentic Identity Provider\",\"url\":\"https://www.crowdstrike.com/en-us
102/blog/crowdstrike-announces-agentic-identity-provider/\",\"domain\":\"crowdstrike.com\"}]},\"4\":{\"guideSlug\":\"green-auto-merge-auto-deploy\",\"guideTitle\":\"A green verdict flows straight to production without a second queue or a second approval - while high-impact actions (token creation, production credentials, webhook edits) require a fresh human re-authentication (GitHub proof of presence)\",\"targetItems\":[{\"text\":\"A green verdict flows straight to production without a second queue or a second approval - while high-impact actions (token creation, production credentials, webhook edits) require a fresh human re-authentication (GitHub proof of presence)\",\"guideSlug\":\"green-auto-merge-auto-deploy\"},{\"text\":\"Throughput well above the pre-agent baseline, with the merge path no longer the constraint\",\"guideSlug\":\"50-pr-day-throughput\"},{\"text\":\"Canary/progressive deployment auto; verifiable provenance per change, with AI-assistance disclosure enforced in CI rather than by convention (Linux 7.2 carried 1,111 `Assisted-by` commits, and at least one maintainer strips the tags)\",\"guideSlug\":\"canary-progressive-deployment-auto\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$33\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$34\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has already configured auto-merge for low-risk PRs in his repositories and it works flawlessly. He wants to extend auto-merge â auto-deploy to the team's main production service, which is higher stakes. He needs a rollback mechanism that's reliable enough to trust with automated production deploys.\\n\\n**What Victor should do:** Victor should implement and test the rollback mechanism before enabling auto-deploy. The test: deploy a known-bad version (one that generates synthetic errors), verify that health checks detect it within the SLA, verify that rollback completes within the SLA, verify that monitoring shows the incident window and resolution. Only after passing this rollback test should auto-deploy be enabled for production. Victor should document the test procedure and run it quarterly as a \\\"rollback drill\\\" - this builds confidence in the automation and ensures degradation in rollback capability is detected before it matters.\"}],\"gettingStarted\":[\"Implement policy-based merge rules first\",\"Enable GitHub auto-merge for approved PRs\",\"Configure Mergify auto-merge rules\"],\"links\":[{\"title\":\"GitHub - Auto-merge Pull Requests\",\"url\":\"https://docs.github.com/en/pull-requests/collaborating-with-pull-requests/incorporating-changes-from-a-pull-request/automatically-merging-a-pull-request\",\"domain\":\"docs.github.com\"},{\"title\":\"Mergify - Auto-merge Configuration\",\"url\":\"https://docs.mergify.com/merge-protections/auto-merge/\",\"domain\":\"docs.mergify.com\"},{\"title\":\"ArgoCD - Automated Sync\",\"url\":\"https://argo-cd.readthedocs.io/en/stable/user-guide/auto_sync/\",\"domain\":\"argo-cd.readthedocs.io\"},{\"title\":\"Flagger - Automated Canary Deployments\",\"url\":\"https://flagger.app/\",\"domain\":\"flagger.app\"},{\"title\":\"Google SRE - Automated Rollbacks\",\"url\":\"https://sre.google/sre-book/managing-incidents/\",\"domain\":\"sre.google\"},{\"title\":\"VentureBeat - AI agents breached 395 organizations using credentials your IAM policy still treats as human\",\"url\":\"https://venturebeat.com/security/ai-agents-breached-395-organizations-using-credentials-your-iam-policy-still-treats-as-human\",\"domain\":\"venturebeat.com\"}]},\"5\":{\"guideSlug\":\"1000-merges-week-stripe-scale\",\"guideTitle\":\"Merge volume limited by product decisions, not by the merge path\",\"targetItems\":[{\"text\":\"Merge volume limited by product decisions, not by the merge path\",\"guideSlug\":\"1000-merges-week-stripe-scale\"},{\"text\":\"Agent produces PR â CI passes â merge â deploy â observe (n8n model: release lifecycle fully delegated to bots)\",\"guideSlug\":\"agent-produces-pr-ci-passes-merge-deploy-observe\"},{\"text\":\"Rollback is agent-driven\",\"guideSlug\":\"rollback-is-agent-driven\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob is excited about Stripe's 1000/week number but his team is at 25 PRs/day. He's trying to build a roadmap toward L5 but doesn't know
102whether to start with CI speed, merge infrastructure, or agent workflow improvements.\\n\\n**What Bob should do:** Bob should sequence the L5 infrastructure investments in the order that removes the current binding constraint. At 25 PRs/day, the binding constraint is almost certainly not CI speed (it's manageable) but merge workflow and review overhead. The right sequence: (1) implement merge queues and policy-based auto-merge to reach 50/day, (2) then optimize CI to under 10 minutes to scale to 100/day, (3) then implement continuous deployment with canary to reach 200+/day, (4) then invest in incremental builds and advanced merge queue for 1000+/week. Each step creates the prerequisite for the next. Bob should present this as a 12-18 month roadmap with measurable milestones at each level, not a single \\\"L5 project.\\\"\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah wants to track progress toward L5 throughput over time. She has throughput (PRs/day) but needs leading indicators that predict whether infrastructure investments are paying off before the throughput number moves.\\n\\n**What Sarah should do:** Sarah should identify the leading indicators for each infrastructure improvement: merge queue adoption (what % of PRs go through the queue?), auto-merge rate (what % are auto-merged?), CI time trend (is P95 CI time decreasing?), deployment frequency (how many times per day does code reach production?). These leading indicators move faster than throughput and tell you whether the infrastructure investments are working before the throughput ceiling is reached. Sarah should publish a monthly \\\"delivery infrastructure scorecard\\\" with these metrics alongside throughput, so the team can see infrastru
102cture progress even when throughput hasn't yet reflected the improvements.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been following Stripe's engineering blog closely and wants to implement a version of their Toolshed model for his team: a curated set of tools and permissions that each agent can access, with audit logging for every tool call. He believes this is the missing piece for scaling agent workflows beyond his personal use.\\n\\n**What Victor should do:** Victor should start with the minimal viable version of Toolshed: a small set of approved tools (run tests, create PR, query codebase, read documentation) with a permission wrapper that logs every call with agent identity and timestamp. This gives each agent session a bounded permission scope and creates an audit trail. Victor should pilot this for 30 days on his own agent sessions, verify that the audit trail is complete and actionable, then propose it as the team standard for agent tooling. The step from \\\"agents with unrestricted access\\\" to \\\"agents with scoped, audited access\\\" is the organizational infrastructure that makes L5 governance feasible.\"}],\"gettingStarted\":[\"Understand your current throughput ceiling and why it exists\",\"Invest in incremental build infrastructure\",\"Implement a high-throughput merge queue\"],\"links\":[{\"title\":\"Stripe Engineering - Minions: Enabling AI Agents\",\"url\":\"https://stripe.dev/blog/minions-stripes-one-shot-end-to-end-coding-agents\",\"domain\":\"stripe.dev\"},{\"title\":\"Stripe - Developer Productivity at Scale\",\"url\":\"https://stripe.com/blog/engineering\",\"domain\":\"stripe.com\"},{\"title\":\"EngFlow - Remote Build Execution\",\"url\":\"https://www.engflow.com/\",\"domain\":\"engflow.com\"},{\"title\":\"Bazel - Remote Caching\",\"url\":\"https://bazel.build/remote/caching\",\"domain\":\"bazel.build\"},{\"title\":\"Trunk.io - Enterprise Merge Queue\",\"url\":\"https://docs.trunk.io/merge\",\"domain\":\"docs.trunk.io\"}]}},\"Metrics\":{\"1\":{\"guideSlug\":\"dora-metrics-if-at-all\",\"guideTitle\":\"Delivery performance measured at all (DORA, SPACE or an equivalent set), if tracked\",\"targetItems\":[{\"text\":\"Delivery performance measured at all (DORA, SPACE or an equivalent set), if tracked\",\"guideSlug\":\"dora-metrics-if-at-all\"},{\"text\":\"Standard delivery metrics (not yet AI-specific)\",\"guideSlug\":\"no-ai-specific-metrics\"},{\"text\":\"ROI of AI not yet measured\",\"guideSlug\":\"how-much-did-we-save-silence\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$35\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been asked to evaluate whether the team's AI tool investments are working. She pulls the data she can find - PR counts, commit frequency, some velocity metrics from Jira - but nothing is connected to outcomes. She can tell that some developers are using AI tools more than others, but she can't tell whether that usage is translating into faster delivery.\\n\\n**What Sarah should do:** Sarah should recognize that the measurement infrastru
102cture needs to come before the impact measurement. Without DORA baselines, any AI impact analysis is speculative. Sarah should propose a 60-day measurement sprint: instrument deployment frequency and lead time for all teams, establish baselines, then overlay AI tool usage data to see if there's a correlation. The correlation won't be causal proof, but it will be the first data-driven signal about whether AI tool adoption is associated with improved delivery performance. That signal is enough to justify continued investment and more rigorous measurement at L2.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor is the team's most advanced AI user and is frustrated that he can't make the case for the AI patterns he's pioneered. He can feel the productivity improvement in his own work - he ships more and context-switches less - but he has no data to back up the claim when he advocates for team-wide adoption.\\n\\n**What Victor should do:** Victor should instrument his own DORA metrics as a case study. Track his personal deployment frequency, lead time from branch creation to production, and PR cycle time for three months before and after adopting parallel agents and CLI-first workflows. Even a single-person case study with clean data is more persuasive than a team-wide claim with no data. Victor can present this at the next engineering all-hands: \\\"Here is my delivery performance before and after. Here is what changed in how I work. Here is what would need to be true for the whole team to see similar results.\\\" A concrete case study with numbers is the strongest possible argument for the AI investment.\"}],\"gettingStarted\":[\"Audit what data you currently have\",\"Define \\\"deployment\\\" precisely\",\"Start with deployment frequency only\"],\"links\":[{\"title\":\"DORA Research Program - Google\",\"url\":\"https://dora.dev/\",\"domain\":\"dora.dev\"},{\"title\":\"DORA Metrics: The Four Key Metrics for DevOps\",\"url\":\"https://cloud.google.com/blog/products/devops-sre/using-the-four-keys-to-measure-your-devops-performance\",\"domain\":\"cloud.google.com\"},{\"title\":\"Four Keys - Open Source DORA Dashboard\",\"url\":\"https://github.com/dora-team/fourkeys\",\"domain\":\"github.com\"},{\"title\":\"Accelerate: The Science of DevOps - Forsgren, Humble, Kim\",\"url\":\"https://itrevolution.com/accelerate-book/\",\"domain\":\"itrevolution.com\"},{\"title\":\"DORA Quick Check - Self-Assessment Tool\",\"url\":\"https://dora.dev/quickcheck/\",\"domain\":\"dora.dev\"}]},\"2\":{\"guideSlug\":\"dora-basic-ai-tracking\",\"guideTitle\":\"A delivery-performance baseline plus basic AI tracking; per-session token spend; input tokens (context), not output, drive spend - watch power users well above the median\",\"targetItems\":[{\"text\":\"A delivery-performance baseline plus basic AI tracking; per-session token spend; input tokens (context), not output, drive spend - watch power users well above the median\",\"guideSlug\":\"dora-basic-ai-tracking\"},{\"text\":\"Licenses vs usage rate - but never token spend or seat activity as an adoption target; both are gamed within weeks, and Meta scrapped its 85k-employee token leaderboard and pulled AI usage from performance reviews\",\"guideSlug\":\"licenses-vs-usage-rate\"},{\"text\":\"PR throughput per dev as a proxy, never a target - and never suggestion acceptance rate, of which 31% is deleted within 15 minutes\",\"guideSlug\":\"pr-throughput-per-dev\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob has established DORA tracking across his teams and is now seeing that deployment frequency is up year-over-year, but he can't tell how much of that is AI tools vs. the new CI pipeline they rebuilt in Q2. He wants to isolate the AI contribution.\\n\\n**What Bob should do:** Bob should implement AI PR labeling retroactively where possible (many git commit messages and PR descriptions contain signals of AI involvement) and prospectively going forward. The key analytical question is: for PRs labeled `ai-assisted` or `ai-authored`, what is the lead time compared to `human-authored` PRs of similar size and complexity? Controlling for PR size removes a major confou
102nd (AI-generated PRs tend to be larger, which inflates lead time). Bob should present this analysis in the next quarterly business review as evidence of AI impact on delivery performance. The retroactive labeling won't be perfect, but it will be good enough to show directional impact.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has developer usage rate data from Copilot Business and basic DORA metrics from GitHub Insights. She wants to connect the two - to show that developers using Copilot more are shipping faster - but she's not sure how to do the analysis rigorously.\\n\\n**What Sarah should do:** Sarah should run a cohort analysis with three groups: non-users (0 Copilot suggestions accepted in a week), light users (1-10 suggestions/day), and heavy users (10+ suggestions/day). For each group, compute average PR cycle time and PR volume per week over the last quarter. The analysis will almost certainly show that heavy users have shorter cycle times and higher PR volume. Sarah should present this with the caveat that it's observational, not causal - but then follow up with a recommendation: target the non-users and light users for structured adoption support. The data doesn't prove causation, but it does identify where intervention will have the highest expected return.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor runs agent-heavy workflows and knows the DORA + basic tracking metrics are table stakes. He's already thinking about ITS and CPI, the L3 metrics. But he recognizes that the team needs to build the L2 foundation before jumping to L3.\\n\\n**What Victor should do:** Victor should help build the L2 instrumentation that will feed into L3. Specifically: the PR labeling convention needs to be more granular than a simple `ai-assisted` label. Victor should propose a tagging schema that captures the agent type (Claude Code, Copilot, Cursor), the task type (new feature, bug fix, refactor, test writing), and the autonomy level (fully autonomous, human-guided, collaborative). This richer metadata makes the L3 metrics - ITS and CPI - much more meaningful because they can be segmented by task type and agent. Building this schema now is the infrastructure investment that makes L3 metrics analysis straightforward rather than a data archaeology project.\"}],\"gettingStarted\":[\"Implement DORA tracking if you haven't already\",\"Create an AI-labeling convention\",\"Instrument AI tool usage rates\"],\"links\":[{\"title\":\"Four Keys - Open Source DORA Implementation\",\"url\":\"https://github.com/dora-team/fourkeys\",\"domain\":\"github.com\"},{\"title\":\"GitHub Copilot Metrics API\",\"url\":\"https://docs.github.com/en/rest/copilot/copilot-metrics\",\"domain\":\"docs.github.com\"},{\"title\":\"LinearB - Engineering Metrics Platform\",\"url\":\"https://linearb.io/\",\"domain\":\"linearb.io\"},{\"title\":\"Accelerate: The Science of DevOps\",\"url\":\"https://itrevolution.com/accelerate-book/\",\"domain\":\"itrevolution.com\"},{\"title\":\"Jellyfish - Engineering Management Platform\",\"url\":\"https://jellyfish.co/\",\"domain\":\"jellyfish.co\"}]},\"3\":{\"guideSlug\":\"cpi-cost-per-iteration-target-0-50\",\"guideTitle\":\"Cost per iteration (CPI) measured per task, not per token: cheaper models can cost more per task (Gemini 3.8 Flash: same token price, $0.40 -\u003e $0.58 per task), and cheaper tokens made sessions 3.3x longer; include CI compute and the production compute the generated code burns (+5-8%)\",\"targetItems\":[{\"text\":\"Cost per iteration (CPI) measured per task, not per token: cheaper models can cost more per task (Gemini 3.8 Flash: same token price, $0.40 -\u003e $0.58 per task), and cheaper tokens made sessions 3.3x longer; include CI compute and the production compute the generated code burns (+5-8%)\",\"guideSlug\":\"cpi-cost-per-iteration-target-0-50\"},{\"text\":\"Iterations to success (ITS): counted per change, limit set by the team\",\"guideSlug\":\"its-iterations-to-success-target-1-3\"},{\"text\":\"Push-to-result time: median recorded per reporting period\",\"guideSlug\":\"ci-feedback-latency-tracking\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$36\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$37\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$38\"}],\"gettingStarted\":[\"Instrument token costs per agent session\",\"Instrument CI runner costs per iteration\",\"Build a per-PR cost rollup\"],\"links\":[{\"title\":\"Anthropic API Pricing\",\"url\":\"https://www.anthropic.com/pricing\",\"domain\":\"anthropic.com\"},{\"title\":\"GitHub Actions Billing and Usage\",\"url\":\"https://docs.github.com/en/billing/managing-billing-for-your-products/managing-billing-for-github-actions/about-billing-for-github-actions\",\"domain\":\"docs.github.com\"},{\"title\":\"Model Context Protocol - Efficient Context Management\",\"url\":\"https://modelcontextprotocol.io/introduction\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"Building Cost-Efficient LLM Applications\",\"url\":\"https://www.anthropic.com/research/building-effective-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"Optimizing AI Agent Economics - Token Usage Strategies\",\"url\":\"https://docs.anthropic.com/en/docs/build-with-claude/prompt-engineering/overview\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"AI-generated C++ in production: 3.52M changes and the compute they cost (arXiv 2608.06640)\",\"url\":\"https://arxiv.org/abs/2608.06640\",\"domain\":\"arxiv.org\"},{\"title\":\"Giles Edwards-Alexander - The economic benefit of refactoring\",\"url\":\"https://martinfowler.com/articles/exploring-gen-ai/refactoring-economic-benefit.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"Anthropic - Claude Opus 5.5 built for coding sessions that use more context\",\"url\":\"https://claude.com/blog/claude-opus-5-5-built-for-coding-sessions-that-use-more-context\",\"domain\":\"claude.com\"},{\"title\":\"Ramp AI Index - September 2026\",\"url\":\"https://ramp.com/data/ai-index-sept-2026\",\"domain\":\"ramp.com\"}
102,{\"title\":\"Uber - Running a software factory efficiently at Uber scale\",\"url\":\"https://www.uber.com/blog/software-factory-uber-scale/\",\"domain\":\"uber.com\"},{\"title\":\"TechCrunch - OpenAI launches GPT-6 Sol and Luna\",\"url\":\"https://techcrunch.com/2026/09/22/openai-launches-gpt-6-sol-and-luna/\",\"domain\":\"techcrunch.com\"}]},\"4\":{\"guideSlug\":\"tors-95\",\"guideTitle\":\"Test-oracle reliability tracked as a metric, alongside model-regression signals (thinking length, files read before edit)\",\"targetItems\":[{\"text\":\"Test-oracle reliability tracked as a metric, alongside model-regression signals (thinking length, files read before edit)\",\"guideSlug\":\"tors-95\"},{\"text\":\"Every agent run terminates in a classified state - success / flawed / blocked / manual - and each non-success class routes to a different fix; only the classification rate justifies expanding the automation boundary\",\"guideSlug\":\"auto-approve-rate-target-60\"},{\"text\":\"Agent Autonomy Score: % tasks without human intervention\",\"guideSlug\":\"agent-autonomy-score-tasks-without-human-intervention\"},{\"text\":\"Merge Queue Wait \u003c 10 min\",\"guideSlug\":\"merge-queue-wait-10-min\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has an auto-approve system that's not performing as expected. The system is set to auto-merge PRs when CI passes, but it's flagging PRs for human review more than expected. The root cause turns out to be that the system can't distinguish real CI failures from flaky ones - and the team's TORS is 82%.\\n\\n**What Bob should do:** Bob should frame the TORS improvement project as a prerequisite for the auto-approve system working correctly. The target is 95% TORS within 90 days. He should assign one senior engineer for 2 weeks to identify and quarantine the 20 flakiest tests (which will likely bring TORS from 82% to ~90%). Then he should budget one engineer-sprint per month for the following 2 months to fix the quarantined tests at root cause. The connection between TORS and auto-approve rate makes the investment easy to justify: every 5% improvement in TORS enables a measurable improvement in auto-approve rate, which directly reduces review queue burden on human developers.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is investigating why some teams using agents have much better throughput than others. She's collected ITS data but the correlation with productivity is weaker than expected. She suspects test reliability is a confounding variable.\\n\\n**What Sarah should do:** Sarah should pull TORS data alongside ITS data for each team and compute the correlation. Her hypothesis: teams with low TORS have artificially inflated ITS (because agents are responding to false failures), which makes their agent workflows appear less efficient than they would be with reliable tests. If the data confirms this, Sarah has a compelling story: improving TORS is the fastest path to reducing ITS for the low-performing teams, because the apparent quality gap between teams is partly a test reliability gap, not a capability gap. This reframes the intervention: instead of retraining teams on agent workflow, fix the tests that are giving agents false signals.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$39\"}],\"gettingStarted\":[\"Implement automatic test retry and failure tagging\",\"Build a TORS dashboard\",\"Identify the flakiest tests by frequency\"],\"links\":[{\"title\":\"Google Testing Blog - Test Flakiness\",\"url\":\"https://testing.googleblog.com/2020/12/test-flakiness-one-of-main-challenges.html\",\"domain\":\"testing.googleblog.com\"},{\"title\":\"GitHub Actions - Retry Failed Tests\",\"url\":\"https://docs.github.com/en/actions/how-tos/write-workflows/choose-when-workflows-run/control-jobs-with-conditions\",\"domain\":\"docs.github.com\"},{\"title\":\"Quarantine Patterns for Flaky Tests\",\"url\":\"https://martinfowler.com/articles/nonDeterminism.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"BuildKite - Flaky Test Management\",\"url\":\"https://buildkite.com/docs/test-engine/flaky-test-management\",\"domain\":\"buildkite.com\"},{\"title\":\"Test Reliability at Scale - Shopify Engineering\",\"url\":\"https://shopify.engineering/unreasonable-effectiveness-test-retries-android-monorepo-case-study\",\"domain\":\"shopify.engineering\"}]},\"5\":{\"guideSlug\":\"cost-per-feature-not-cost-per-pr\",\"guideTitle\":\"Cost-per-feature (not cost-per-PR); CFO scorecard: Useful Work, Cost per Successful Task, Return on Compute - paired with incidents-per-merged-change and firefighting hours, which move in the opposite direction\",\"targetItems\":[{\"text\":\"Cost-per-feature (not cost-per-PR); CFO scorecard: Useful Work, Cost per Successful Task, Return on Compute - paired with incidents-per-merged-change and firefighting hours, which move in the opposite direction\",\"guideSlug\":\"cost-per-feature-not-cost-per-pr\"},{\"text\":\"Business value throughput, not activity metrics\",\"guideSlug\":\"business-value-throughput-not-activity-metrics\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$3a\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$3b\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$3c\"}],\"gettingStarted\":[\"Define \\\"feature\\\" in your context\",\"Connect engineering work to feature tags\",\"Aggregate costs by feature\"],\"links\":[{\"title\":\"Value Stream Mapping for Software Delivery\",\"url\":\"https://itrevolution.com/articles/value-stream-mapping/\",\"domain\":\"itrevolution.com\"},{\"title\":\"Engineering Economics: Measuring Return on Investment\",\"url\":\"https://www.mckinsey.com/industries/technology-media-and-telecommunications/our-insights/yes-you-can-measure-software-developer-productivity\",\"domain\":\"mckinsey.com\"},{\"title\":\"Feature Flags and Feature-Level Delivery Tracking\",\"url\":\"https://launchdarkly.com/blog/feature-flag-driven-development/\",\"domain\":\"launchdarkly.com\"},{\"title\":\"Flow Metrics for Software Delivery\",\"url\":\"https://flowframework.org/\",\"domain\":\"flowframework.org\"},{\"title\":\"Measuring the Cost of Delay in Software Development\",\"url\":\"https://www.infoq.com/news/2015/02/cost-of-delay/\",\"domain\":\"infoq.com\"},{\"title\":\"OpenAI - A scorecard for the AI age\",\"url\":\"https://openai.com/index/a-scorecard-for-the-ai-age\",\"domain\":\"openai.com\"}]}},\"Governance \u0026 Compliance\":{\"1\":{\"guideSlug\":\"shadow-ai-devs-with-private-subscriptions\",\"guideTitle\":\"Individual devs use their own AI subscriptions\",\"targetItems\":[{\"text\":\"Individual devs use their own AI subscriptions\",\"guideSlug\":\"shadow-ai-devs-with-private-subscriptions\"},{\"text\":\"AI usage not yet audited\",\"guideSlug\":\"zero-audit-trail\"},{\"text\":\"AI usage is informal, policy not yet defined\",\"guideSlug\":\"banning-ai-answer-from-2024\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$3d\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$3e\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$3f\"}],\"gettingStarted\":[\"Do a shadow AI census first\",\"Classify the risk by tool t
102ype\",\"Identify the supply gap\"],\"links\":[{\"title\":\"GitHub Copilot Enterprise - Data Handling\",\"url\":\"https://docs.github.com/en/copilot/github-copilot-enterprise/overview/about-github-copilot-enterprise\",\"domain\":\"docs.github.com\"},{\"title\":\"Shadow AI in Enterprise: 2024 Survey - GitLab\",\"url\":\"https://about.gitlab.com/developer-survey/\",\"domain\":\"about.gitlab.com\"},{\"title\":\"SOC2 and AI Tools: What Auditors Are Asking - AICPA\",\"url\":\"https://www.aicpa-cima.com/resources/download/2017-trust-services-criteria-with-revised-points-of-focus-2022\",\"domain\":\"aicpa-cima.com\"},{\"title\":\"GDPR and Generative AI: Data Processor Obligations\",\"url\":\"https://www.edpb.europa.eu/our-work-tools/our-documents/guidelines/guidelines-42019-article-25-data-protection-design-and_en\",\"domain\":\"edpb.europa.eu\"},{\"title\":\"Anthropic Claude for Enterprise - Privacy and Security\",\"url\":\"https://privacy.claude.com/en/\",\"domain\":\"privacy.claude.com\"},{\"title\":\"EY - Autonomous AI implementation outpaces oversight (September 2026)\",\"url\":\"https://www.ey.com/en_us/newsroom/2026/09/ey-survey-finds-that-autonomous-ai-implementation-outpaces-oversight-yielding-an-ai-governance-gap\",\"domain\":\"ey.com\"}]},\"2\":{\"guideSlug\":\"official-ai-tool-policy\",\"guideTitle\":\"Official AI tool policy; per-session spend caps, short-lived keys and kill switches; autonomy set by a declarative deny/ask ruleset that is reviewed and version-controlled, not by a human clicking approve; a deliberate default for new vendor features instead of letting them auto-enable (Copilot's \\\"Default policy for new features\\\")\",\"targetItems\":[{\"text\":\"Official AI tool policy; per-session spend caps, short-lived keys and kill switches; autonomy set by a declarative deny/ask ruleset that is reviewed and version-controlled, not by a human clicking approve; a deliberate default for new vendor features instead of letting them auto-enable (Copilot's \\\"Default policy for new features\\\")\",\"guideSlug\":\"official-ai-tool-policy\"},{\"text\":\"Basic audit: who uses what - and every agent has a named human owner, because 51% of organisations cannot say who owns their AI identities\",\"guideSlug\":\"basic-audit-who-uses-what\"},{\"text\":\"Regulatory obligations for your jurisdiction and sector are identified and owned (EU AI Act Article 50 transparency applicable since Aug 2 2026, high-risk duties deferred to Dec 2027 / Aug 2028; China tiers agents by decision authority); data-residency routing\",\"guideSlug\":\"eu-ai-act-awareness\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob needs to deliver an AI governance framework to the CISO within 30 days as a condition of getting budget for official AI tool procurement. He has done a shadow AI census and knows what tools developers are using and what data handling risks exist. Now he needs to turn that knowledge into a policy document.\\n\\n**What Bob should do:** Bob should start with the one-page draft approach. In a single document: approved tools list (those with enterprise DPAs that address the CISO's concerns), data classification table (what can and cannot go to AI systems), PR disclosure requirement (text for the PR template), and policy owner designation. Bob should review this draft with the CISO and the legal team before publishing - one round of feedback with a one-week turnaround. The output of that review is the v1 policy. Bob should publish it with a 30-day grace period and use the grace period to get enterprise licenses procured and deployed. Policy without tooling is ineffective; they need to ship together.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$40\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$41\"}],\"gettingStarted\":[\"Draft a one-page policy first\",\"Start the approved tools list with what developers already use\",\"Create the data classification section carefully\"],\"links\":[{\"title\":\"Anthropic Acceptable Use Policy\",\"url\":\"https://www.anthropic.com/legal/aup\",\"domain\":\"anthropic.com\"},{\"title\":\"GitHub Copilot Policy for Organizations\",\"url\":\"https://docs.github.com/en/copilot/managing-copilot/managing-github-copilot-in-your-organization\",\"domain\":\"docs.github.com\"},{\"title\":\"NIST AI RMF: Govern Function\",\"url\":\"https://www.nist.gov/itl/ai-risk-management-framework\",\"domain\":\"nist.gov\"},{\"title\":\"CISA: Guidelines for Secure AI System Development\",\"url\":\"https://www.cisa.gov/news-events/alerts/2023/11/26/cisa-and-uk-ncsc-unveil-joint-guidelines-secure-ai-system-development\",\"domain\":\"cisa.gov\"},{\"title\":\"ISO 42001: AI Management System Standard\",\"url\":\"https://www.iso.org/standard/81230.html\",\"domain\":\"iso.org\"},{\"title\":\"The Register - Claude Code puts auto mode in the driver's seat\",\"url\":\"https://www.theregister.com/ai-and-ml/2026/08/10/claude-code-puts-auto-mode-in-the-drivers-seat/5285326\",\"domain\":\"theregister.com\"}
102,{\"title\":\"EY - Autonomous AI implementation outpaces oversight (September 2026)\",\"url\":\"https://www.ey.com/en_us/newsroom/2026/09/ey-survey-finds-that-autonomous-ai-implementation-outpaces-oversight-yielding-an-ai-governance-gap\",\"domain\":\"ey.com\"}]},\"3\":{\"guideSlug\":\"minimum-viable-audit-trail-model-timestamp-context-approver\",\"guideTitle\":\"Minimum viable audit trail: model, timestamp, context, approver - and the agent's own identity, registered as a distinct principal in the IdP (Entra Agent ID, Okta, Google Agent Identity), not a developer's token\",\"targetItems\":[{\"text\":\"Minimum viable audit trail: model, timestamp, context, approver - and the agent's own identity, registered as a distinct principal in the IdP (Entra Agent ID, Okta, Google Agent Identity), not a developer's token\",\"guideSlug\":\"minimum-viable-audit-trail-model-timestamp-context-approver\"},{\"text\":\"Policy-as-code; all model traffic through an AI gateway (LiteLLM, Kong AI Gateway 2.0, Agent Router, Claude apps gateway, Cloudflare): central keys with BYOK-only, per-team budgets, model allowlist with exact-version pinning and hard denies (`availableModelsMatch`, `deniedModels`, Codex `requirements.toml`); repository-supplied agent config (`.claude/`, `.vscode/`, `.cursor/`, `.git/config`, `build.rs`) on a mandatory-diff path, because it executes on folder open\",\"guideSlug\":\"policy-as-code\"},{\"text\":\"Compliance gates in CI\",\"guideSlug\":\"compliance-gates-in-ci\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$42\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah wants to use the MVAT data to build model version performance analysis - correlating which Claude model version generated code with downstream defect rates and review round counts. This would be the first data-driven input to AI tool version adoption decisions.\\n\\n**What Sarah should do:** Sarah should start by verifying that the model field is captured consistently and accurately before building the analysis on top of it. She should pull the last 90 days of MVAT records and check: what model versions appear, what's the distribution, are there obviously wrong values (free-text instead of structured model IDs)? Once she has confidence in the data quality, she can join the MVAT records with defect tracking data (bugs filed in the two weeks after a PR merged) and review round counts. The resulting analysis - \\\"PRs generated
102with claude-3-5-sonnet have X% lower defect rate than claude-3-opus\\\" - gives the team a data-driven basis for model selection rather than relying on intuition or marketing claims.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$43\"}],\"gettingStarted\":[\"Define the schema as a structured git trailer\",\"Build a commit hook or CLI wrapper\",\"Add CI validation\"],\"links\":[{\"title\":\"Git Interpret-Trailers Documentation\",\"url\":\"https://git-scm.com/docs/git-interpret-trailers\",\"domain\":\"git-scm.com\"},{\"title\":\"SLSA Provenance Specification\",\"url\":\"https://slsa.dev/provenance/v1\",\"domain\":\"slsa.dev\"},{\"title\":\"SOC2 CC8: Change Management Trust Service Criteria\",\"url\":\"https://us.aicpa.org/interestareas/informationtechnology/resources/systemsandorganizationcontrolsforserviceorganizations\",\"domain\":\"us.aicpa.org\"},{\"title\":\"OpenTelemetry for AI Observability\",\"url\":\"https://opentelemetry.io/docs/concepts/signals/traces/\",\"domain\":\"opentelemetry.io\"},{\"title\":\"Sigstore Rekor: Immutable Transparency Log\",\"url\":\"https://docs.sigstore.dev/rekor/overview/\",\"domain\":\"docs.sigstore.dev\"},{\"title\":\"Okta - AI innovations announced at Oktane 2026\",\"url\":\"https://www.okta.com/newsroom/press-releases/ai-innovations-oktane-2026/\",\"domain\":\"okta.com\"},{\"title\":\"CrowdStrike - Agentic Identity Provider\",\"url\":\"https://www.crowdstrike.com/en-us/blog/crowdstrike-announces-agentic-identity-provider/\",\"domain\":\"crowdstrike.com\"}]},\"4\":{\"guideSlug\":\"full-provenance-tracking-per-change\",\"guideTitle\":\"Full provenance tracking per change (cryptographic agent traces; commit-to-prompt lineage via gateway prompt IDs; prompt and response logs in the SIEM with retention beyond vendor defaults - Cursor keeps 30 days); agent access certified like human access and revoked when the owner leaves\",\"targetItems\":[{\"text\":\"Full provenance tracking per change (cryptographic agent traces; commit-to-prompt lineage via gateway prompt IDs; prompt and response logs in the SIEM with retention beyond vendor defaults - Cursor keeps 30 days); agent access certified like human access and revoked when the owner leaves\",\"guideSlug\":\"full-provenance-tracking-per-change\"},{\"text\":\"Automated compliance checks; the AI gateway run as critical infrastructure with a patch SLA (LiteLLM's MCP auth bypass reached the CISA KEV list), DLP at the gateway, and unapproved agents detectable - 26% of large firms cannot; skills and MCP servers allowlisted and pinned rather than scanned, assessed against OWASP Agentic Skills Top 10\",\"guideSlug\":\"automated-compliance-checks\"},{\"text\":\"AI code vs human code distinction in VCS (Kubernetes model: disclosure mandatory, AI commit messages banned; humans write the why)\",\"guideSlug\":\"ai-code-vs-human-code-distinction-in-vcs\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$44\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking AI adoption metrics but wants to build a more sophisticated analysis: what types of tasks are AI most effective at, and does effectiveness correlate with any provenance attributes (model version, human modification rate, task type)? The provenance graph gives her the data to answer these questions systematically.\\n\\n**What Sarah should do:** Sarah should build an analysis pipeline on top of the provenance graph. Starting with a cohort of the last 200 AI-assisted PRs, she can compute: percentage of AI-generated code that was modified during human review (high modification = AI less effective for this task type), time from AI session to PR merge (efficiency measure), downstream defect rate by task type and model version. The goal is a \\\"what works and what doesn't\\\" map that guides how the team uses AI tools. Sarah should present this analysis quarterly and use it to update the guidance on which tasks are best suited for AI assistance.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$45\"}],\"gettingStarted\":[\"Implement SLSA Level 2 or 3 for your build pipeline\",\"Define the AI provenance schema extension\",\"Integrate Claude Code session export\"],\"links\":[{\"title\":\"SLSA Framework: Supply Chain Levels for Software Artifacts\",\"url\":\"https://slsa.dev/\",\"domain\":\"slsa.dev\"},{\"title\":\"SLSA GitHub Actions Generator\",\"url\":\"https://github.com/slsa-framework/slsa-github-generator\",\"domain\":\"github.com\"},{\"title\":\"Sigstore: Keyless Signing for Software Artifacts\",\"url\":\"https://www.sigstore.dev/\",\"domain\":\"sigstore.dev\"},{\"title\":\"in-toto: A Framework to Protect Software Supply Chains\",\"url\":\"https://in-toto.io/\",\"domain\":\"in-toto.io\"},{\"title\":\"EU AI Act Article 12: Transparency and Record-Keeping\",\"url\":\"https://eur-lex.europa.eu/legal-content/EN/TXT/?uri=CELEX%3A32024R1689\",\"domain\":\"eur-lex.europa.eu\"},{\"title\":\"Claude Code changelog - gateway prompt ID header\",\"url\":\"https://code.claude.com/docs/en/changelog\",\"domain\":\"code.claude.com\"},{\"title\":\"Okta - AI innovations announced at Oktane 2026\",\"url\":\"https://www.okta.com/newsroom/press-releases/ai-innovations-oktane-2026/\",\"domain\":\"okta.com\"}]},\"5\":{\"guideSlug\":\"continuous-compliance-agent-monitors-regulatory-changes\",\"guideTitle\":\"Continuous compliance: agent monitors regulatory changes\",\"targetItems\":[{\"text\":\"Continuous compliance: agent monitors regulatory changes\",\"guideSlug\":\"continuous-compliance-agent-monitors-regulatory-changes\"},{\"text\":\"Self-documenting audit trail: every agent action traces to one agent identity and one accountable human, and the urgent fast path runs through the same controls (47% of firms with written policies skipped them for urgent deployments)\",\"guideSlug\":\"self-documenting-audit-trail\"},{\"text\":\"Enterprise-grade RBAC per agent: task-scoped permissions with
102no standing access, new agents start on probation and earn scope from their track record\",\"guideSlug\":\"enterprise-grade-rbac-per-agent-stripe-toolshed-model-400-mc\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$46\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$47\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$48\"}],\"gettingStarted\":[\"Define the regulatory scope\",\"Build the regulatory feed subscriptions\",\"Implement the impact assessment pipeline\"],\"links\":[{\"title\":\"EUR-Lex: EU AI Act and Implementing Regulations\",\"url\":\"https://eur-lex.europa.eu/legal-content/EN/TXT/?uri=CELEX%3A32024R1689\",\"domain\":\"eur-lex.europa.eu\"},{\"title\":\"EU AI Office: Official Updates and Guidance\",\"url\":\"https://digital-strategy.ec.europa.eu/en/policies/ai-office\",\"domain\":\"digital-strategy.ec.europa.eu\"},{\"title\":\"AICPA: SOC2 Standards Update Notifications\",\"url\":\"https://us.aicpa.org/interestareas/informationtechnology/resources/systemsandorganizationcontrolsforserviceorganizations\",\"domain\":\"us.aicpa.org\"},{\"title\":\"NIST AI RMF: Continuous Monitoring Guidance\",\"url\":\"https://www.nist.gov/artificial-intelligence\",\"domain\":\"nist.gov\"},{\"title\":\"GDPR Enforcement Tracker - DLA Piper\",\"url\":\"https://www.dlapiperdataprotection.com/\",\"domain\":\"dlapiperdataprotection.com\"}]}}},\"organization\":{\"AI Adoption Model\":{\"1\":{\"guideSlug\":\"big-bang-let-s-buy-100-licenses\",\"guideTitle\":\"Adoption via bulk license purchase\",\"targetItems\":[{\"text\":\"Adoption via bulk license purchase\",\"guideSlug\":\"big-bang-let-s-buy-100-licenses\"},{\"text\":\"Initial enthusiasm fades to low usage\",\"guideSlug\":\"enthusiasm-silence-shelfware\"},{\"text\":\"Powerful tools on an unprepared process\",\"guideSlug\":\"ferrari-engine-in-fiat-126p\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob is getting pressure from the CTO to \\\"move faster on AI.\\\" Three of Bob's peers at other companies have announced company-wide Copilot deployments. The board is asking about AI strategy. Bob is tempted to just buy the licenses and declare victory - at least it signals momentum.\\n\\n**What Bob should do:** Bob needs to reframe the conversation with the CTO. The question isn't \\\"do we have AI tools?\\\" - it's \\\"are our developers actually getting faster?\\\" Bob should propose a 90-day structured pilot with 2 teams, measurable outcomes, and a clear expansion plan if the pilot succeeds. This is a harder story to tell in a board deck than \\\"100 licenses deployed,\\\" but it's the story that leads to actual results. Bob should also identify who will own the initiative day-to-day - without a named owner, even a well-structured pilot fails. The owner doesn't have to be full-time on AI, but they need 20-30% of their time committed to the program.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been asked to report on AI tool adoption after a big-bang license deployment six months ago. She pulls the usage data: 23% weekly active usage, concentrated in 4-5 developers. The other 75 license holders haven't logged in in two months. Leadership is asking whether to renew.\\n\\n**What Sarah should do:** Sarah should present the data honestly and use it to make the case for a structured reset. The current deployment is in the shelfware zone - not zero value, but not the ROI that justified the investment. The 4-5 active power users are the seed of a proper pilot cohort. Sarah should propose converting from a broad deployment to a focused program: identify the 10-15 most engaged developers, pair them with a champion (Victor), build a proper onboarding track, and measure outcomes at 30-60-90 days. The renewal decision should be contingent on the pilot producing measurable outcomes, not on optimistic projections.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor is one of the 4-5 active users from the big-bang deployment. He's gotten real value from the tool - his PR throughput is up, he's using it for code review prep and test generation. But he's isolated. Nobody else on the team is using it consistently, there's no shared knowledge of what works, and he feels like he's doing something niche rather than something the org is investing in.\\n\\n**What Victor should do:** Victor should document his workflow and make it visible. A short internal post - \\\"here's how I'm actually using Copilot and what I'm getting from it\\\" - does two things: it surfaces the concrete value story that justifies continued investment, and it recruits other developers to try the specific workflows that work. Victor should also propose to Bob that the org formalize his role as the AI champion for a structured pilot. The transition from \\\"one guy who figured it out\\\" to \\\"internal expert running a program\\\" is the difference between individual productivity and organizational capability.\"}
102],\"gettingStarted\":[\"Resist the big-bang impulse\",\"Buy a small number of licenses first\",\"Pair licenses with a champion\"],\"links\":[{\"title\":\"How GitHub Measures Copilot Adoption\",\"url\":\"https://github.blog/news-insights/research/research-quantifying-github-copilots-impact-on-developer-productivity-and-happiness/\",\"domain\":\"github.blog\"},{\"title\":\"The Hype Cycle and Why Enterprise Software Adoptions Fail\",\"url\":\"https://www.gartner.com/en/research/methodologies/gartner-hype-cycle\",\"domain\":\"gartner.com\"},{\"title\":\"Change Management for Technology Adoption - Prosci\",\"url\":\"https://www.prosci.com/en/change-management-in-it\",\"domain\":\"prosci.com\"},{\"title\":\"Developer Experience and Productivity - McKinsey\",\"url\":\"https://www.mckinsey.com/industries/technology-media-and-telecommunications/our-insights/yes-you-can-measure-software-developer-productivity\",\"domain\":\"mckinsey.com\"}]},\"2\":{\"guideSlug\":\"pilot-teams-2-3-teams\",\"guideTitle\":\"Pilot teams (2-3 teams)\",\"targetItems\":[{\"text\":\"Pilot teams (2-3 teams)\",\"guideSlug\":\"pilot-teams-2-3-teams\"},{\"text\":\"Adoption spreads through peer networks rather than mandates\",\"guideSlug\":\"internal-champion\"},{\"text\":\"Pilot metrics; track cost per task from day one - cheaper tokens lengthen sessions rather than shrink bills\",\"guideSlug\":\"pilot-metrics\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob has been asked by the CTO to \\\"get AI deployed\\\" before the next board meeting in 90 days. The pressure is to announce something visible. Bob's instinct is to buy broad and announce it as a win. But he's also seen the enthusiasm-silence-shelfware arc play out before with other tool rollouts and knows what happens when access is not paired with adoption infrastructure.\\n\\n**What Bob should do:** Bob should reframe the 90-day deadline as an opportunity to run a proper pilot rather than a reason to skip it. A well-structured pilot with 2-3 teams, measurable outcomes, and an expansion plan ready to activate is a better board story than a broad deployment with 20% usage. Bob should identify which 2-3 teams are best positioned to succeed, assign champions, and set a 60-day pilot with a built-in expansion decision. The board story becomes: \\\"We ran a structured pilot, here are the results, here is the expansion plan we are executing.\\\" That is a more credible AI story than license counts.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is responsible for tracking AI adoption across the engineering organization. She has usage data from a previous unstructured rollout that shows 22% weekly active usage after 4 months, with usage concentrated in 6-7 developers. Leadership wants to know if the investment is working and what to do next.\\n\\n**What Sarah should do:** Sarah should use the existing usage data to identify the 2-3 teams where adoption is strongest and propose converting from a broad unstructured deployment to a focused pilot structure. The 6-7 active users are the seed champions. The teams they're on are the pilot candidates. Sarah should propose a 60-day structured pilot with defined metrics, team-level champions, and a clear expansion criterion. This converts a stalled broad deployment into a structured learning exercise without requiring a restart - the existing license investment is repurposed rather than written off.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been one of the active users from the previous rollout. He has developed genuine workflow expertise - he knows which tasks benefit most from AI assistance, which prompting approaches work best for the codebase, and which pitfalls to avoid. This knowledge lives in his head and hasn't been transferred to anyone else.\\n\\n**What Victor should do:** Victor should volunteer to be the champion for one of the pilot teams and use the pilot structure to externalize his knowledge. He should document three to five specific workflows - with before/after examples, specific prompts, and honest notes on failure modes - as the pilot playbook. The playbook is more valuable than the pilot itself, because it is the thing that makes expansion possible. Victor should also propose to Bob that champion responsibilities be formally recognized - not as extra work on top of his
102normal role, but as a legitimate allocation of 20-30% of his time during the pilot period.\"}],\"gettingStarted\":[\"Define the pilot scope before selecting tools\",\"Select teams for success probability\",\"Assign a champion per team\"],\"links\":[{\"title\":\"How to Run a Successful Technology Pilot - MIT Sloan\",\"url\":\"https://sloanreview.mit.edu/article/a-better-way-to-pilot-emerging-technologies/\",\"domain\":\"sloanreview.mit.edu\"},{\"title\":\"The Lean Startup - Eric Ries\",\"url\":\"https://theleanstartup.com/\",\"domain\":\"theleanstartup.com\"},{\"title\":\"Developer Experience Pilots at Spotify\",\"url\":\"https://engineering.atspotify.com/2020/08/how-we-use-golden-paths-to-solve-fragmentation-in-our-software-ecosystem\",\"domain\":\"engineering.atspotify.com\"},{\"title\":\"Measuring Software Delivery Performance - DORA\",\"url\":\"https://dora.dev/research/\",\"domain\":\"dora.dev\"}]},\"3\":{\"guideSlug\":\"platform-team-owns-ai-tooling\",\"guideTitle\":\"Platform team owns AI tooling: a central proxy or agent registry supplying identity, cost tracking, sandboxing, observability and evals by default - declare an agent once, get production-readiness in minutes rather than weeks\",\"targetItems\":[{\"text\":\"Platform team owns AI tooling: a central proxy or agent registry supplying identity, cost tracking, sandboxing, observability and evals by default - declare an agent once, get production-readiness in minutes rather than weeks\",\"guideSlug\":\"platform-team-owns-ai-tooling\"},{\"text\":\"Internal Developer Platform with AI layer, plus enablement run as a named programme with attendance you can count (a standing guild, guided hackathons, hands-on labs) rather than a launch email\",\"guideSlug\":\"internal-developer-platform-with-ai-layer\"},{\"text\":\"Standardized agent setup per team, but no mandated tool - the platform is standard, the choice is free, and outcomes are what get measured; a default-model policy set centrally (Ramp: firms restricting frontier use cut spend per employee 9.7% while token prices fell 41%); \\\"bad day protocol\\\" for model and harness regressions, plus a vendor-exit plan\",\"guideSlug\":\"standardized-agent-setup-per-team\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$49\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$4a\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$4b\"}],\"gettingStarted\":[\"Audit what exists before building\",\"Define the platform team's scope explicitly\",\"Start with the things that should never be per-team\"],\"links\":[{\"title\":\"Agentic Engineering at Zalando: A Snapshot\",\"url\":\"https://engineering.zalando.com/posts/2026/08/agentic-engineering-at-zalando-a-snapshot.html\",\"domain\":\"engineering.zalando.com\"},{\"title\":\"Why Ramp Built Inspect - The Pragmatic Engineer\",\"url\":\"https://newsletter.pragmaticengineer.com/p/why-ramp-built-inspect\",\"domain\":\"newsletter.pragmaticengineer.com\"},{\"title\":\"An Operating Model for Enterprise AI Agent Reliability - Thoughtworks\",\"url\":\"https://www.thoughtworks.com/insights/blog/generative-ai/operating-model-enterprise-ai-agent-reliability\",\"domain\":\"thoughtworks.com\"}
102,{\"title\":\"Team Topologies - Matthew Skelton and Manuel Pais\",\"url\":\"https://teamtopologies.com/\",\"domain\":\"teamtopologies.com\"},{\"title\":\"Platform Engineering at Spotify - Golden Paths\",\"url\":\"https://engineering.atspotify.com/2020/08/how-we-use-golden-paths-to-solve-fragmentation-in-our-software-ecosystem\",\"domain\":\"engineering.atspotify.com\"},{\"title\":\"Internal Developer Platforms - backstage.io\",\"url\":\"https://backstage.io/\",\"domain\":\"backstage.io\"},{\"title\":\"The Platform Engineering Guide - CNCF\",\"url\":\"https://tag-app-delivery.cncf.io/whitepapers/platforms/\",\"domain\":\"tag-app-delivery.cncf.io\"},{\"title\":\"Kong AI Gateway 2.0 GA - Kong\",\"url\":\"https://konghq.com/blog/product-releases/kong-ai-gateway-2-0-ga\",\"domain\":\"konghq.com\"},{\"title\":\"LiteLLM CVE-2026-59822 added to CISA KEV - Tech Insider\",\"url\":\"https://tech-insider.org/litellm-mcp-vulnerability-cve-2026-59822-cisa-kev-2026/\",\"domain\":\"tech-insider.org\"}]},\"4\":{\"guideSlug\":\"ai-first-development-culture\",\"guideTitle\":\"AI-assisted work is the default path rather than an initiative, and the org advances by removing its next bottleneck rather than by buying tokens\",\"targetItems\":[{\"text\":\"AI-assisted work is the default path rather than an initiative, and the org advances by removing its next bottleneck rather than by buying tokens\",\"guideSlug\":\"ai-first-development-culture\"},{\"text\":\"Agent fleet management as discipline\",\"guideSlug\":\"agent-fleet-management-as-discipline\"},{\"text\":\"Developer supervises agents rather than authoring most changes\",\"guideSlug\":\"developer-agent-supervisor-yegge-stage-6-7\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$4c\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$4d\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$4e\"}],\"gettingStarted\":[\"Start with leadership behavior, not just leadership endorsement.\",\"Update hiring criteria to reflect AI-first expectations.\",\"Build AI tool use into performance and growth conversations.\"],\"links\":[{\"title\":\"Steve Yegge on the Coming AGI Transition\",\"url\":\"https://steve-yegge.medium.com/\",\"domain\":\"steve-yegge.medium.com\"},{\"title\":\"Boris Cherny - Steps of AI Adoption\",\"url\":\"https://x.com/bcherny/status/2077929379661844559\",\"domain\":\"x.com\"},{\"title\":\"How AI Coding Tools Spread Through Organizations - Microsoft (arXiv 2607.01418)\",\"url\":\"https://arxiv.org/abs/2607.01418\",\"domain\":\"arxiv.org\"},{\"title\":\"Employers Who Laid Off Workers for AI Are Reversing Their Decisions - CNBC\",\"url\":\"https://www.cnbc.com/2026/07/01/employers-who-laid-off-workers-for-ai-are-reversing-their-decisions.html\",\"domain\":\"cnbc.com\"},{\"title\":\"The Culture Map - Erin Meyer\",\"url\":\"https://erinmeyer.com/books/the-culture-map/\",\"domain\":\"erinmeyer.com\"},{\"title\":\"Psychological Safety and High-Performance Teams - Google re:Work\",\"url\":\"https://rework.withgoogle.com/print/guides/5721312655835136/\",\"domain\":\"rework.withgoogle.com\"},{\"title\":\"Gartner - AI Agents in Enterprise Applications 2026 Forecast\",\"url\":\"https://www.gartner.com/en/information-technology/insights/artificial-intelligence\",\"domain\":\"gartner.com\"}]},\"5\":{\"guideSlug\":\"kubernetes-for-agents-centralized-orchestration\",\"guideTitle\":\"Centralized agent orchestration: scheduling, placement and lifecycle handled by a platform, not per team\",\"targetItems\":[{\"text\":\"Centralized agent orchestration: scheduling, placement and lifecycle handled by a platform, not per team\",\"guideSlug\":\"kubernetes-for-agents-centralized-orchestration\"},{\"text\":\"Human-at-the-wheel, not human-in-the-loop\",\"guideSlug\":\"human-at-the-wheel-not-human-in-the-loop\"},{\"text\":\"Organization optimized for agent throughput, not human throughput\",\"guideSlug\":\"organization-optimized-for-agent-throughput-not-human-throug\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$4f\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah's measurement systems are hitting the limits of L4 fleet management. She can see aggregate cost and total agent usage but cannot answer the questions leadership is now asking: which task types have the best cost-to-value ratio, which teams are using agents most effectively per dollar spent, and what would be the impact of shifting 20% of agent workload from frontier models to smaller models.\\n\\n**What Sarah should do:** Sarah should work with Victor to define the measurement ar
102chitecture for the orchestration layer before it is built. The questions she needs to answer - cost per task type, value per dollar of inference spend, model routing effectiveness - are architectural requirements for the orchestration system's observability layer. Sarah should write these requirements explicitly and ensure they are included in the orchestration system design. Retrofitting observability into an orchestration system after it's built is significantly harder than designing for it from the start.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$50\"}],\"gettingStarted\":[\"Map your current agent workload before designing the orchestration\",\"Start with task queuing and cost tracking\",\"Implement model routing with a clear policy\"],\"links\":[{\"title\":\"Kubernetes Documentation - Concepts\",\"url\":\"https://kubernetes.io/docs/concepts/\",\"domain\":\"kubernetes.io\"},{\"title\":\"LangGraph - Multi-Agent Orchestration\",\"url\":\"https://langchain-ai.github.io/langgraph/\",\"domain\":\"langchain-ai.github.io\"},{\"title\":\"Anthropic - Building Effective Agents\",\"url\":\"https://www.anthropic.com/research/building-effective-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"The Gas Town Pattern - Agent Infrastructure at Scale\",\"url\":\"https://docs.anthropic.com/en/docs/agents-and-agentic-frameworks\",\"domain\":\"docs.anthropic.com\"}]}},\"Knowledge Management\":{\"1\":{\"guideSlug\":\"folk-tradition-run-x-means-run-x-after-y-and-z\",\"guideTitle\":\"Processes passed on verbally, not documented\",\"targetItems\":[{\"text\":\"Processes passed on verbally, not documented\",\"guideSlug\":\"folk-tradition-run-x-means-run-x-after-y-and-z\"},{\"text\":\"Little written documentation\",\"guideSlug\":\"nobody-wrote-docs-because-nobody-read-them\"},{\"text\":\"Key knowledge held by senior staff\",\"guideSlug\":\"tribal-knowledge-in-seniors-heads\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$51\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$52\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$53\"}],\"gettingStarted\":[\"Run a \\\"traps audit\\\"\",\"Instrument failures\",\"Pair on onboarding to surface traps\"],\"links\":[{\"title\":\"The Documentation System - Divio\",\"url\":\"https://documentation.divio.com/\",\"domain\":\"documentation.divio.com\"},{\"title\":\"How to Write Good Documentation - Google Developer Documentation Style Guide\",\"url\":\"https://developers.google.com/style\",\"domain\":\"developers.google.com\"},{\"title\":\"What is a Runbook - PagerDuty\",\"url\":\"https://www.pagerduty.com/resources/automation/learn/what-is-a-runbook/\",\"domain\":\"pagerduty.com\"},{\"title\":\"The Knowledge Curse - Why Experts Are Bad Teachers\",\"url\":\"https://hbr.org/2006/12/the-curse-of-knowledge\",\"domain\":\"hbr.org\"},{\"title\":\"CLAUDE.md and Agent Context Files - Anthropic\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"}]},\"2\":{\"guideSlug\":\"docs-refresh-initiative\",\"guideTitle\":\"Docs refresh initiative\",\"targetItems\":[{\"text\":\"Docs refresh initiative\",\"guideSlug\":\"docs-refresh-initiative\"},{\"text\":\"Architecture Decision Records (ADRs)\",\"guideSlug\":\"architecture-decision-records-adrs\"},{\"text\":\"Written onboarding paths\",\"guideSlug\":\"written-onboarding-paths\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$54\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$55\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$56\"}],\"gettingStarted\":[\"Scope the refresh explicitly\",\"Assign ownership, not just tasks\",\"Verify by following, not by reading\"],\"links\":[{\"title\":\"Docs as Code - Anne Gentle\",\"url\":\"https://www.docslikecode.com/\",\"domain\":\"docslikecode.com\"},{\"title\":\"Documentation Debt - Write the Docs\",\"url\":\"https://www.writethedocs.org/guide/writing/docs-principles/\",\"domain\":\"writethedocs.org\"},{\"title\":\"Technical Writing Best Practices - Google\",\"url\":\"https://developers.google.com/tech-writing\",\"domain\":\"developers.google.com\"},{\"title\":\"The Half-Life of Documentation - Increment Magazine\",\"url\":\"https://increment.com/documentation/\",\"domain\":\"increment.com\"},{\"title\":\"Engineering Documentation Culture - Stripe\",\"url\":\"https://stripe.com/blog/payment-api-design\",\"domain\":\"stripe.com\"}]}
102,\"3\":{\"guideSlug\":\"documentation-infrastructure-not-an-hr-problem\",\"guideTitle\":\"Documentation = infrastructure (not an HR problem)\",\"targetItems\":[{\"text\":\"Documentation = infrastructure (not an HR problem)\",\"guideSlug\":\"documentation-infrastructure-not-an-hr-problem\"},{\"text\":\"Lint rules \u003e docs (enforced \u003e suggested)\",\"guideSlug\":\"lint-rules-docs-enforced-suggested\"},{\"text\":\"A queryable map of the codebase (structure, ownership, change history); agentic search and plain-text memory over vector databases\",\"guideSlug\":\"knowledge-graph-codebase-codetale-graph-buddy\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$57\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$58\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$59\"}],\"gettingStarted\":[\"Audit documentation as you would audit infrastructure\",\"Add documentation coverage to your engineering metrics\",\"Build CI checks for documentation requirements\"],\"links\":[{\"title\":\"Documentation as Code - Write the Docs\",\"url\":\"https://www.writethedocs.org/guide/docs-as-code/\",\"domain\":\"writethedocs.org\"},{\"title\":\"The Documentation System - Divio\",\"url\":\"https://documentation.divio.com/\",\"domain\":\"documentation.divio.com\"},{\"title\":\"Docs for Developers - Alyssa Rock \u0026 Jared Bhatti\",\"url\":\"https://docsfordevelopers.com/\",\"domain\":\"docsfordevelopers.com\"},{\"title\":\"Measuring Documentation Quality - Google Developer Documentation\",\"url\":\"https://developers.google.com/style/highlights\",\"domain\":\"developers.google.com\"},{\"title\":\"CLAUDE.md and Agent Context Files - Anthropic\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"}]},\"4\":{\"guideSlug\":\"context-fabric-mcp-servers-feed-agents-automatically\",\"guideTitle\":\"Context Fabric: MCP servers feed agents automatically, discoverable and assessable from published metadata before a client ever connects\",\"targetItems\":[{\"text\":\"Context Fabric: MCP servers feed agents automatically, discoverable and assessable from published metadata before a client ever connects\",\"guideSlug\":\"context-fabric-mcp-servers-feed-agents-automatically\"},{\"text\":\"Skills and MCP servers packaged once and installed across clients as vendor-neutral plugins, then treated as maintained assets with a review cadence and an eviction rule - installing a useful skill and keeping it forever are separate decisions\",\"guideSlug\":\"autonomous-requirements-unclear-ticket-spec-auto\"},{\"text\":\"Docs auto-updated by agents on code change\",\"guideSlug\":\"docs-auto-updated-by-agents-on-code-change\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$5a\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been watching engineers spend 15-30 minutes assembling context before each significant agent interaction - pulling in relevant documentation, finding the right ADRs, looking up ownership, checking current monitoring. This overhead makes AI assistance less attractive than it should be and creates inconsistency based on how diligent each engineer is about context preparation.\\n\\nThe Context Fabric eliminates most of this overhead. Sarah should calculate the time saved per engineer per week when context assembly is automated: if 20 engineers each save 20 minutes per day of manual context assembly, that is 400 minutes - nearly 7 hours
102- of engineering time freed daily. She should present this calculation to Bob as the business case for Context Fabric investment. She should also track engineer satisfaction with AI tools before and after Fabric deployment as a qualitative complement to the time-savings calculation.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$5b\"}],\"gettingStarted\":[\"Inventory your context sources\",\"Deploy your first high-value MCP server\",\"Configure agents to connect to deployed MCP servers\"],\"links\":[{\"title\":\"Model Context Protocol - Anthropic\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"MCP Specification - modelcontextprotocol.io\",\"url\":\"https://modelcontextprotocol.io/\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"Building MCP Servers - Anthropic Documentation\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp#adding-mcp-servers\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"CLAUDE.md and Agent Context Files - Anthropic\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Context Engineering for AI Agents - Anthropic\",\"url\":\"https://www.anthropic.com/research\",\"domain\":\"anthropic.com\"},{\"title\":\"The MCP Roadmap - Model Context Protocol Blog\",\"url\":\"https://blog.modelcontextprotocol.io/posts/mcp-roadmap/\",\"domain\":\"blog.modelcontextprotocol.io\"},{\"title\":\"Making Data Ready for Agentic AI - Sadalage and Chandrasekaran\",\"url\":\"https://martinfowler.com/articles/making-data-ready-for-agentic-ai.html\",\"domain\":\"martinfowler.com\"}]},\"5\":{\"guideSlug\":\"self-evolving-knowledge-base\",\"guideTitle\":\"Self-evolving knowledge base\",\"targetItems\":[{\"text\":\"Self-evolving knowledge base\",\"guideSlug\":\"self-evolving-knowledge-base\"},{\"text\":\"The written record is kept current by agents: drift is detected, corrected and validated\",\"guideSlug\":\"agent-detects-stale-context-updates-validates\"},{\"text\":\"Organizational memory = Git-backed, agent-readable, always current\",\"guideSlug\":\"organizational-memory-git-backed-agent-readable-always-curre\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$5c\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$5d\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$5e\"}],\"gettingStarted\":[\"Validate each component individually before integrating\",\"Define the human review surface area\",\"Build staleness detection as a first priority\"],\"links\":[{\"title\":\"CLAUDE.md and Agent Context Files - Anthropic\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/memory\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Model Context Protocol - Anthropic\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Continuous Documentation - Write the Docs\",\"url\":\"https://www.writethedocs.org/guide/docs-as-code/\",\"domain\":\"writethedocs.org\"},{\"title\":\"Building Evolutionary Architectures - O'Reilly\",\"url\":\"https://evolutionaryarchitecture.com/\",\"domain\":\"evolutionaryarchitecture.com\"},{\"title\":\"Knowledge Management Systems - Davenport \u0026 Prusak\",\"url\":\"https://store.hbr.org/product/working-knowledge-how-organizations-manage-what-they-know/3014\",\"domain\":\"store.hbr.org\"}]}},\"Team Structure \u0026 Roles\":{\"1\":{\"guideSlug\":\"traditional-roles-dev-qa-pm\",\"guideTitle\":\"Traditional roles: dev, QA, PM\",\"targetItems\":[{\"text\":\"Traditional roles: dev, QA, PM\",\"guideSlug\":\"traditional-roles-dev-qa-pm\"},{\"text\":\"Seniors review and fix AI-generated code; human-skill preservation (reject code you can't understand even if it works)\",\"guideSlug\":\"senior-debugs-ai-code\"},{\"text\":\"AI being evaluated in the team's stack\",\"guideSlug\":\"ai-doesn-t-work-in-our-environment\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$5f\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$60\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor is one of the eight developers on Bob's team. He has been using GitHub Copilot and Claude for six months and has a clear sense of what works and what doesn't. He's seen his individual productivity increase but also feels frustrated that the QA queue and vague PM specs limit how much of that productivity actually reaches production.\\n\\n**What Victor should do:** Victor should make the system bottleneck visible. He should track his own cycle time carefully: how long does he spend implementing? How long does the PR sit waiting for QA? How many clarification rounds does he have with PM per ticket? This data, presented to Bob, makes the case for expanding AI adoption beyond just developer tools. It also frames Victor as a thoughtful systems thinker rather than a developer who just wants faster tools. Victor is positioned to become the AI champion (L2) precisely because he can see the full pipeline, not just his own slice of it.\"}],\"gettingStarted\":[\"Document your current role structure explicitly\",\"Measure per-role throughput\",\"Identify the first pressure point\"],\"links\":[{\"title\":\"The Evolution of Softw
102are Engineering Roles in the Age of AI - Pragmatic Engineer\",\"url\":\"https://newsletter.pragmaticengineer.com/p/the-impact-of-ai-on-software-engineers-2026\",\"domain\":\"newsletter.pragmaticengineer.com\"},{\"title\":\"Software Engineering at Google - Team Structure Chapters\",\"url\":\"https://abseil.io/resources/swe-book/html/ch05.html\",\"domain\":\"abseil.io\"},{\"title\":\"Accelerate: The Science of Lean Software and DevOps - DORA Metrics\",\"url\":\"https://dora.dev/research/\",\"domain\":\"dora.dev\"},{\"title\":\"Staff Engineer: Leadership Beyond the Management Track\",\"url\":\"https://staffeng.com/book\",\"domain\":\"staffeng.com\"}]},\"2\":{\"guideSlug\":\"ai-champion-per-team\",\"guideTitle\":\"A named AI champion per team, with time actually allocated to the role\",\"targetItems\":[{\"text\":\"A named AI champion per team, with time actually allocated to the role\",\"guideSlug\":\"ai-champion-per-team\"},{\"text\":\"Context engineer role (initial)\",\"guideSlug\":\"context-engineer-role-initial\"},{\"text\":\"Training: how to instruct confidently and then verify confidently, which is not the same as reading every line\",\"guideSlug\":\"training-how-to-write-good-prompts-tasks\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob has four teams. Each has at least one developer who uses AI tools more than the others. But there's no coordination between these informal early adopters, no sharing of what works, and no clear time allocation. Bob's VP is asking for an AI adoption update and Bob isn't sure what to report.\\n\\n**What Bob should do:** Bob should formalize the existing informal structure. He should identify the one developer per team who is already doing champion-like work, have a direct conversation with each of them about formalizing the role (including time protection and a community with peers), and set up a monthly champion sync. The first champion sync should have one goal: each champion shares the top three AI practices that have worked best for their team. This immediately creates cross-team learning that wasn't happening before. Bob can then report to his VP: \\\"We have identified team-level AI champions and are building a structured knowledge-sharing program.\\\" That's honest, credible, and actionable.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$61\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$62\"}],\"gettingStarted\":[\"Identify the organic champion\",\"Allocate dedicated time\",\"Create a champion community\"],\"links\":[{\"title\":\"Anthropic - Building Effective AI Champions in Engineering Teams\",\"url\":\"https://www.anthropic.com/engineering/claude-code-best-practices\",\"domain\":\"anthropic.com\"},{\"title\":\"Developer Experience as Organizational Capability - Gartner\",\"url\":\"https://www.gartner.com/en/articles/how-to-build-developer-experience-for-ai\",\"domain\":\"gartner.com\"},{\"title\":\"The Staff Engineer's Path - Champions and Influence\",\"url\":\"https://staffeng.com/guides/being-visible\",\"domain\":\"staffeng.com\"},{\"title\":\"Communities of Practice in Software Teams - Martin Fowler\",\"url\":\"https://martinfowler.com/bliki/ActivityOriented.html\",\"domain\":\"martinfowler.com\"}]},\"3\":{\"guideSlug\":\"review-shifts-from-writing-to-evaluating-code\",\"guideTitle\":\"Review shifts up the lifecycle: judgment relocates rather than disappears - problem selection, architecture, the quality bar, which signals to trust, and shipping authority stay human even when authorship does not\",\"targetItems\":[{\"text\":\"Review shifts up the lifecycle: judgment relocates rather than disappears - problem selection, architecture, the quality bar, which signals to trust, and shipping authority stay human even when authorship does not\",\"guideSlug\":\"review-shifts-from-writing-to-evaluating-code\"},{\"text\":\"Platform Engineer (AI tooling); Harness Engineer as the consolidated named skill - the harness, not the model, is the asset that survives a vendor swap\",\"guideSlug\":\"platform-engineer-ai-tooling\"},{\"text\":\"Context Engineer = full role (now mainstream - dedicated job postings across industry)\",\"guideSlug\":\"context-engineer-full-role\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$63\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$64\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$65\"}],\"gettingStarted\":[\"Adopt an intent-first review protocol\",\"Create an AI code review checklist\",\"Review the diff against the spec, not against the previous code\"],\"links\":[{\"title\":\"Own the Outer Loop - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/own-the-outer-loop\",\"domain\":\"addyo.substack.com\"},{\"title\":\"Code Review Best Practices for AI-Generated Code - Google Engineering\",\"url\":\"https://google.github.io/eng-practices/review/\",\"domain\":\"google.github.io\"},{\"title\":\"How to Review AI-Generated Pull Requests - GitHub Blog\",\"url\":\"https://github.blog/ai-and-ml/generative-ai/agent-pull-requests-are-everywhere-heres-how-to-review-them/\",\"domain\":\"github.blog\"},{\"title\":\"Intent-Based Code Review - Thoughtworks Technology Radar\",\"url\":\"https://www.thoughtworks.com/radar/techniques/code-reviews\",\"domain\":\"thoughtworks.com\"},{\"title\":\"Effective Code Review at Scale - Netflix Tech Blog\",\"url\":\"https://netflixtechblog.com/\",\"domain\":\"netflixtechblog.com\"},{\"title\":\"Human Judgment Doesn't Leave the Software Factory. It Relocates. - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/human-judgment-doesnt-leave-the-software\",\"domain\":\"addyo.substack.com\"},{\"title\":\"Agentic Code Quality - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/agentic-code-quality\",\"domain\":\"addyo.substack.com\"}
102,{\"title\":\"Building a Software Factory for AI SDK - Vercel\",\"url\":\"https://vercel.com/blog/building-a-software-factory-for-ai-sdk\",\"domain\":\"vercel.com\"},{\"title\":\"Software Factory 2026 AI Benchmarks - LinearB\",\"url\":\"https://linearb.io/blog/software-factory-2026-ai-benchmarks-code-review-roi\",\"domain\":\"linearb.io\"},{\"title\":\"Turn One Giant AI-Generated Pull Request Into a Reviewable Stack - GitHub\",\"url\":\"https://github.blog/engineering/turn-one-giant-ai-generated-pull-request-to-a-reviewable-stack/\",\"domain\":\"github.blog\"},{\"title\":\"More Than Just Code Review - Simon Willison\",\"url\":\"https://simonwillison.net/2026/Aug/22/more-than-just-code-review/\",\"domain\":\"simonwillison.net\"}]},\"4\":{\"guideSlug\":\"span-of-control-how-many-agents-you-can-effectively-supervis\",\"guideTitle\":\"Span of control = how many agents you can effectively supervise; the binding limit is the orchestrator's context, not tokens - batches capped at 2-4, status polling restricted, overlapping file ownership read as a signal to consolidate\",\"targetItems\":[{\"text\":\"Span of control = how many agents you can effectively supervise; the binding limit is the orchestrator's context, not tokens - batches capped at 2-4, status polling restricted, overlapping file ownership read as a signal to consolidate\",\"guideSlug\":\"span-of-control-how-many-agents-you-can-effectively-supervis\"},{\"text\":\"Developer = manager of agent fleet (now a product default: Cursor Run Mode, Claude agent view, Antigravity, MultiDevin)\",\"guideSlug\":\"developer-manager-of-agent-fleet\"},{\"text\":\"Each running agent has a visible health state, so a stalled or drifting one is noticed without being hunted for\",\"guideSlug\":\"keep-your-tamagotchi-alive-yegge\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$66\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is designing the organizational metrics dashboard for AI adoption. She wants to include a \\\"productivity\\\" metric but is struggling to define what to measure. She's been asked to demonstrate that the AI tooling investment is delivering value, and stakeholder expectations are high.\\n\\n**What Sarah should do:** Sarah should build the metrics around effective span of control because it captures the right thing: how effectively are developers using the AI capacity available to them? She should track, per developer per sprint: number of parallel agents run, number of tasks completed by agents versus directly, and review-to-agent ratio (how many PRs did the developer review versus write). The trend over time should show increasing effective span as developers develop fleet management skills, which translates to increasing throughput per developer. This metric is honest - it measures actual agent utilization, not just tool installation - and it provides actionable data for improving AI practices.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$67\"}],\"gettingStarted\":[\"Establish your current effective span\",\"Identify your span-limiting factor\",\"Address logistics before cognitive load\"],\"links\":[{\"title\":\"Span of Control in Management - Harvard Business Review\",\"url\":\"https://hbr.org/2012/04/how-many-direct-reports\",\"domain\":\"hbr.org\"},{\"title\":\"Cognitive Load Theory and Team Management - Research Overview\",\"url\":\"https://www.nngroup.com/articles/minimize-cognitive-load/\",\"domain\":\"nngroup.com\"},{\"title\":\"Claude Code Parallel Agent Workflows - Anthropic\",\"url\":\"https://code.claude.com/docs/en/common-workflows\",\"domain\":\"code.claude.com\"},{\"title\":\"Managing AI Agent Fleets at Scale - Anthropic Engineering\",\"url\":\"https://www.anthropic.com/engineering/claude-code-best-practices\",\"domain\":\"anthropic.com\"},{\"title\":\"The Orchestrator's Tax - Rahul Garg\",\"url\":\"https://martinfowler.com/articles/orchestrator-tax.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"Practical Loop Engineering - Addy Osmani\",\"url\":\"https://addyo.substack.com/p/practical-loop-engineering\",\"domain\":\"addyo.substack.com\"},{\"title\":\"The Conductor Developer - Rachel Laycock\",\"url\":\"https://martinfowler.com/rachels-ramblings/conductor-developer.html\",\"domain\":\"martinfowler.com\"}]},\"5\":{\"guideSlug\":\"agentic-engineer-orchestration-supervision-architecture\",\"guideTitle\":\"Agentic Engineer: orchestration + supervision + architecture\",\"targetItems\":[{\"text\":\"Agentic Engineer: orchestration + supervision + architecture\",\"guideSlug\":\"agentic-engineer-orchestration-supervision-architecture\"},{\"text\":\"PEV loop: Plan â Execute â Verify\",\"guideSlug\":\"pev-loop-plan-execute-verify\"},{\"text\":\"Non-coder contributors via agent interfaces\",\"guideSlug\":\"non-coder-contributors-via-agent-interfaces\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$68\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$69\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$6a\"}],\"gettingStarted\":[\"Master the full L4 skill set first\",\"Study orchestration systems design\",\"Design a supervision architecture for your organization\"],\"links\":[{\"title\":\"Anthropic Engineering Blog - Building Agentic Systems\",\"url\":\"https://www.anthropic.com/engineering\",\"domain\":\"anthropic.com\"},{\"title\":\"Multi-Agent Orchestration Patterns - Anthropic Documentation\",\"url\":\"https://docs.anthropic.com/en/docs/build-with-claude/agents\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Temporal Workflow Engine - Durable Agent Orchestration\",\"url\":\"https://temporal.io/\",\"domain\":\"temporal.io\"},{\"title\":\"The Rise of the AI Engineer - Swyx\",\"url\":\"https://www.latent.space/p/ai-engineer\",\"domain\":\"latent.space\"},{\"title\":\"LLM Agent Architecture Patterns - LangChain Documentation\",\"url\":\"https://python.langchain.com/docs/concepts/agents/\",\"domain\":\"python.langchain.com\"}]}}
102,\"Tech Debt \u0026 Modernization\":{\"1\":{\"guideSlug\":\"debt-grows\",\"guideTitle\":\"Tech debt accumulates\",\"targetItems\":[{\"text\":\"Tech debt accumulates\",\"guideSlug\":\"debt-grows\"},{\"text\":\"Legacy code left untouched\",\"guideSlug\":\"legacy-don-t-touch-it\"},{\"text\":\"Multi-year migration backlog\",\"guideSlug\":\"migration-backlog-years\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob manages a team that is consistently slower than stakeholders expect, and he suspects the codebase is a significant factor. Developers complain about certain modules being \\\"impossible to change,\\\" but there is no formal record of what the debt is or what it costs. Bob's instinct is that fixing this requires a dedicated cleanup sprint, but every time he proposes one, feature priorities push it out.\\n\\nBob needs to reframe debt as infrastructure, not housekeeping. The conversation with stakeholders should be: \\\"Our development speed is capped by the state of our codebase. Here is a specific area where we are 2x slower than we should be, and here is what it would take to fix it.\\\" That framing makes debt reduction a velocity investment, not a distraction from delivery. Bob should also establish the 10% sprint capacity rule as a team norm, so that debt reduction happens continuously rather than competing for a dedicated sprint that never arrives.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah is trying to measure developer productivity and keeps running into the same problem: the numbers vary wildly depending on which part of the codebase a developer is working in. A developer working in the clean microservices ships features quickly; the same developer working in the legacy monolith moves at half the speed. The variance makes aggregate productivity metrics nearly meaningless.\\n\\nSarah should use this variance as a debt measurement tool. The difference in velocity between clean and debt-laden areas of the codebase is a direct measure of what the debt is costing the organization. Sarah should instrument this: track PR cycle time and story point completion rate by code area, then map the slow areas to known debt. This gives both a debt priority list (fix the slowest areas first) and a business case (fixing this area would recover X% of lost velocity). Debt reduction tracked this way becomes a productivity initiative with measurable ROI, not a vague engineering preference.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$6b\"}],\"gettingStarted\":[\"Name the problem explicitly\",\"Do a one-hour debt audit\",\"Attach a cost estimate to one item\"],\"links\":[{\"title\":\"Martin Fowler - Technical Debt\",\"url\":\"https://martinfowler.com/bliki/TechnicalDebt.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"Ward Cunningham - Debt Metaphor (original)\",\"url\":\"https://wiki.c2.com/?WardExplainsDebtMetaphor\",\"domain\":\"wiki.c2.com\"},{\"title\":\"The Hidden Cost of Technical Debt - Stripe Developer Survey\",\"url\":\"https://stripe.com/files/reports/the-developer-coefficient.pdf\",\"domain\":\"stripe.com\"},{\"title\":\"Accelerate - DORA Research on Software Delivery Performance\",\"url\":\"https://dora.dev/research/\",\"domain\":\"dora.dev\"}]},\"2\":{\"guideSlug\":\"debt-categorized-and-prioritized\",\"guideTitle\":\"Debt categorized and prioritized; separate Disposable Software (deliberate throwaway) from durable systems - and track comprehension debt, the debt that accumulates silently where machines verify machines\",\"targetItems\":[{\"text\":\"Debt categorized and prioritized; separate Disposable Software (deliberate throwaway) from durable systems - and track comprehension debt, the debt that accumulates silently where machines verify machines\",\"guideSlug\":\"debt-categorized-and-prioritized\"},{\"text\":\"Manual migration attempts\",\"guideSlug\":\"manual-migration-attempts\"},{\"text\":\"OpenRewrite basic recipes\",\"guideSlug\":\"openrewrite-basic-recipes\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$6c\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been measuring velocity by area of the codebase and has the data that shows where debt is costing developer time. What she has been missing is the bridge between her productivity data and the engineering team's debt work - the debt inventory is that bridge. If the debt inventory includes the code areas that map to Sarah's slow-velocity zones, she can validate that the priority order makes sense (high-impact debt areas should be prioritized) and measure the productivity improvement when those items are addressed.\\n\\nSarah should request access to the debt inventory and add a column for \\\"estimated velocity impact\\\" - her data contribution to the scoring rubric. This makes the inventory more accurate and gives Sarah a direct line of sight from debt remediation work to productivity outcomes. When the authentication module debt is addressed and PR cycle time in that area drops by 30%, that is a measurable outcome that Sarah can report and that justifies the debt reduction investment.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has been the informal mental debt registry for years. He knows every item that should be on the list. The transition to a formal inventory is partly about mov
102ing his knowledge into a shared artifact and partly about adding rigor to the prioritization that has previously been his judgment call.\\n\\nVictor should lead the initial inventory sprint and be the named owner of the inventory going forward. He should also use AI tools at this stage to accelerate categorization: given a codebase, an agent can scan for common debt indicators - outdated dependencies, missing test coverage, deprecated API usage, high cyclomatic complexity - and produce a candidate debt list that Victor reviews and refines. This is not full AI-assisted remediation; it is AI-assisted discovery, which is appropriate and effective at L2. The categorized inventory Victor produces is also the input that L3 AI remediation agents will need - building a clean inventory now is an investment in higher maturity levels.\"}],\"gettingStarted\":[\"Establish debt categories\",\"Run an initial inventory sprint\",\"Create a scoring rubric\"],\"links\":[{\"title\":\"A Taxonomy of Technical Debt - Erin Allard, Thoughtworks\",\"url\":\"https://www.thoughtworks.com/insights/blog/technical-debt-taxonomy\",\"domain\":\"thoughtworks.com\"},{\"title\":\"Managing Technical Debt - SEI Carnegie Mellon\",\"url\":\"https://www.sei.cmu.edu/library/managing-technical-debt-in-software-engineering-2/\",\"domain\":\"sei.cmu.edu\"},{\"title\":\"OpenRewrite - Migration and Debt Recipes Catalog\",\"url\":\"https://docs.openrewrite.org/recipes\",\"domain\":\"docs.openrewrite.org\"},{\"title\":\"OWASP Dependency-Check - Security Debt Scanning\",\"url\":\"https://owasp.org/www-project-dependency-check/\",\"domain\":\"owasp.org\"},{\"title\":\"Agentic Technical Debt and the Stochastic Tax - arXiv 2605.29129\",\"url\":\"https://arxiv.org/abs/2605.29129\",\"domain\":\"arxiv.org\"},{\"title\":\"Bun Rewritten in Rust in 11 Days - Bun Blog\",\"url\":\"https://bun.com/blog/bun-in-rust\",\"domain\":\"bun.com\"}]},\"3\":{\"guideSlug\":\"continuous-modernization-agent-pays-off-debt-in-background\",\"guideTitle\":\"Continuous Modernization: agent pays off debt in the background, and the payoff is now measurable in tokens - one refactor cut the input cost of every future change to that code by 83%\",\"targetItems\":[{\"text\":\"Continuous Modernization: agent pays off debt in the background, and the payoff is now measurable in tokens - one refactor cut the input cost of every future change to that code by 83%\",\"guideSlug\":\"continuous-modernization-agent-pays-off-debt-in-background\"},{\"text\":\"Library bumps, version upgrades auto\",\"guideSlug\":\"library-bumps-version-upgrades-auto\"},{\"text\":\"OpenRewrite + agent = systematic refactoring\",\"guideSlug\":\"openrewrite-agent-systematic-refactoring\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$6d\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$6e\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$6f\"}],\"gettingStarted\":[\"Ensure the debt inventory is current and structured\",\"Establish the agent's toolchain\",\"Start with one category of debt\"],\"links\":[{\"title\":\"OpenRewrite - Automated Refactoring at Scale\",\"url\":\"https://docs.openrewrite.org/\",\"domain\":\"docs.openrewrite.org\"},{\"title\":\"Dependabot - Automated Dependency Updates\",\"url\":\"https://docs.github.com/en/code-security/dependabot\",\"domain\":\"docs.github.com\"},{\"title\":\"Renovate Bot - Dependency Update Automation\",\"url\":\"https://docs.renovatebot.com/\",\"domain\":\"docs.renovatebot.com\"},{\"title\":\"Continuous Modernization with Moderne\",\"url\":\"https://www.moderne.ai/blog/enterprise-tech-debt-refactoring-at-scale\",\"domain\":\"moderne.ai\"},{\"title\":\"The Debt Trap - Google SRE Book\",\"url\":\"https://sre.google/sre-book/being-on-call/\",\"domain\":\"sre.google\"},{\"title\":\"The Economic Benefit of Refactoring - Giles Edwards-Alexander\",\"url\":\"https://martinfowler.com/articles/exploring-gen-ai/refactoring-economic-benefit.html\",\"domain\":\"martinfowler.com\"},{\"title\":\"AI-Generated C++ in Production - arXiv 2608.06640\",\"url\":\"https://arxiv.org/abs/2608.06640\",\"domain\":\"arxiv.org\"}]},\"4\":{\"guideSlug\":\"dead-project-too-expensive-to-modernize-agent-modernizes-for\",\"guideTitle\":\"\\\"Dead project too expensive to modernize\\\" â agent modernizes for pennies (sqlite-utils 4.0 for $149; reverse engineering that never penciled out now does)\",\"targetItems\":[{\"text\":\"\\\"Dead project too expensive to modernize\\\" â agent modernizes for pennies (sqlite-utils 4.0 for $149; reverse engineering that never penciled out now does)\",\"guideSlug\":\"dead-project-too-expensive-to-modernize-agent-modernizes-for\"},{\"text\":\"Cross-repo migration agents: the permanently-deferred migration is now a two-week job, so the backlog is a choice rather than a constraint\",\"guideSlug\":\"cross-repo-migration-agents\"},{\"text\":\"Java 8 â 21, Angular.js â Angular 17 via agents; the Bun model for full rewrites (conformance test suite first, then 64 concurrent agents)\",\"guideSlug\":\"java-8-21-angular-js-angular-17-via-agents\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$70\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking these seven internal tools as a compliance risk: they run on software with known CVEs, they are not compliant with the current security standards, and any audit would flag them. She has been asking for modernization budget for two years and has been told it is not economically justified.\\n\\nThe agent-based modernization approach changes Sarah's business case. She no longer needs to justify a six-month engineering project. She needs to justify two weeks of agent work and 20 hours of human review. This is a case where the economic argument was always valid - the risk cost of running CVE-exposed software clearly exceeds the modernization cost - but the modernization cost was too high for the argument to win. At L4 costs, the argument wins easily. Sarah should use this case to establish a policy: any application running on EOL software with known CVEs is automatically authorized for agent-based modernization without requiring a separate business case.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$71\"}],\"gettingStarted\":[\"Inventory the dead project list\",\"Triage by risk\",\"Run a pilot on one project\"],\"links\":[{\"title\":\"OpenRewrite - Legacy Migration Recipes\",\"url\":\"https://docs.openrewrite.org/recipes/java/migrate\",\"domain\":\"docs.openrewrite.org\"},{\"title\":\"OWASP Dependency-Check for EOL Software Scanning\",\"url\":\"https://owasp.org/www-project-dependency-check/\",\"domain\":\"owasp.org\"},{\"title\":\"Characterization Testing for Legacy Code - Michael Feathers\",\"url\":\"https://www.oreilly.com/library/view/working-effectively-with/0131177052/\",\"domain\":\"oreilly.com\"},{\"title\":\"The Cost of Technical Debt - Stepsize\",\"url\":\"https://www.stepsize.com/blog/cost-of-technical-debt\",\"domain\":\"stepsize.com\"}]},\"5\":{\"guideSlug\":\"tech-debt-near-zero-steady-state\",\"guideTitle\":\"Tech debt = near-zero steady state\",\"targetItems\":[{\"text\":\"Tech debt = near-zero steady state\",\"guideSlug\":\"tech-debt-near-zero-steady-state\"},{\"text\":\"Agent fleet maintains, upgrades, patches 24/7\",\"guideSlug\":\"agent-fleet-maintains-upgrades-patches-24-7\"},{\"text\":\"CVE remediation: detect â fix â test â ship autonomous\",\"guideSlug\":\"cve-remediation-detect-fix-test-ship-autonomous\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's organization has been at L4 for 18 months. The migration backlog is cleared. Dependency versions are current. The dead projects have been modernized or retired. The debt inventory, which had 240 items when he started tracking it two years ago, now has 12 items, all recent and most in review. For the first time in his career running this team, the codebase is clean.\\n\\nThe transition to steady-state maintenance requires a shift in how Bob manages the agent fleet. The backlog-clearance agents that ran high-throughput migrations are replaced by maintenance agents that run on a daily cycle, monitoring for new debt creation and addressing it immediately. Bob should define the steady-state policy: what categories of debt, at what SLA, and what escalation criteria. He should also communicate to the team that the old concept of \\\"tech debt reduction initiatives\\\" is retired - debt is now handled continuously by the maintenance system, and developers should not expect to spend sprint time on debt cleanup.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah's metrics look different at L5 than at any previous level. Developer velocity is no longer eroding over time as it was at L1-L2. PR cycle time is stable rather than growing. Incident rate attributable to technical debt has dropped to near zero. Onboarding time for new developers has declined because the codebase is comprehensible and documented.\\n\\nSarah should establish a \\\"codebase quality index\\\" as a standing organizational metric: a composite of debt inventory size, average dependency age, test coverage, and lint error count. At L5, this index should be stable or improving. If it degrades, the maintenance system is not keeping up and needs attention. Publishing the codebase quality index monthly gives the organization visibility into technical health at a glance and makes the investment in agent-based maintenance legible to non-technical stakeholders.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$72\"}],\"gettingStarted\":[\"Measure current debt velocity\",\"Achieve backlog clearance first\",\"Establish debt creation detection\"],\"links\":[{\"title\":\"DORA Metrics - Software Delivery Performance Research\",\"url\":\"https://dora.dev/research/\",\"domain\":\"dora.dev\"},{\"title\":\"OpenRewrite - Maintaining Code Currency\",\"url\":\"https://docs.openrewrite.org/\",\"domain\":\"docs.openrewrite.org\"},{\"title\":\"Continuous Improvement in Software Engineering - IEEE\",\"url\":\"https://ieeexplore.ieee.org/document/9520328\",\"domain\":\"ieeexplore.ieee.org\"},{\"title\":\"The Phoenix Project - Continuous Improvement Model\",\"url\":\"https://itrevolution.com/product/the-phoenix-project/\",\"domain\":\"itrevolution.com\"}]}}}
102,\"infrastructure\":{\"Agent Runtime \u0026 Sandboxing\":{\"1\":{\"guideSlug\":\"agent-in-developer-s-ide\",\"guideTitle\":\"Agent in developer's IDE\",\"targetItems\":[{\"text\":\"Agent in developer's IDE\",\"guideSlug\":\"agent-in-developer-s-ide\"},{\"text\":\"Agent runs in the developer's local environment\",\"guideSlug\":\"no-isolation\"},{\"text\":\"Agent access is coarse-grained (all or none)\",\"guideSlug\":\"agent-has-access-to-everything-or-nothing\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$73\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$74\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$75\"}],\"gettingStarted\":[\"Install a capable IDE agent\",\"Start with read-only tasks\",\"Enable write access incrementally\"],\"links\":[{\"title\":\"Cursor Agent Mode Documentation\",\"url\":\"https://docs.cursor.com/agent\",\"domain\":\"docs.cursor.com\"},{\"title\":\"Claude Code - Running in VS Code\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/overview\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Understanding AI Agent Security Risks\",\"url\":\"https://owasp.org/www-project-top-10-for-large-language-model-applications/\",\"domain\":\"owasp.org\"},{\"title\":\"GitHub Copilot Agent Mode\",\"url\":\"https://docs.github.com/en/copilot/how-tos/chat-with-copilot/chat-in-ide\",\"domain\":\"docs.github.com\"}]},\"2\":{\"guideSlug\":\"dedicated-dev-environments\",\"guideTitle\":\"Dedicated dev environments\",\"targetItems\":[{\"text\":\"Dedicated dev environments\",\"guideSlug\":\"dedicated-dev-environments\"},{\"text\":\"Basic sandboxing (Docker, bubblewrap, eBPF directory confinement), and untrusted repositories opened with auto-run hooks disabled - repo-supplied agent, editor and git config (`core.fsmonitor`, GitSpawn) executes on folder open, before any install step\",\"guideSlug\":\"basic-sandboxing-docker\"},{\"text\":\"Agent credentials scoped per project and short-lived, never a personal PAT or a shared long-lived key; spend caps and baseline alerts on every provider account - the first mass agent-run campaign found Claude, Cursor and Gemini tokens on 5,871 machines and burned $600k of credits through one dashboard\",\"guideSlug\":\"agent-credentials-scoped-per-project\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob wants to move his team from laptop-based agents to a more controlled environment. He has heard about GitHub Codespaces and thinks it might be the right step, but he is not sure how to drive adoption without mandating it and facing resistance. A few developers have already been using Codespaces for their own reasons and seem happy with it.\\n\\n**What Bob should do:** Bob should start with the developers who are already using Codespaces and work with them to build a devcontainer configuration that includes agent tooling. Once that configuration exists, he can point other developers at it as \\\"here is a working setup you can use immediately.\\\" Mandate is the wrong lever here - demonstration is. Bob should also make the cost case: cloud workspace costs for 40 developers are likely $500-2,000 per month, which is negligible compared to the security incident cost the laptop model risks. Frame it as cheap insurance plus productivity improvement.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking developer satisfaction with their tooling and has seen consistent complaints about environment setup time: \\\"I spent two hours getting the new service to run locally.\\\" This setup friction is a direct productivity cost and also a reason developers use agent-less workflows for tasks that would be better done with agents. Cloud dev environments could fix the setup problem while also enabling safer agent use.\\n\\n**What Sarah should do:** Sarah should measure the time developers spend on environment setup and configuration. Even a rough estimate (30-minute survey across the team) will reveal the total cost. Then she should run a 30-day pilot with Codespaces or Gitpod for a subset of the team and measure whether setup time decreases. The productivity data from the pilot - reduced environment setup friction plus safer agent use - makes the case for org-wide adoption. Sarah should frame this as a developer experience improvement rather than an infrastru
102cture security project, because it is both.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$76\"}],\"gettingStarted\":[\"Add a devcontainer configuration\",\"Provision workspace-scoped credentials\",\"Configure the agent within the workspace\"],\"links\":[{\"title\":\"GitHub Codespaces Documentation\",\"url\":\"https://docs.github.com/en/codespaces\",\"domain\":\"docs.github.com\"},{\"title\":\"Dev Containers Specification\",\"url\":\"https://containers.dev/\",\"domain\":\"containers.dev\"},{\"title\":\"Gitpod Documentation\",\"url\":\"https://www.gitpod.io/docs\",\"domain\":\"gitpod.io\"},{\"title\":\"Coder - Self-Hosted Dev Environments\",\"url\":\"https://coder.com/docs\",\"domain\":\"coder.com\"},{\"title\":\"Dev Container Features - Pre-built Tooling\",\"url\":\"https://containers.dev/features\",\"domain\":\"containers.dev\"}]},\"3\":{\"guideSlug\":\"isolated-agent-environments-devbox-model\",\"guideTitle\":\"Isolated agent environments (devbox model); credentials injected at run time from a vault or broker, never in prompts, files or eval sandboxes - an injected evaluation sandbox handed over production keys for several providers\",\"targetItems\":[{\"text\":\"Isolated agent environments (devbox model); credentials injected at run time from a vault or broker, never in prompts, files or eval sandboxes - an injected evaluation sandbox handed over production keys for several providers\",\"guideSlug\":\"isolated-agent-environments-devbox-model\"},{\"text\":\"Pre-warmed containers with codebase\",\"guideSlug\":\"pre-warmed-containers-with-codebase\"},{\"text\":\"Network isolation with egress denied by default and destinations allowlisted; the evaluation and test environment is inside the security boundary, not outside it; audit everything the agent WRITES, because escapes work by planting files that trusted host tools later read\",\"guideSlug\":\"network-isolation-agent-can-t-see-production\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$77\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has noticed that developers are serializing agent tasks that could be parallelized because they are worried about conflicts in shared environments. The theoretical throughput of parallel execution is not being realized because the infrastructure does not safely support it. Developers who could be running 3-5 parallel agents are running 1-2 out of caution.\\n\\n**What Sarah should do:** Sarah should quantify the parallelism gap. If developers are running agents sequentially out of conflict avoidance, how many hours per week are being lost to that serialization? The calculation is not hard: estimate tasks per day, average task duration, and the fraction that could have run in parallel. Even conservative assumptions will show a significant lost-throughput number that justifies the devbox infrastructure investment. Sarah should present this calculation alongside the devbox proposal: \\\"here is what parallel execution would give us, here is what it costs to build the isolation that makes it safe.\\\"\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$78\"}],\"gettingStarted\":[\"Define the devbox specification\",\"Build the base image\",\"Implement task-scoped credential injection\"],\"links\":[{\"title\":\"Firecracker - Secure MicroVM Technology\",\"url\":\"https://firecracker-microvm.github.io/\",\"domain\":\"firecracker-microvm.github.io\"},{\"title\":\"Docker Containers vs. VMs - Security Tradeoffs\",\"url\":\"https://docs.docker.com/guides/docker-overview/\",\"domain\":\"docs.docker.com\"},{\"title\":\"HashiCorp Vault - Dynamic Credentials\",\"url\":\"https://developer.hashicorp.com/vault/docs/secrets/databases\",\"domain\":\"developer.hashicorp.com\"},{\"title\":\"Stripe Engineering - Building AI Agents at Scale\",\"url\":\"https://stripe.com/blog/engineering\",\"domain\":\"stripe.com\"},{\"title\":\"Anthropic Threat Intelligence Report, September 2026\",\"url\":\"https://www.anthropic.com/threat-intelligence-report-september-2026\",\"domain\":\"anthropic.com\"}]},\"4\":{\"guideSlug\":\"ephemeral-devboxes-10s-spin-up-stripe-benchmark\",\"guideTitle\":\"Ephemeral devboxes spin up fast enough that the agent never waits on the environment\",\"targetItems\":[{\"text\":\"Ephemeral devboxes spin up fast enough that the agent never waits on the environment\",\"guideSlug\":\"ephemeral-devboxes-10s-spin-up-stripe-benchmark\"},{\"text\":\"Pre-loaded services, code, MCP tools\",\"guideSlug\":\"pre-loaded-services-code
102-mcp-tools\"},{\"text\":\"MicroVM, hardware-isolated execution as default (kubernetes-sigs agent-sandbox standard, AWS Lambda MicroVMs, Docker Cloud Sandboxes, Microsoft MXC); assume escape, including over DNS: workload identity per agent (SPIFFE, WIF, Google Agent Identity) with task-scoped tokens that expire in minutes + cryptographic run provenance\",\"guideSlug\":\"kernel-level-policy-enforcement\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$79\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$7a\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$7b\"}],\"gettingStarted\":[\"Baseline your current startup time\",\"Implement Firecracker for microVM-speed isolation\",\"Switch to snapshot-based initialization\"],\"links\":[{\"title\":\"Stripe Minions - Agent Infrastructure Blog Post\",\"url\":\"https://stripe.dev/blog/minions-stripes-one-shot-end-to-end-coding-agents\",\"domain\":\"stripe.dev\"},{\"title\":\"Firecracker MicroVM - Fast Snapshot and Restore\",\"url\":\"https://github.com/firecracker-microvm/firecracker/blob/main/docs/snapshotting/snapshot-support.md\",\"domain\":\"github.com\"},{\"title\":\"Firecracker Design - Why Millisecond Boot Times\",\"url\":\"https://firecracker-microvm.github.io/\",\"domain\":\"firecracker-microvm.github.io\"},{\"title\":\"CRIU - Checkpoint/Restore In Userspace\",\"url\":\"https://criu.org/Main_Page\",\"domain\":\"criu.org\"},{\"title\":\"BuildKit - Efficient Docker Build Caching\",\"url\":\"https://docs.docker.com/build/buildkit/\",\"domain\":\"docs.docker.com\"}]},\"5\":{\"guideSlug\":\"agent-fleet-on-dedicated-compute\",\"guideTitle\":\"Agent fleet on dedicated compute, on a harness whose parts are replaceable - model adapter, tool registry, session log and the agent loop itself swappable without a rewrite\",\"targetItems\":[{\"text\":\"Agent fleet on dedicated compute, on a harness whose parts are replaceable - model adapter, tool registry, session log and the agent loop itself swappable without a rewrite\",\"guideSlug\":\"agent-fleet-on-dedicated-compute\"},{\"text\":\"Agent execution environments scale with demand, independently of CI runner capacity\",\"guideSlug\":\"auto-scaling-agents-scale-with-load\"},{\"text\":\"Each agent = isolated machine or managed-agent-on-your-hardware (Devin Outposts model); sender-constrained tokens (DPoP, mTLS) and every session of one agent revocable in seconds\",\"guideSlug\":\"each-agent-isolated-machine-cursor-approach-or-shared-with-s\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$7c\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$7d\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$7e\"}],\"gettingStarted\":[\"Size the initial fleet based on observed agent load\",\"Create a Kubernetes namespace for agent workloads\",\"Deploy a dedicated node pool for agents\"],\"links\":[{\"title\":\"Kubernetes Node Pools and Taints\",\"url\":\"https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/\",\"domain\":\"kubernetes.io\"},{\"title\":\"AWS EC2 Storage Optimized Instances (i3, i4)\",\"url\":\"https://aws.amazon.com/ec2/instance-types/#Storage_Optimized\",\"domain\":\"aws.amazon.com\"},{\"title\":\"Cursor Engineering - Disk I/O at Agent Scale\",\"url\":\"https://cursor.com/blog/scaling-agents\",\"domain\":\"cursor.com\"},{\"title\":\"Kubernetes Resource Management\",\"url\":\"https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/\",\"domain\":\"kubernetes.io\"},{\"title\":\"Spot Instances for Batch Workloads - AWS\",\"url\":\"https://aws.amazon.com/ec2/spot/\",\"domain\":\"aws.amazon.com\"},{\"title\":\"deepseek-ai/deepseek-harness - Everything Is a Plugin\",\"url\":\"https://github.com/deepseek-ai/deepseek-harness\",\"domain\":\"github.com\"},{\"title\":\"DeepSeek Harness and Unbundled Agent Infrastructure (The New Stack)\",\"url\":\"https://thenewstack.io/deepseek-harness-open-source-plugins/\",\"domain\":\"thenewstack.io\"}]}},\"MCP \u0026 Tool Integration\":{\"1\":{\"guideSlug\":\"zero-mcp\",\"guideTitle\":\"Agent uses built-in tools only\",\"targetItems\":[{\"text\":\"Agent uses built-in tools only\",\"guideSlug\":\"zero-mcp\"},{\"text\":\"Agent relies on public / general knowledge\",\"guideSlug\":\"agent-knows-only-public-api-knowledge\"}
102,{\"text\":\"Integrations done by copy-paste\",\"guideSlug\":\"integrations-copy-paste\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$7f\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$80\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$81\"}],\"gettingStarted\":[\"Audit your current AI tool usage\",\"Identify the highest-value context gaps\",\"Evaluate MCP-compatible clients\"],\"links\":[{\"title\":\"Model Context Protocol - Official Introduction\",\"url\":\"https://modelcontextprotocol.io/introduction\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"MCP Server Registry - Available Servers\",\"url\":\"https://github.com/modelcontextprotocol/servers\",\"domain\":\"github.com\"},{\"title\":\"Claude Code MCP Documentation\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"Anthropic MCP Quickstart Guide\",\"url\":\"https://modelcontextprotocol.io/quickstart\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"Claude Code - Connecting to External Tools\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/overview\",\"domain\":\"docs.anthropic.com\"}]},\"2\":{\"guideSlug\":\"1-3-basic-mcp-servers-git-jira-docs\",\"guideTitle\":\"1-3 basic MCP servers (Git, Jira, docs)\",\"targetItems\":[{\"text\":\"1-3 basic MCP servers (Git, Jira, docs)\",\"guideSlug\":\"1-3-basic-mcp-servers-git-jira-docs\"},{\"text\":\"Manual MCP setup per developer\",\"guideSlug\":\"manual-mcp-setup-per-developer\"},{\"text\":\"Basic tool authorization; the stateless MCP core is shipping, so servers drop sticky sessions and run serverless - and Roots, Sampling and Logging are on a 12-month clock\",\"guideSlug\":\"basic-tool-authorization\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's team has been manually pasting context into AI tools for months. He's identified that git history and Jira tickets are the two most frequently pasted context types. He wants to fix this but is concerned about the engineering overhead of maintaining custom MCP servers alongside normal product work.\\n\\n**What Bob should do:** Bob should explicitly allocate one sprint for MCP infrastructure. The goal: deploy Git and Jira MCP servers for the whole team. The work should be assigned to one engineer (preferably Victor or whoever is already experimenting with AI tools) with a clear deliverable: every developer on the team has both servers configured and working by the end of the sprint. Bob should not allow this to be split across multiple sprints or treated as a \\\"when we have time\\\" item - the cumulative daily cost of manual context loading is already paying for this sprint many times over.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been running AI tool adoption programs and needs to show progress metrics to leadership. She wants concrete before/after numbers, not anecdotes. The transition from Zero MCP to 1-3 MCP servers is the first place in the maturity journey where a clean before/after measurement is feasible.\\n\\n**What Sarah should do:** Sarah should instrument the transition. Before deploying the Jira MCP server, have developers count how many times per day they paste Jira ticket text into AI tools. After deployment, run the same count for two weeks. The delta is a concrete productivity metric: \\\"we eliminated 47 manual context pastes per day, saving an estimated 94 developer-minutes per day.\\\" This number is what Sarah needs for the executive reporting and what justifies the next MCP infrastructure investment. She should build this measurement discipline in from the start: every MCP server deployment should have a before/after measurement of the specific paste operation it replaces.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$82\"}],\"gettingStarted\":[\"Install a Git MCP server\",\"Connect your issue tracker\",\"Add an internal docs server\"],\"links\":[{\"title\":\"MCP Server Registry - Official Servers\",\"url\":\"https://github.com/modelcontextprotocol/servers\",\"domain\":\"github.com\"},{\"title\":\"GitHub MCP Server\",\"url\":\"https://github.com/github/github-mcp-server\",\"domain\":\"github.com\"},{\"title\":\"MCP Server for Jira (Community)\",\"url\":\"https://github.com/sooperset/mcp-atlassian\",\"domain\":\"github.com\"},{\"title\":\"MCP Filesystem Server for Docs\",\"url\":\"https://github.com/modelcontextprotocol/servers/tree/main/src/filesystem\",\"domain\":\"github.com\"},{\"title\":\"Claude Code MCP Configuration\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp\",\"domain\":\"docs.anthropic.com\"}]},\"3\":{\"guideSlug\":\"mcp-platform-centralized-server-management\",\"guideTitle\":\"MCP platform: centralized server management\",\"targetItems\":[{\"text\":\"MCP platform: centralized server management\",\"guideSlug\":\"mcp-platform-centralized-server-management\"},{\"text\":\"Servers for architecture, ownership and SLA data are run as products: versioned, owned and monitored\",\"guideSlug\":\"architecture-mcp-ownership-mcp-sla-mcp\"},{\"text\":\"RBAC per MCP tool through an MCP gateway that shows each caller only the tools it may use (Kong MCP bundling); clients registered via Client ID Metadata Documents, tokens bound to one server; lazy tool-loading cuts tokens and live attack surface\",\"guideSlug\":\"rbac-per-mcp-tool\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$83\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$84\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$85\"}],\"gettingStarted\":[\"Inventory current MCP usage\",\"Choose a configuration format\",\"Set up centralized secret storage\"],\"links\":[{\"title\":\"Claude Code Teams MCP Configuration\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"HashiCorp Vault for Secrets Management\",\"url\":\"https://developer.hashicorp.com/vault/docs\",\"domain\":\"developer.hashicorp.com\"},{\"title\":\"MCP Server Architecture Reference\",\"url\":\"https://modelcontextprotocol.io/docs/concepts/architecture\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"1Password Secrets Automation\",\"url\":\"https://developer.1password.com/docs/connect/\",\"domain\":\"developer.1password.com\"},{\"title\":\"Doppler Team Secrets Management\",\"url\":\"https://docs.doppler.com/docs/secretops-beginners-series-team-management\",\"domain\":\"docs.doppler.com\"}]},\"4\":{\"guideSlug\":\"toolshed-model-400-tools-behind-one-mcp-stripe\",\"guideTitle\":\"The organisation's tool surface reachable through one governed MCP gateway, with access granted centrally by the IdP (Okta Cross App Access / ID-JAG, MCP Enterprise-Managed Authorization) instead of per-user consent sprawl, and a kill switch that revokes an agent's live tokens at the gateway\",\"targetItems\":[{\"text\":\"The organisation's tool surface reachable through one governed MCP gateway, with access granted centrally by the IdP (Okta Cross App Access / ID-JAG, MCP Enterprise-Managed Authorization) instead of per-user consent sprawl, and a kill switch that revokes an agent's live tokens at the gateway\",\"guideSlug\":\"toolshed-model-400-tools-behind-one-mcp-stripe\"},{\"text\":\"Agent discovery: agent knows what tools are available\",\"guideSlug\":\"agent-discovery-agent-knows-what-tools-are-available\"},{\"text\":\"MCP governance: every server on a lifecycle from intake to deprecation with a per-server pause switch; tool metadata watched for change at runtime, not only at install (Deadbugz rewrote its tool descriptions after three calls); plugins pinned by full commit SHA, never a branch name (Plugin4Shell); injection resistance tested in the IDE configuration you actually ship\",\"guideSlug\":\"mcp-governance-lifecycle-versioning-audit\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$86\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$87\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$88\"}],\"gettingStarted\":[\"Audit your current MCP server footprint\",\"Design the gateway layer\",\"Implement tool namespace conventions\"],\"links\":[{\"title\":\"Stripe Engineering - Minions: One-Shot End-to-End Coding Agents\",\"url\":\"https://stripe.dev/blog/minions-stripes-one-shot-end-to-end-coding-agents\",\"domain\":\"stripe.dev\"},{\"title\":\"MCP Protocol - Tools Specification\",\"url\":\"https://modelcontextprotocol.io/docs/concepts/tools\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"Building MCP Aggregator Servers\",\"url\":\"https://modelcontextprotocol.io/docs/concepts/architecture\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"Claude Code Enterprise MCP Configuration\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/mcp\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"MCP Server Registry - Reference Implementations\",\"url\":\"https://github.com/modelcontextprotocol/servers\",\"domain\":\"github.com\"},{\"title\":\"Okta AI innovations at Oktane 2026 - Agent Gateway, kill switch, Agent-to-Agent Connections\",\"url\":\"https://www.okta.com/newsroom/press-releases/ai-innovations-oktane-2026/\",\"domain\":\"okta.com\"},{\"title\":\"LiteLLM CVE-2026-59822 added to CISA KEV\",\"url\":\"https://tech-insider.org/litellm-mcp-vulnerability-cve-2026-59822-cisa-kev-2026/\",\"domain\":\"tech-insider.org\"}]},\"5\":{\"guideSlug\":\"mcp-as-nervous-system-bidirectional-context-flow\",\"guideTitle\":\"MCP as nervous system: bidirectional context flow\",\"targetItems\":[{\"text\":\"MCP as nervous system: bidirectional context flow\",\"guideSlug\":\"mcp-as-nervous-system-bidirectional-context-flow\"},{\"text\":\"Production â MCP â Agent â Code â Deploy â Production\",\"guideSlug\":\"production-mcp-agent-code-deploy-production\"},{\"text\":\"Agent-to-Agent Protocol (A2A) + MCP combined\",\"guideSlug\":\"agent-to-agent-protocol-a2a-mcp-combined\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$89\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$8a\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$8b\"}],\"gettingStarted\":[\"Implement MCP notification support\",\"Define the event taxonomy\",\"Connect your event streaming infrastru
102cture\"],\"links\":[{\"title\":\"MCP Protocol - Notifications Specification\",\"url\":\"https://modelcontextprotocol.io/docs/concepts/architecture\",\"domain\":\"modelcontextprotocol.io\"},{\"title\":\"Anthropic Claude Agents Documentation\",\"url\":\"https://docs.anthropic.com/en/docs/build-with-claude/agents\",\"domain\":\"docs.anthropic.com\"},{\"title\":\"OpenTelemetry Events and Spans\",\"url\":\"https://opentelemetry.io/docs/concepts/signals/traces/\",\"domain\":\"opentelemetry.io\"},{\"title\":\"Apache Kafka for Event Streaming\",\"url\":\"https://kafka.apache.org/documentation/\",\"domain\":\"kafka.apache.org\"},{\"title\":\"Claude Code Advanced Workflows\",\"url\":\"https://docs.anthropic.com/en/docs/claude-code/overview\",\"domain\":\"docs.anthropic.com\"}]}},\"Build System\":{\"1\":{\"guideSlug\":\"maven-gradle-default-config\",\"guideTitle\":\"Maven/Gradle default config\",\"targetItems\":[{\"text\":\"Maven/Gradle default config\",\"guideSlug\":\"maven-gradle-default-config\"},{\"text\":\"Full rebuild on every change\",\"guideSlug\":\"full-rebuild-on-every-change\"},{\"text\":\"Each build recomputes the full graph from source, on every machine\",\"guideSlug\":\"ci-shared-queue-everyone-waits\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob's teams have adopted Claude Code for individual developer use. Agents are running, PRs are getting created, but developers are complaining that agent loops feel slow - the agent runs, waits, adjusts, waits again. Bob hasn't connected this to build performance; he thinks it's a model latency issue.\\n\\n**What Bob should do:** Bob should ask his DevEx lead (Sarah) to instrument agent iteration cycle time and identify how much of that time is build wait time. The answer is almost always \\\"more than you think.\\\" Once the bottleneck is quantified, the fix is a one-day Gradle configuration project: enable caching, the daemon, and parallel execution. Bob should treat this as an infrastructure investment with a direct return on agent throughput. A 50% reduction in build time doubles the number of agent iterations in a given time window - that compounds across every agent, every developer, every day.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah tracks developer experience metrics and has noticed that agent usage has plateaued. Developers use agents for greenfield tasks but switch back to manual coding for iterative work. When she asks why, the answer is consistent: \\\"waiting for the build is frustrating enough that it breaks the flow.\\\"\\n\\n**What Sarah should do:** Sarah should add build time to her DevEx dashboard - specifically, the 50th and 95th percentile build time on agent-driven branches. She should then set a target: sub-60-second warm builds for any module an agent is likely to touch. The path from L1 to that target is well-understood: Gradle daemon, caching, parallel execution. Sarah should run the configuration change as a sprint task for one engineer, measure the before/after impact on agent session length (agents that get fast feedback iterate more), and use the result to justify further investment in L3 build tooling.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has set up Gradle caching locally and his builds are fast. But he's noticed that when he runs 3-4 parallel agents in worktrees, the build times are slower than when he runs a single agent - cache contention, lock files, and occasional corruption.\\n\\n**What Victor should do:** Victor should set up a local remote cache backend using the Gradle Build Cache Docker image or a simple HTTP cache server. This gives all his parallel worktrees a shared, concurrency-safe cache without requiring a full Gradle Enterprise setup. He should also configure each worktree's `gradle.properties` to point at the shared cache. Once parallel agents are hitting a shared cache rather than competing for a local file-system cache, build times should drop to near-single-agent levels. Victor should document this setup as the standard for parallel agent development and push for a team-wide remote cache at L3.\"}],\"gettingStarted\":[\"Enable the Gradle Build Cache\",\"Enable the Gradle daemon\",\"Enable parallel execution\"],\"links\":[{\"title\":\"Gradle Build Cache Documentation\",\"url\":\"https://docs.gradle.org/current/userguide/build_cache.html\",\"domain\":\"docs.gradle.org\"},{\"title\":\"Gradle Performance Guide\",\"url\":\"https://docs.gradle.org/current/userguide/performance.html\",\"domain\":\"docs.gradle.org\"},{\"title\":\"Maven Incremental Compilation\",\"url\":\"https://maven.apache.org/plugins/maven-compiler-plugin/compile-mojo.html\",\"domain\":\"maven.apache.org\"},{\"title\":\"Gradle Build Scan\",\"url\":\"https://scans.gradle.com/\",\"domain\":\"scans.gradle.com\"},{\"title\":\"Gradle Daemon Documentation\",\"url\":\"https://docs.gradle.org/current/userguide/gradle_daemon.html\",\"domain\":\"docs.gradle.org\"}]},\"2\":{\"guideSlug\":\"basic-build-caching\",\"guideTitle\":\"Basic build caching; packages an agent installs from docs or llms.txt are verified against a registry allowlist first (237 of 8,565 llms.txt files pointed at dead, typo'd or unregistered packages)\",\"targetItems\":[{\"text\":\"Basic build caching; packages an agent installs from docs or llms.txt are verified against a registry allowlist first (237 of 8,565 llms.txt files pointed at dead, typo'd or unregistered packages)\",\"guideSlug\":\"basic-build-caching\"},{\"text\":\"Parallel build steps\",\"guideSlug\":\"parallel-build-steps\"},{\"text\":\"The build cache is shared between developers and CI, not rebuilt per machine\",\"guideSlug\":\"dedicated-ci-resources\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"Bob has approved an initiative to improve developer productivity and wants to show a quick win. His DevEx lead has identified that CI pipeline time averages 12 minutes, with 4 minutes spent downloading dependencies on every run. This is a clear target: depen
102dency caching should eliminate those 4 minutes immediately.\\n\\n**What Bob should do:** Bob should have his infrastructure team implement CI dependency caching for the main repositories this week - it's a half-day task with immediate measurable impact. He should track average CI time before and after, and share the result with the team as a productivity win. This builds momentum for the more significant investment in L3 build infrastructure. Bob should also ask whether agent-generated CI runs are being measured separately - if they are, the before/after comparison for agent iteration loop time will be even more dramatic than for human PR builds.\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been collecting CI time data and sees that the distribution is bimodal: fast builds under 4 minutes (cache hits) and slow builds over 12 minutes (cache misses). The slow builds happen predictably when lock files change or new CI runners start. She wants to narrow this distribution.\\n\\n**What Sarah should do:** Sarah should work with the infrastructure team to understand what's causing the slow tail. For each major source of cache misses - lock file changes, cold CI runners, first build on a new branch - there's a specific caching strategy. Lock file changes should trigger cache refresh automatically. Cold CI runners should be pre-warmed with a daily scheduled build that populates the cache. New branches should seed from the main branch cache rather than starting cold. Each of these is a targeted intervention that narrows the build time distribution and makes CI time more predictable for both developers and agents.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"Victor has implemented all local and CI caching and is getting consistent warm build times of 45 seconds for his main module. But he's realized that local caches are not shared between his 4 parallel worktrees - each worktree maintains its own Gradle cache, so the same compilation work happens 4 times instead of once.\\n\\n**What Victor should do:** Victor should configure a local remote cache server to share between his worktrees. The Gradle Build Cache Docker image runs as a local HTTP server that all worktrees can point at. Each worktree's `gradle.properties` gets `org.gradle.caching=true` and a remote cache URL pointing at the local server. With this setup, the first worktree to compile a given module populates the local server, and all other worktrees get cache hits. Victor should measure the shared cache hit rate across his 4 worktrees and target 70%+ - this is the configuration that makes local parallel agents nearly as fast as a single agent for shared infrastructure code.\"}],\"gettingStarted\":[\"Enable Gradle local build cache\",\"Cache CI dependencies\",\"Enable npm/yarn/pnpm caching\"],\"links\":[{\"title\":\"Gradle Build Cache User Guide\",\"url\":\"https://docs.gradle.org/current/userguide/build_cache.html\",\"domain\":\"docs.gradle.org\"},{\"title\":\"GitHub Actions Caching Dependencies\",\"url\":\"https://docs.github.com/en/actions/writing-workflows/choosing-what-your-workflow-does/caching-dependencies-to-speed-up-workflows\",\"domain\":\"docs.github.com\"},{\"title\":\"Docker BuildKit Cache\",\"url\":\"https://docs.docker.com/build/cache/backends/\",\"domain\":\"docs.docker.com\"},{\"title\":\"npm CI Caching Best Practices\",\"url\":\"https://docs.npmjs.com/cli/v10/commands/npm-ci\",\"domain\":\"docs.npmjs.com\"},{\"title\":\"Maven Dependency Caching in CI\",\"url\":\"https://maven.apache.org/extensions/maven-build-cache-extension/\",\"domain\":\"maven.apache.org\"}]},\"3\":{\"guideSlug\":\"incremental-builds-only-changed-targets\",\"guideTitle\":\"Incremental builds: only changed targets\",\"targetItems\":[{\"text\":\"Incremental builds: only changed targets\",\"guideSlug\":\"incremental-builds-only-changed-targets\"},{\"text\":\"Remote execution distributing build steps across machines\",\"guideSlug\":\"remote-execution-engflow\"},{\"text\":\"A build tool with an explicit dependency graph and a shared cache (Bazel, Buck2, Pants, Nx, Turborepo)\",\"guideSlug\":\"bazel-buck2-pants\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$8c\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has added \\\"agent build time by change type\\\" to her DevEx dashboard. She can see that 90% of agent builds complete under 15 seconds, but 10% take over 90 seconds. She wants to reduce the 10% tail. The data shows the long-tail builds are consistently triggered by changes to 12 specific \\\"high-fan-out\\\" targets.\\n\\n**What Sarah should do:** Sarah should prioritize refactoring those 12 high-fan-out targets as a DevEx investment. Each one is a hot spot: agents that touch these targets pay a disproportionate build time penalty. She should work with the owners of those targets to reduce their dependency surface. Techniques include: splitting large targets into smaller ones, moving stable code to a separate target that changes less frequently, using interface abstractions to decouple implementations from interfaces. Sarah should track \\\"number of high-fan-out targets with \u003e100 dependents\\\" as a metric and drive it to zero over two quarters.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$8d\"}],\"gettingStarted\":[\"Understand your current dependency graph shape\",\"Split large targets into smaller, more focused targets\",\"Measure your actual affected-target set for common change types\"],\"links\":[{\"title\":\"Bazel Query Language\",\"url\":\"https://bazel.build/query/language\",\"domain\":\"bazel.build\"},{\"title\":\"Bazel Test Caching\",\"url\":\"https://bazel.build/docs/output_directories#out-structure\",\"domain\":\"bazel.build\"},{\"title\":\"Bazel Dependency Graph Visualization\",\"url\":\"https://bazel.build/query/guide\",\"domain\":\"bazel.build\"},{\"title\":\"Buck2 Incremental Builds\",\"url\":\"https://buck2.build/docs/rule_authors/incremental_actions/\",\"domain\":\"buck2.build\"},{\"title\":\"How Google Handles Monorepo Scale with Bazel\",\"url\":\"https://cacm.acm.org/research/why-google-stores-billions-of-lines-of-code-in-a-single-repository/\",\"domain\":\"cacm.acm.org\"}]},\"4\":{\"guideSlug\":\"agent-specific-build-profiles\",\"guideTitle\":\"Agent-specific build profiles; multi-root workspaces and worktree isolation per agent, now shipped as a default by agents themselves rather than assembled by hand\",\"targetItems\":[{\"text\":\"Agent-specific build profiles; multi-root workspaces and worktree isolation per agent, now shipped as a default by agents themselves rather than assembled by hand\",\"guideSlug\":\"agent-specific-build-profiles\"},{\"text\":\"Build system aware of agent iteration patterns\",\"guideSlug\":\"build-system-aware-of-agent-iteration-patterns\"},{\"text\":\"Sub-2min feedback on any change\",\"guideSlug\":\"sub-2min-feedback-on-any-change\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$8e\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has been tracking agent CI costs and sees that they're growing proportionally with agent adoption - each new developer adding agents adds a linear increment to CI costs. She wants to decouple agent adoption from CI cost growth.\\n\\n**What Sarah should do:** Sarah should frame agent build profiles as a cost management strategy, not just a performance optimization. If agent CI can run for 10% of the cost of full CI, the cost-per-agent-iteration drops by 90%. This means the organization can support
10210x more agent iterations for the same CI budget, or achieve the same agent throughput at 10% of the current CI cost. Sarah should model this: current CI cost per agent iteration, projected cost per iteration with a lean profile, breakeven point. This financial framing often moves faster through budget discussions than performance framing alone.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$8f\"}],\"gettingStarted\":[\"Audit your current CI pipeline\",\"Define the agent iteration profile contents\",\"Create a `.bazelrc` config for agent builds\"],\"links\":[{\"title\":\"Bazel Build Configurations (.bazelrc)\",\"url\":\"https://bazel.build/docs/bazelrc\",\"domain\":\"bazel.build\"},{\"title\":\"GitHub Actions Workflow Triggers\",\"url\":\"https://docs.github.com/en/actions/writing-workflows/choosing-when-your-workflow-runs/triggering-a-workflow\",\"domain\":\"docs.github.com\"},{\"title\":\"Bazel Test Tag Filters\",\"url\":\"https://bazel.build/reference/command-line-reference#flag--test_tag_filters\",\"domain\":\"bazel.build\"},{\"title\":\"Gradle Task Profiles and Exclusions\",\"url\":\"https://docs.gradle.org/current/userguide/command_line_interface.html#sec:excluding_tasks_from_the_command_line\",\"domain\":\"docs.gradle.org\"},{\"title\":\"Nx Affected Commands for Monorepos\",\"url\":\"https://nx.dev/ci/features/affected\",\"domain\":\"nx.dev\"},{\"title\":\"git worktree - Manage Multiple Working Trees\",\"url\":\"https://git-scm.com/docs/git-worktree\",\"domain\":\"git-scm.com\"}]},\"5\":{\"guideSlug\":\"build-commodity-near-instant-for-agents\",\"guideTitle\":\"Build = commodity (near-instant for agents)\",\"targetItems\":[{\"text\":\"Build = commodity (near-instant for agents)\",\"guideSlug\":\"build-commodity-near-instant-for-agents\"},{\"text\":\"Compilation bottleneck eliminated via crate/module architecture (the Bun lesson: a conformance test suite makes even a runtime rewrite verifiable)\",\"guideSlug\":\"compilation-bottleneck-eliminated-via-crate-module-architect\"},{\"text\":\"Disk I/O optimized for concurrent agent workloads (Cursor lesson)\",\"guideSlug\":\"disk-i-o-optimized-for-concurrent-agent-workloads-cursor-les\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$90\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$91\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$92\"}],\"gettingStarted\":[\"Audit your current build time distribution\",\"Verify the full stack is in place\",\"Validate under concurrent agent load\"],\"links\":[{\"title\":\"Google's Billion-Line Repository: Why Bazel?\",\"url\":\"https://cacm.acm.org/research/why-google-stores-billions-of-lines-of-code-in-a-single-repository/\",\"domain\":\"cacm.acm.org\"},{\"title\":\"Stripe's Build Infrastructure at Scale\",\"url\":\"https://stripe.dev/blog/fast-secure-builds-choose-two\",\"domain\":\"stripe.dev\"},{\"title\":\"EngFlow: Bazel Scales More Than Just Builds\",\"url\":\"https://blog.engflow.com/2024/01/31/bazel-scales-more-than-just-builds/\",\"domain\":\"blog.engflow.com\"},{\"title\":\"Bazel Performance Tuning Guide\",\"url\":\"https://bazel.build/configure/best-practices\",\"domain\":\"bazel.build\"},{\"title\":\"BuildBuddy: Achieving Fast Builds\",\"url\":\"https://www.buildbuddy.io/blog/debugging-slow-bazel-builds/\",\"domain\":\"buildbuddy.io\"}]}},\"Observability \u0026 Feedback Loop\":{\"1\":{\"guideSlug\":\"basic-logging\",\"guideTitle\":\"Basic logging\",\"targetItems\":[{\"text\":\"Basic logging\",\"guideSlug\":\"basic-logging\"},{\"text\":\"Alerting on errors\",\"guideSlug\":\"alerting-on-errors\"},{\"text\":\"Prod feedback and token-cost visibility not yet wired to dev\",\"guideSlug\":\"no-connection-prod-dev-feedback\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$93\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$94\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$95\"}],\"gettingStarted\":[\"Audit your current log output\",\"Add consistent severity levels\",\"Stop writing logs to files; write to stdout\"],\"links\":[{\"title\":\"The Twelve-Factor App: Logs\",\"url\":\"https://12factor.net/logs\",\"domain\":\"12factor.net\"},{\"title\":\"OpenTelemetry Logging\",\"url\":\"https://opentelemetry.io/
102docs/concepts/signals/logs/\",\"domain\":\"opentelemetry.io\"},{\"title\":\"Datadog Log Management\",\"url\":\"https://docs.datadoghq.com/logs/\",\"domain\":\"docs.datadoghq.com\"},{\"title\":\"Grafana Loki - Log Aggregation\",\"url\":\"https://grafana.com/oss/loki/\",\"domain\":\"grafana.com\"},{\"title\":\"Structured Logging Best Practices\",\"url\":\"https://www.honeycomb.io/blog/structured-events-basis-observability\",\"domain\":\"honeycomb.io\"}]},\"2\":{\"guideSlug\":\"structured-logging\",\"guideTitle\":\"Structured logging\",\"targetItems\":[{\"text\":\"Structured logging\",\"guideSlug\":\"structured-logging\"},{\"text\":\"OpenTelemetry basic\",\"guideSlug\":\"opentelemetry-basic\"},{\"text\":\"Post-deploy monitoring; per-session token cost as table stakes; input tokens (context) drive spend - track them, not output\",\"guideSlug\":\"post-deploy-monitoring\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$96\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah has heard developers complain that finding the relevant log line during an incident is harder than
102it should be. She measures the time from \\\"alert fires\\\" to \\\"root cause identified\\\" and knows it is too long. She suspects that moving to structured logging would cut this time significantly.\\n\\n**What Sarah should do:** Sarah should run a structured logging pilot with one team and measure the before/after investigation time for a comparable set of incidents. The hypothesis is clear: structured logs with field-based queries dramatically reduce time-to-root-cause compared to unstructured text search. If the pilot confirms the hypothesis (it will), the data makes the case for the rest of the organization. Sarah should also note that structured logging is the prerequisite for agent-assisted investigation: without structured, queryable logs, agents cannot meaningfully participate in incident response. Every minute invested in structured logging migration is a minute that enables future agent automation.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$97\"}],\"gettingStarted\":[\"Choose a structured logging library for each service\",\"Define your mandatory field schema\",\"Add a request-scoped context carrier\"],\"links\":[{\"title\":\"OpenTelemetry Semantic Conventions for Logs\",\"url\":\"https://opentelemetry.io/
102docs/specs/semconv/general/logs/\",\"domain\":\"opentelemetry.io\"},{\"title\":\"Grafana Loki - Structured Logging\",\"url\":\"https://grafana.com/docs/loki/latest/fundamentals/labels/\",\"domain\":\"grafana.com\"},{\"title\":\"Datadog Log Management - Parsing\",\"url\":\"https://docs.datadoghq.com/logs/log_configuration/parsing/\",\"domain\":\"docs.datadoghq.com\"},{\"title\":\"structlog - Python Structured Logging\",\"url\":\"https://www.structlog.org/\",\"domain\":\"structlog.org\"},{\"title\":\"Uber zap - High Performance Go Logging\",\"url\":\"https://github.com/uber-go/zap\",\"domain\":\"github.com\"}]},\"3\":{\"guideSlug\":\"full-observability-stack-otel-grafana\",\"guideTitle\":\"Full observability stack (OTel + Grafana)\",\"targetItems\":[{\"text\":\"Full observability stack (OTel + Grafana)\",\"guideSlug\":\"full-observability-stack-otel-grafana\"},{\"text\":\"Production metrics â dashboards; agent telemetry through a governed gateway (self-hosted control plane: identity, policy, telemetry - Claude Apps Gateway model; catch shadow AI via proxy)\",\"guideSlug\":\"production-metrics-dashboards\"},{\"text\":\"Incident data available for context; usage-truth reconciliation (client-reported tokens vs agent-claimed work); every tool call traced to the agent identity, the human it acts for and the token used\",\"guideSlug\":\"incident-data-available-for-context\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$98\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"$99\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$9a\"}],\"gettingStarted\":[\"Deploy the Grafana OSS stack\",\"Configure the OTel Collector as the central pipeline\",\"Enable exemplars in Prometheus and Grafana\"],\"links\":[{\"title\":\"Grafana OSS Stack - Getting Started\",\"url\":\"https://grafana.com/docs/grafana/latest/getting-started/\",\"domain\":\"grafana.com\"},{\"title\":\"Prometheus + Grafana Helm Chart\",\"url\":\"https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack\",\"domain\":\"github.com\"},{\"title\":\"Grafana Tempo - Distributed Tracing\",\"url\":\"https://grafana.com/oss/tempo/\",\"domain\":\"grafana.com\"},{\"title\":\"Google SRE - SLO Implementation\",\"url\":\"https://sre.google/workbook/implementing-slos/\",\"domain\":\"sre.google\"},{\"title\":\"OpenTelemetry Collector Configuration\",\"url\":\"https://opentelemetry.io/
102docs/collector/configuration/\",\"domain\":\"opentelemetry.io\"}]},\"4\":{\"guideSlug\":\"production-anomaly-auto-ticket-agent-investigation\",\"guideTitle\":\"Production anomaly â auto-ticket â agent investigation; incident rate and firefighting hours tracked against change volume, because the two move in opposite directions\",\"targetItems\":[{\"text\":\"Production anomaly â auto-ticket â agent investigation; incident rate and firefighting hours tracked against change volume, because the two move in opposite directions\",\"guideSlug\":\"production-anomaly-auto-ticket-agent-investigation\"},{\"text\":\"Self-healing basic: known patterns auto-fixed, with diagnosis kept human; anomalous agent behaviour (scope drift, credential reuse, unusual egress) revokes the agent's credentials automatically - a training-sandbox escape in September got past partial monitoring because the automatic shutdown failed\",\"guideSlug\":\"self-healing-basic-known-patterns-auto-fixed\"},{\"text\":\"Infrastructure recommends code changes back to the team, and agent sessions are audited for quality regression over time\",\"guideSlug\":\"vercel-sdi-model-infra-recommends-code-changes\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$9b\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah tracks on-call sustainability as a developer experience metric. She knows that unsustainable on-call rotations cause senior engineers to leave and make hiring harder. The agent investigation pipeline is a direct intervention on this problem.\\n\\n**What Sarah should do:** Sarah should instrument the full incident response timeline with the agent pipeline in place: time from anomaly detection to agent starting investigation, time from agent investigation to human decision, time from human decision to resolution. Compare this to the pre-agent baseline. The improvement in \\\"time from anomaly to human decision\\\" is the clearest evidence that the pipeline is working. Sarah should also interview on-call engineers after their first month with the agent pipeline: are they more confident? Are incidents less stressful? Do they feel the agent's analyses are useful? This qualitative data supplements the quantitative metrics and surfaces issues with agent output quality that numbers alone might miss.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$9c\"}],\"gettingStarted\":[\"Define structured anomaly event schemas\",\"Build the auto-ticket creation webhook\",\"Implement the investigation agent\"],\"links\":[{\"title\":\"PagerDuty - Event Intelligence\",\"url\":\"https://www.pagerduty.com/platform/aiops/\",\"domain\":\"pagerduty.com\"},{\"title\":\"Prometheus Alertmanager Webhook Integration\",\"url\":\"https://prometheus.io/docs/alerting/latest/configuration/#webhook_config\",\"domain\":\"prometheus.io\"},{\"title\":\"Datadog Monitors and Webhooks\",\"url\":\"https://docs.datadoghq.com/monitors/notify/\",\"domain\":\"docs.datadoghq.com\"},{\"title\":\"Anthropic - Building Effective Agents\",\"url\":\"https://www.anthropic.com/engineering/building-effective-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"Google SRE - Incident Management\",\"url\":\"https://sre.google/sre-book/managing-incidents/\",\"domain\":\"sre.google\"},{\"title\":\"AI-Generated C++ in Production: Quality and Compute Cost (arXiv 2608.06640)\",\"url\":\"https://arxiv.org/abs/2608.06640\",\"domain\":\"arxiv.org\"}]},\"5\":{\"guideSlug\":\"full-production-agent-loop\",\"guideTitle\":\"Full production â agent loop\",\"targetItems\":[{\"text\":\"Full production â agent loop\",\"guideSlug\":\"full-production-agent-loop\"},{\"text\":\"Anomaly â investigate â fix â test â deploy autonomous\",\"guideSlug\":\"anomaly-investigate-fix-test-deploy-autonomous\"},{\"text\":\"Infrastructure self-drives: code defines infra, production informs code\",\"guideSlug\":\"infrastructure-self-drives-code-defines-infra-production-inf\"}],\"personas\":[{\"name\":\"Bob\",\"role\":\"Head of Engineering\",\"comment\":\"$9d\"},{\"name\":\"Sarah\",\"role\":\"Productivity Lead\",\"comment\":\"Sarah wants development teams to experience the production-agent loop as a reduction in operational burden rather than a loss of control. Developer anxiety about \\\"AI changing our code\\\" is a real adoption barrier that needs to be addressed proactively.\\n\\n**What Sarah should do:** Sarah should design the developer experience of the loop around transparency and opt-in. Every agent-generated change should appear in the normal PR workflow with clear labeling (\\\"generated by production-optimization-agent based on trace data from 2024-01-15\\\"). Developers should be able to review, modify, and reject agent PRs using the same workflow they use for human PRs. The first opt-in should be for the lowest-risk category (documentation, comments, test improvements) so developers can experience the loop as helpful before it operates on production code. Sarah should track developer sentiment toward the loop in monthly surveys and address concerns directly with evidence from the loop's audit trail.\"},{\"name\":\"Victor\",\"role\":\"Staff Engineer - AI Champion\",\"comment\":\"$9e\"}
102],\"gettingStarted\":[\"Validate the prerequisite infrastructure before building the loop\",\"Define the agent policy framework\",\"Implement the work item queue\"],\"links\":[{\"title\":\"Anthropic - Building Effective Agents\",\"url\":\"https://www.anthropic.com/engineering/building-effective-agents\",\"domain\":\"anthropic.com\"},{\"title\":\"Argo Workflows - Kubernetes Workflow Engine\",\"url\":\"https://argoproj.github.io/workflows/\",\"domain\":\"argoproj.github.io\"},{\"title\":\"Temporal - Durable Execution for Agent Workflows\",\"url\":\"https://temporal.io/\",\"domain\":\"temporal.io\"},{\"title\":\"Google SRE - Eliminating Toil\",\"url\":\"https://sre.google/sre-book/eliminating-toil/\",\"domain\":\"sre.google\"},{\"title\":\"OpenTelemetry - Continuous Profiling\",\"url\":\"https://opentelemetry.io/docs/concepts/signals/profiles/\",\"domain\":\"opentelemetry.io\"}]}}}},\"editionSlug\":\"2026-10\"}]]}],[\"$L9f\",\"$La0\",\"$La1\",\"$La2\",\"$La3\",\"$La4\"],\"$La5\"]}]\n"])</script>
102<script>self.__next_f.push([1,"8:[\"$\",\"$1\",\"h\",{\"children\":[null,[\"$\",\"$La6\",null,{\"children\":\"$La7\"}],[\"$\",\"div\",null,{\"hidden\":true,\"children\":[\"$\",\"$La8\",null,{\"children\":[\"$\",\"$a9\",null,{\"name\":\"Next.Metadata\",\"children\":\"$Laa\"}]}]}],[\"$\",\"meta\",null,{\"name\":\"next-size-adjust\",\"content\":\"\"}]]}]\n"])</script>
102<script>self.__next_f.push([1,"ab:I[44887,[\"/_next/static/chunks/8a2c28f82c6d4f4c.js\",\"/_next/static/chunks/3e8b73794b0ba35e.js\"],\"OutletBoundary\"]\n9f:[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/chunks/18b6abd70bbeface.js\",\"async\":true,\"nonce\":\"$undefined\"}]\na0:[\"$\",\"script\",\"script-1\",{\"src\":\"/_next/static/chunks/8827248f36da4bf2.js\",\"async\":true,\"nonce\":\"$undefined\"}]\na1:[\"$\",\"script\",\"script-2\",{\"src\":\"/_next/static/chunks/47f82402146413fb.js\",\"async\":true,\"nonce\":\"$undefined\"}]\na2:[\"$\",\"script\",\"script-3\",{\"src\":\"/_next/static/chunks/b7312459423597a8.js\",\"async\":true,\"nonce\":\"$undefined\"}]\na3:[\"$\",\"script\",\"script-4\",{\"src\":\"/_next/static/chunks/705056c08acab783.js\",\"async\":true,\"nonce\":\"$undefined\"}]\na4:[\"$\",\"script\",\"script-5\",{\"src\":\"/_next/static/chunks/1a514990f47add69.js\",\"async\":true,\"nonce\":\"$undefined\"}]\na5:[\"$\",\"$Lab\",null,{\"children\":[\"$\",\"$a9\",null,{\"name\":\"Next.MetadataOutlet\",\"children\":\"$@ac\"}]}]\n"])</script>
102<script>self.__next_f.push([1,"a7:[[\"$\",\"meta\",\"0\",{\"charSet\":\"utf-8\"}],[\"$\",\"meta\",\"1\",{\"name\":\"viewport\",\"content\":\"width=device-width, initial-scale=1\"}]]\n"])</script>
102<script>self.__next_f.push([1,"ad:I[410839,[\"/_next/static/chunks/8a2c28f82c6d4f4c.js\",\"/_next/static/chunks/3e8b73794b0ba35e.js\"],\"IconMark\"]\n"])</script>
102<script>self.__next_f.push([1,"aa:[[\"$\",\"title\",\"0\",{\"children\":\"Workshop Assessment | Visdom Maturity Matrix\"}],[\"$\",\"meta\",\"1\",{\"name\":\"description\",\"content\":\"Walk 20 areas, check criteria you have, find the level blocking your agents. Generates a shareable maturity report by perspective.\"}],[\"$\",\"link\",\"2\",{\"rel\":\"canonical\",\"href\":\"https://visdom-maturity-matrix.virtuslab.com/workshop\"}],[\"$\",\"meta\",\"3\",{\"property\":\"og:title\",\"content\":\"Workshop assessment\"}],[\"$\",\"meta\",\"4\",{\"property\":\"og:description\",\"content\":\"Walk 16 capabilities, check the criteria you have, and find the level blocking your agents. Generates a shareable maturity report.\"}],[\"$\",\"meta\",\"5\",{\"property\":\"og:url\",\"content\":\"https://visdom-maturity-matrix.virtuslab.com/workshop\"}],[\"$\",\"meta\",\"6\",{\"property\":\"og:image:alt\",\"content\":\"Workshop assessment - find the level blocking your agents\"}],[\"$\",\"meta\",\"7\",{\"property\":\"og:image:type\",\"content\":\"image/png\"}],[\"$\",\"meta\",\"8\",{\"property\":\"og:image\",\"content\":\"https://visdom-maturity-matrix.virtuslab.com/workshop/opengraph-image?c1d725e4417a00ac\"}],[\"$\",\"meta\",\"9\",{\"property\":\"og:image:width\",\"content\":\"1200\"}],[\"$\",\"meta\",\"10\",{\"property\":\"og:image:height\",\"content\":\"630\"}],[\"$\",\"meta\",\"11\",{\"property\":\"og:type\",\"content\":\"website\"}],[\"$\",\"meta\",\"12\",{\"name\":\"twitter:card\",\"content\":\"summary_large_image\"}],[\"$\",\"meta\",\"13\",{\"name\":\"twitter:title\",\"content\":\"Workshop assessment\"}],[\"$\",\"meta\",\"14\",{\"name\":\"twitter:description\",\"content\":\"Walk 16 capabilities, check the criteria you have, and find the level blocking your agents. Generates a shareable maturity report.\"}],[\"$\",\"meta\",\"15\",{\"name\":\"twitter:image\",\"content\":\"https://visdom-maturity-matrix.virtuslab.com/workshop/opengraph-image\"}],[\"$\",\"link\",\"16\",{\"rel\":\"icon\",\"href\":\"/favicon.ico\",\"sizes\":\"32x32\"}],[\"$\",\"link\",\"17\",{\"rel\":\"icon\",\"href\":\"/favicon/favicon-light-32x32.png\",\"type\":\"image/png\",\"sizes\":\"32x32\",\"media\":\"(prefers-color-scheme: light)\"}],[\"$\",\"link\",\"18\",{\"rel\":\"icon\",\"href\":\"/favicon/favicon-dark-32x32.png\",\"type\":\"image/png\",\"sizes\":\"32x32\",\"media\":\"(prefers-color-scheme: dark)\"}],[\"$\",\"link\",\"19\",{\"rel\":\"apple-touch-icon\",\"href\":\"/favicon/apple-touch-icon-light.png\",\"sizes\":\"180x180\"}],[\"$\",\"$Lad\",\"20\",{}]]\n"])</script>
102<script>self.__next_f.push([1,"ac:null\n"])</script>
102</body></html>
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.