1"use strict";(self.webpackChunkevopolicygym_site=self.webpackChunkevopolicygym_site||[]).push([[4812],{8400(e,n,t){t.r(n),t.d(n,{assets:()=>c,contentTitle:()=>o,default:()=>h,frontMatter:()=>r,metadata:()=>i,toc:()=>d});var i=t(3659),s=t(4848),a=t(8453);const r={locale:"en",page:"designing-evopolicygym",title:"EvoPolicyGym: environments for agents that build strategy systems",description:"Why EvoPolicyGym uses interactive Environments and executable Programs to study and train coding agents.",lead:"A coding agent studies an Environment, learns from bounded Feedback, and turns its conclusions into an executable strategy system.",publishedAt:"2026-07-27",date:"2026-07-27",authors:["evopolicygym"],tags:["Design","Motivation","Architecture"],status:"published"},o=void 0,c={authorsImageUrls:[void 0]},d=[{value:"Coding agents as policy-system experts",id:"coding-agents-as-policy-system-experts",level:2},{value:"Separate the expert from the Policy",id:"separate-the-expert-from-the-policy",level:2},{value:"Make the artifact first-class",id:"make-the-artifact-first-class",level:2}
1,{value:"Let each Benchmark define useful Feedback",id:"let-each-benchmark-define-useful-feedback",level:2},{value:"Treat evidence access as part of the experiment",id:"treat-evidence-access-as-part-of-the-experiment",level:2},{value:"Keep the Kernel focused",id:"keep-the-kernel-focused",level:2},{value:"What EvoPolicyGym enables",id:"what-evopolicygym-enables",level:2},{value:"Study agents that evolve strategy systems",id:"study-agents-that-evolve-strategy-systems",level:3},{value:"Train coding agents with interactive Environments",id:"train-coding-agents-with-interactive-environments",level:3},{value:"Continue reading",id:"continue-reading",level:2}];function l(e){const n={a:"a",code:"code",em:"em",h2:"h2",h3:"h3",li:"li",p:"p",pre:"pre",strong:"strong",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",ul:"ul",...(0,a.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(n.h2,{id:"coding-agents-as-policy-system-experts",children:"Coding agents as policy-system experts"}),"\n",(0,s.jsxs)(n.p,{children:["EvoPolicyGym was directly inspired by Jiayi Weng's\n",(0,s.jsx)(n.a,{href:"https://trinkle23897.github.io/learning-beyond-gradients/",children:"Learning Beyond Gradients"}),".\nThe article describes ",(0,s.jsx)(n.em,{children:"Heuristic Learning"}),": a coding agent absorbs rewards,\nfailures, tests, logs, and replays, then improves a programmatic policy by\nediting the software system itself. It showed us that agentic coding can serve\nas a learning process whose evolving state is explicit in code."]}),"\n",(0,s.jsx)(n.p,{children:"EvoPolicyGym begins from that insight and asks how to make the process bounded,\nreproducible, and comparable across interactive Environments."}),"\n",(0,s.jsx)(n.p,{children:"An EvoPolicyGym Run places a coding agent in the role of a policy-system\nexpert. The agent studies the Environment and its interface, inspects an\ninitial Program, forms hypotheses about successful behavior, and writes those\nideas into a complete executable Policy system."}),"\n",(0,s.jsx)(n.p,{children:"That system may combine domain knowledge, state estimation, rules, planning,\nsearch, memory, algorithms, or tuned parameters. The coding agent is free to\nchange its internal design as evidence accumulates. Its responsibility is to\nturn what it learns into source code that can make decisions on its own."}),"\n",(0,s.jsx)(n.p,{children:"Authoring and execution occupy two distinct phases. During Evaluation, the\nsubmitted Policy independently receives observations and produces Actions.\nThe result is a separable strategy artifact that can be frozen, inspected,\nrerun, and compared."}),"\n",(0,s.jsx)(n.p,{children:"Environment Feedback closes the engineering loop:"}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-text",children:"study the Environment\n \u2193\nauthor an executable Policy system\n \u2193\nsubmit and evaluate it\n \u2193\ninspect scores, traces, and artifacts\n \u2193\ndiagnose, redesign, and submit again\n"})}),"\n",(0,s.jsx)(n.p,{children:"This is the motivation for Autonomous Policy Evolution in EvoPolicyGym. The\ncoding agent contributes expertise and software engineering; the Environment\ncontributes empirical evidence; the evolving Program records the resulting\nstrategy. The central question is how effectively an agent can transform\nlimited Environment Feedback into a better executable decision system."}),"\n",(0,s.jsx)(n.h2,{id:"separate-the-expert-from-the-policy",children:"Separate the expert from the Policy"}),"\n",(0,s.jsx)(n.p,{children:"EvoPolicyGym gives the coding agent and the Policy different roles."}),"\n",(0,s.jsxs)(n.p,{children:["The coding agent is the outer policy engineer and optimizer. It reads\ninstructions and public Feedback, edits ",(0,s.jsx)(n.code,{children:"workspace/program/"}),", and decides when\nto submit another candidate. The Policy is the inner decision system. It\nreceives observations through a small ABI and returns Actions while an Episode\nis running."]}),"\n",(0,s.jsxs)(n.p,{children:["A Policy may retain state between ",(0,s.jsx)(n.code,{children:"act()"})," calls inside one Episode. Each new\nEpisode receives a fresh process and Policy instance, while cross-Episode\nimprovement is represented by a new Program."]}),"\n",(0,s.jsx)(n.p,{children:"EvoPolicyGym locates learning at the Program level between Episodes.\nEpisode-local state supports temporal behavior; Program revision captures\nlasting improvement. Each change therefore has a visible source snapshot and\na clear relationship to its Evaluation evidence."}),"\n",(0,s.jsx)(n.h2,{id:"make-the-artifact-first-class",children:"Make the artifact first-class"}),"\n",(0,s.jsxs)(n.p,{children:["The workspace supports live authoring. Every accepted submission turns its\ncurrent source tree into an immutable, content-addressed ",(0,s.jsx)(n.code,{children:"Program"}),". The\nEvaluation, Feedback, and artifacts belong to that exact snapshot, and the\nfinal result returns the retained Program selected from submitted candidates."]}),"\n",(0,s.jsx)(n.p,{children:"This choice makes a Run understandable as a sequence of authored artifacts:"}),"\n",(0,s.jsxs)(n.table,{children:[(0,s.jsx)(n.thead,{children:(0,s.jsxs)(n.tr,{children:[(0,s.jsx)(n.th,{children:"Object"}),(0,s.jsx)(n.th,{children:"Responsibility"})]})}),(0,s.jsxs)(n.tbody,{children:[(0,s.jsxs)(n.tr,{children:[(0,s.jsx)(n.td,{children:(0,s.jsx)(n.code,{children:"Program"})}),(0,s.jsx)(n.td,{children:"The executable Policy source being evaluated"})]}),(0,s.jsxs)(n.tr,{children:[(0,s.jsx)(n.td,{children:(0,s.jsx)(n.code,{children:"Submission"})}),(0,s.jsx)(n.td,{children:"One immutable Program, an explicit training-index selector, and committed Feedback"})]}),(0,s.jsxs)(n.tr,{children:[(0,s.jsx)(n.td,{children:(0,s.jsx)(n.code,{children:"Run"})}),(0,s.jsx)(n.td,{children:"A bounded sequence of submissions and a final handoff"})]}),(0,s.jsxs)(n.tr,{children:[(0,s.jsx)(n.td,{children:(0,s.jsx)(n.code,{children:"Validation"})}),(0,s.jsx)(n.td,{children:"Host-side selection among finished candidates"})]}),(0,s.jsxs)(n.tr,{children:[(0,s.jsx)(n.td,{children:(0,s.jsx)(n.code,{children:"Assessment"})}),(0,s.jsx)(n.td,{children:"Held-out measurement of the selected Program"})]})]})]}),"\n",(0,s.jsx)(n.p,{children:"The Program is the durable result. The agent transcript and process logs\nprovide supporting diagnostics."}),"\n",(0,s.jsx)(n.h2,{id:"let-each-benchmark-define-useful-feedback",children:"Let each Benchmark define useful Feedback"}),"\n",(0,s.jsx)(n.p,{children:"Different Environments expose different kinds of evidence. A control task may\nbenefit from state trajectories and termination causes. A card game may need\nround summaries, economy decisions, or compact replays."}),"\n",(0,s.jsx)(n.p,{children:"EvoPolicyGym standardizes the Feedback carrier while each Benchmark defines\nits useful domain content. Feedback always has a scalar score, and the\nBenchmark may add bounded public values and artifacts. The Benchmark also owns\nEpisode planning, Environment c
1onstruction, Action validation, and scoring."}),"\n",(0,s.jsx)(n.p,{children:"The Kernel owns what must remain consistent across Benchmarks: budgets,\nimmutable submissions, lifecycle ordering, publication, selection, records,\nand the Policy ABI. Environment packages remain independently installable and\ndepend only on the public authoring interface."}),"\n",(0,s.jsx)(n.p,{children:"This division lets the project grow like a Gym-style ecosystem while the\nKernel remains stable and domain-independent."}),"\n",(0,s.jsx)(n.h2,{id:"treat-evidence-access-as-part-of-the-experiment",children:"Treat evidence access as part of the experiment"}),"\n",(0,s.jsx)(n.p,{children:"A Run's submission limit, total Episode budget, fixed training-pool size, and\noptional per-Submission cap define its experimental condition. For example,\nsixteen submissions, forty-eight Episode units, and a pool of ninety-six\nEpisode identities grant forty-eight total observations selected from a wider\nset; the larger pool does not increase the interaction budget."}),"\n",(0,s.jsx)(n.p,{children:"The Host constructs this indexed pool before the agent starts. Each Submission\nnames a non-empty set of public Run-local indices. Reusing an index keeps its\nhidden Episode specification and Policy seed fixed, so two immutable Programs\ncan be compared on matched evidence. Every use still creates a fresh\nEnvironment and Policy runtime and consumes budget again. Actual seeds,\nscenarios, and pool construction remain Host-owned."}),"\n",(0,s.jsx)(n.p,{children:"Once an Evaluation begins, its reserved Episode allocation is consumed. Policy\nfailures and invalid Actions are reported as observed behavior, preserving the\nexact semantics of the submitted Program. Feedback maps every sanitized\nEpisode outcome back to its public index, while comparisons over different\nselectors remain unmatched evidence."}),"\n",(0,s.jsxs)(n.p,{children:["The agent uses public search Feedback to decide what to try next. When it\nfinishes, authority returns to the Host. Private Validation selects among the\nhanded-off candidates, and held-out Assessment measures the selected Program.\nOptimization Feedback closes at ",(0,s.jsx)(n.code,{children:"finish"}),"; selection and final measurement\nremain Host-side."]}),"\n",(0,s.jsx)(n.p,{children:"Keeping search, selection, and final measurement distinct makes the reported\nresult easier to interpret."}),"\n",(0,s.jsx)(n.h2,{id:"keep-the-kernel-focused",children:"Keep the Kernel focused"}),"\n",(0,s.jsx)(n.p,{children:"EvoPolicyGym is infrastructure with deliberately focused ownership."}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"Agent integrations translate a Host-owned task into a provider invocation."}),"\n",(0,s.jsx)(n.li,{children:"Benchmark distributions own domain semantics, dependencies, baselines,\nFeedback, and tests."}),"\n",(0,s.jsx)(n.li,{children:"The Kernel owns the shared evaluation and Program-evolution lifecycle."}),"\n",(0,s.jsx)(n.li,{children:"The Policy boundary carries bounded public values."}),"\n"]}),"\n",(0,s.jsx)(n.p,{children:"Today Codex is the first supported coding-agent integration, and local process\nexecution is the active backend. The contracts remain provider- and\nbackend-independent, preserving the meaning of Program, Submission,\nEvaluation, and Run as more integrations arrive."}),"\n",(0,s.jsx)(n.h2,{id:"what-evopolicygym-enables",children:"What EvoPolicyGym enables"}),"\n",(0,s.jsx)(n.p,{children:"EvoPolicyGym connects scalable interactive Environments, coding agents,\nversioned Programs, and verifiable Benchmark evidence. The Agent is the subject\nof study and training; the Program is the executable evidence it leaves\nbehind."}),"\n",(0,s.jsx)(n.h3,{id:"study-agents-that-evolve-strategy-systems",children:"Study agents that evolve strategy systems"}),"\n",(0,s.jsx)(n.p,{children:"With the Environment, initial Program, interaction budget, Feedback visibility,\nand selection rules held constant, repeated Runs can study:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Agent capability:"}
1)," which coding agent authors the strongest Policy system\nunder the same conditions?"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Improvement efficiency:"})," how much Environment interaction produces stable\nProgram improvement?"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Feedback value:"})," which traces, diagnostics, replays, and aggregate signals\nlead to effective revisions?"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Evolution dynamics:"})," how does Program structure change across Submissions,\nand which changes produce durable gains?"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Selection validity:"})," does the candidate chosen by Validation retain its\nadvantage in held-out Assessment?"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Policy-system design:"})," which state representations, rules, planners,\nmemories, and controllers do different agents encode?"]}),"\n",(0,s.jsxs)(n.li,{children:[(0,s.jsx)(n.strong,{children:"Scaling and generalization:"})," how do these results change across budgets,\nprofiles, seeds, task complexity, and Environment families?"]}),"\n"]}),"\n",(0,s.jsx)(n.p,{children:"The final score measures the Agent's selected artifact. The sequence of\nimmutable Programs, Feedback, artifacts, and outcomes explains how the Agent\nreached it."}),"\n",(0,s.jsx)(n.h3,{id:"train-coding-agents-with-interactive-environments",children:"Train coding agents with interactive Environments"}),"\n",(0,s.jsx)(n.p,{children:"An Environment and Benchmark together form a task generator, evidence\ngenerator, and verifier. Profiles, scenarios, and seeds create task variation;\nProgram evaluations, public Feedback, diagnostics, and held-out outcomes\nprovide training signals."}),"\n",(0,s.jsx)(n.p,{children:"The same Environment ecosystem can support:"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:"coding-agent RL and RLVR using Program evaluation and held-out performance as\nverifiable outcomes;"}),"\n",(0,s.jsx)(n.li,{children:"SFT from successful long-horizon Agent trajectories;"}),"\n",(0,s.jsx)(n.li,{children:"agent distillation from observable evolution records: task context, public\nFeedback, Program changes, Submissions, and outcomes;"}),"\n",(0,s.jsx)(n.li,{children:"rejection sampling of high-quality Agent trajectories based on their final\nartifacts and results;"}),"\n",(0,s.jsx)(n.li,{children:"curriculum learning across task profiles and difficulty;"}),"\n",(0,s.jsx)(n.li,{children:"process supervision from intermediate failures, revisions, and evaluations."}),"\n"]}),"\n",(0,s.jsx)(n.p,{children:"An Agent evolution trajectory spans the full task: understanding the\nEnvironment, authoring a Program, reading Feedback, diagnosing behavior,\nrevising the strategy system, submitting candidates, and completing the final\nhandoff. These long-horizon records provide training material for coding agents\nand policy-engineering agents."}),"\n",(0,s.jsx)(n.pre,{children:(0,s.jsx)(n.code,{className:"language-text",children:"Environment + Benchmark\n \u2502\n \u25bc\nAgent authors and revises a Program\n \u2502\n \u251c\u2500\u2500 evolution trajectory \u2500\u2500\u2500\u2500\u2500\u25b6 Agent SFT / distillation\n \u2514\u2500\u2500 evaluation outcomes \u2500\u2500\u2500\u2500\u2500\u2500\u25b6 Agent RL / RLVR\n"})}),"\n",(0,s.jsx)(n.p,{children:"The Kernel provides the common task, Evaluation, Run, and evidence contracts.\nDataset exporters and training systems can turn retained Agent trajectories\ninto SFT, RL, and distillation data, then return trained agents for held-out\nmeasurement. In this way, the Environment catalog is both an Agent Benchmark\nsurface and a scalable source of verifiable long-horizon experience."}),"\n",(0,s.jsx)(n.h2,{id:"continue-reading",children:"Continue reading"}),"\n",(0,s.jsxs)(n.ul,{children:["\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/docs/concepts/",children:"Core concepts \u2192"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/docs/evaluation/",children:"Evaluation and Runs \u2192"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/environments/",children:"Environment catalog \u2192"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"/results/",children:"Core16 results \u2192"})}),"\n",(0,s.jsx)(n.li,{children:(0,s.jsx)(n.a,{href:"https://arxiv.org/abs/2607.02440",children:"Paper \u2197"})}),"\n"]})]})}function h(e={}){const{wrapper:n}={...(0,a.R)(),...e.components};return n?(0,s.jsx)(n,{...e,children:(0,s.jsx)(l,{...e})}):l(e)}},8453(e,n,t){t.d(n,{R:()=>r,x:()=>o});var i=t(6540);const s={},a=i.createContext(s);function r(e){const n=i.useContext(a);return i.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function o(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:r(e.components),i.createElement(a.Provider,{value:n},e.children)}},3659(e){e.exports=JSON.parse('{"permalink":"/EvoPolicyGym/blog/designing-evopolicygym","source":"@site/blog/designing-evopolicygym.md","title":"EvoPolicyGym: environments for agents that build strategy systems","de
1scription":"Why EvoPolicyGym uses interactive Environments and executable Programs to study and train coding agents.","date":"2026-07-27T00:00:00.000Z","tags":[{"inline":true,"label":"Design","permalink":"/EvoPolicyGym/blog/tags/design"},{"inline":true,"label":"Motivation","permalink":"/EvoPolicyGym/blog/tags/motivation"},{"inline":true,"label":"Architecture","permalink":"/EvoPolicyGym/blog/tags/architecture"}],"readingTime":7.11,"hasTruncateMarker":true,"authors":[{"name":"EvoPolicyGym contributors","title":"Research and engineering team","url":"https://github.com/Linzwcs/EvoPolicyGym","key":"evopolicygym","page":null}],"frontMatter":{"locale":"en","page":"designing-evopolicygym","title":"EvoPolicyGym: environments for agents that build strategy systems","description":"Why EvoPolicyGym uses interactive Environments and executable Programs to study and train coding agents.","lead":"A coding agent studies an Environment, learns from bounded Feedback, and turns its conclusions into an executable strategy system.","publishedAt":"2026-07-27","date":"2026-07-27","authors":["evopolicygym"],"tags":["Design","Motivation","Architecture"],"status":"published"},"unlisted":false,"prevItem":{"title":"Letting Coding Agents Build Strategy Systems for Balatro","permalink":"/EvoPolicyGym/blog/balatro-policy-evolution"}}')}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.