1"use strict";(self.webpackChunkevopolicygym_site=self.webpackChunkevopolicygym_site||[]).push([[4e3],{7515(e,n,i){i.r(n),i.d(n,{assets:()=>c,contentTitle:()=>d,default:()=>h,frontMatter:()=>r,metadata:()=>s,toc:()=>a});const s=JSON.parse('{"id":"concepts","title":"Core concepts","description":"The domain model and trust boundaries of EvoPolicyGym 0.3.","source":"@site/docs/concepts.md","sourceDirName":".","slug":"/concepts","permalink":"/EvoPolicyGym/docs/concepts","draft":false,"unlisted":false,"tags":[],"version":"current","lastUpdatedAt":1786448798000,"frontMatter":{"locale":"en","page":"concepts","section":"core","title":"Core concepts","navTitle":"Core concepts","description":"The domain model and trust boundaries of EvoPolicyGym 0.3.","lead":"Programs are immutable, Evaluations are bounded, Feedback is committed, and each Episode receives a fresh Policy lifecycle.","index":"D2","order":2,"docsVersion":"v0.3","status":"draft"},"sidebar":"docsSidebar","previous":{"title":"Getting started","permalink":"/EvoPolicyGym/docs/getting-started"},"next":{"title":"Programs","permalink":"/EvoPolicyGym/docs/programs"}}');var o=i(4848),t=i(8453);const r={locale:"en",page:"concepts",section:"core",title:"Core concepts",navTitle:"Core concepts",description:"The domain model and trust boundaries of EvoPolicyGym 0.3.",lead:"Programs are immutable, Evaluations are bounded, Feedback is committed, and each Episode receives a fresh Policy lifecycle.",index:"D2",order:2,docsVersion:"v0.3",status:"draft"},d=void 0,c={},a=[{value:"Domain vocabulary",id:"domain-vocabulary",level:2},{value:"Evaluation lifecycle",id:"evaluation-lifecycle",level:2},{value:"Program-evolution lifecycle",id:"program-evolution-lifecycle",level:2},{value:"Trust boundary",id:"trust-boundary",level:2},{value:"Failure ownership",id:"failure-ownership",level:2},{value:"Package boundaries",id:"package-boundaries",level:2},{value:"Next",id:"next",level:2}];function l(e){const n={a:"a",code:"code",h2:"h2",li:"li",p:"p",pre:"pre",table:"table",tbody:"tbody",td:"td",th:"th",thead:"thead",tr:"tr",ul:"ul",...(0,t.R)(),...e.components};return(0,o.jsxs)(o.Fragment,{children:[(0,o.jsx)(n.h2,{id:"domain-vocabulary",children:"Domain vocabulary"}),"\n",(0,o.jsxs)(n.table,{children:[(0,o.jsx)(n.thead,{children:(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.th,{children:"Value"}),(0,o.jsx)(n.th,{children:"Meaning"})]})}),(0,o.jsxs)(n.tbody,{children:[(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"Program"})}),(0,o.jsx)(n.td,{children:"A detached, immutable, content-addressed snapshot of one Policy source directory."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"Episode"})}),(0,o.jsx)(n.td,{children:"One trusted scenario, one fresh Environment, and one fresh Policy process and instance."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"Evaluation"})}),(0,o.jsx)(n.td,{children:"One Program evaluated over a finite deterministic Episode plan."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"Feedback"})}),(0,o.jsx)(n.td,{children:"A Benchmark-defined public projection with one scalar score, bounded content, and optional artifacts."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"Submission"})}),(0,o.jsx)(n.td,{children:"One Program and the committed Feedback produced when a Coding Agent requests Evaluation."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"ProgramEvolutionRun"})}),(0,o.jsx)(n.td,{children:"One bounded outer loop in which a Coding Agent edits Programs, submits candidates, reads Feedback, and hands published candidates to Host-side selection."})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:(0,o.jsx)(n.code,{children:"Experiment"})}),(0,o.jsx)(n.td,{children:"Reserved for a future collection of comparable Runs."})]})]})]}),"\n",(0,o.jsxs)(n.p,{children:["The public SDK uses ",(0,o.jsx)(n.code,{children:"Program"}),", not ",(0,o.jsx)(n.code,{children:"ProgramVersion"}),". A Program retains no Host\nsource path and cannot change when the caller later edits its original\ndirectory."]}),"\n",(0,o.jsx)(n.h2,{id:"evaluation-lifecycle",children:"Evaluation lifecycle"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-text",children:"Program\n \u2193\ndeterministic Episode plan\n \u2193\nfresh Environment + fresh Policy process\n \u2193\nunmodified Actions and trusted Steps\n \u2193\nsanitized Episode summaries\n \u2193\nBenchmark-defined Feedback\n"})}),"\n",(0,o.jsxs)(n.p,{children:["Policy state may persist between ",(0,o.jsx)(n.code,{children:"act()"})," calls in one Episode. It never\npersists into another Episode. Cross-Episode improvement happens only when the\nouter Coding Agent authors a new Program."]}),"\n",(0,o.jsx)(n.h2,{id:"program-evolution-lifecycle",children:"Program-evolution lifecycle"}),"\n",(0,o.jsx)(n.pre,{children:(0,o.jsx)(n.code,{className:"language-text",children:"initial Program\n \u2193\nHost fixes indexed training Episode pool\n \u2193\nCoding Agent edits workspace/program/\n \u2193\nSubmission(selector) \u2192 fresh runtimes \u2192 committed Feedback\n \u2193\nCoding Agent reads
1indexed outcomes in workspace/feedback/\n \u2193\nnext Program or finish(candidate IDs)\n \u2193\nHost-side final selection\n"})}),"\n",(0,o.jsxs)(n.p,{children:["A ",(0,o.jsx)(n.code,{children:"RunConfig"})," fixes the split, maximum submissions, total Episode budget,\nfixed training Episode-pool size, optional per-Submission Episode cap, seed,\nand timeouts before the Agent starts. Pool size and budget are different\nlimits: the pool controls how many Episode identities are available, while the\nbudget controls the total number of selected indices across all Submissions.\nThe pool size defaults to the total budget."]}),"\n",(0,o.jsxs)(n.p,{children:["The Agent selects public Run-local indices from that pool. The same index\npreserves its hidden Episode specification and Policy seed across Submissions,\nwhich supports matched Program comparisons. Every use still creates fresh\nruntime state and consumes budget again. Pool indices are experimental handles,\nnot Environment seeds, and neither the Agent nor the Policy receives the\nunderlying seed. The optional per-Submission cap defaults to ",(0,o.jsx)(n.code,{children:"None"}),"."]}),"\n",(0,o.jsx)(n.h2,{id:"trust-boundary",children:"Trust boundary"}),"\n",(0,o.jsxs)(n.table,{children:[(0,o.jsx)(n.thead,{children:(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.th,{children:"Trusted Host and Benchmark own"}),(0,o.jsx)(n.th,{children:"Policy can observe"})]})}),(0,o.jsxs)(n.tbody,{children:[(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:"Environment parameter selection"}),(0,o.jsxs)(n.td,{children:["Public ",(0,o.jsx)(n.code,{children:"environment_parameters"})," fixed before Evaluation"]})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:"Episode scenario, Environment seed, and pool mapping"}),(0,o.jsxs)(n.td,{children:[(0,o.jsx)(n.code,{children:"PolicyContext"})," without a pool index or Case identity"]})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:"Environment state and transitions"}),(0,o.jsx)(n.td,{children:"Public observations"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:"Action validation"}),(0,o.jsx)(n.td,{children:"Its own Episode-local state"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:"Rewards, scoring, and private metrics"}),(0,o.jsx)(n.td,{children:"Committed public Feedback only"})]}),(0,o.jsxs)(n.tr,{children:[(0,o.jsx)(n.td,{children:"Run budget and publication"}),(0,o.jsx)(n.td,{children:"No Host path, credential, scorer, or runtime evidence"})]})]})]}),"\n",(0,o.jsxs)(n.p,{children:["The Policy boundary carries only bounded ",(0,o.jsx)(n.code,{children:"PolicyValue"})," data. Paths, file\ndescriptors, credentials, arbitrary Python objects, and pickle graphs never\ncross it."]}),"\n",(0,o.jsx)(n.h2,{id:"failure-ownership",children:"Failure ownership"}),"\n",(0,o.jsx)(n.p,{children:"Policy exceptions, timeouts, protocol errors, and invalid Actions become\nsanitized Policy failures. Invalid Actions are never clipped, repaired,\nsampled, or replaced."}),"\n",(0,o.jsx)(n.p,{children:"Trusted Environment, Benchmark, process-control, and cleanup faults abort the\nEvaluation. They never become Policy penalties."}),"\n",(0,o.jsx)(n.h2,{id:"package-boundaries",children:"Package boundaries"}),"\n",(0,o.jsxs)(n.p,{children:["The base ",(0,o.jsx)(n.code,{children:"evopolicygym"})," wheel owns the portable Kernel. Independent Benchmark\ndistributions depend only on public SDK facades and ",(0,o.jsx)(n.code,{children:"evopolicygym.authoring"}),".\nThe Kernel does not import those distributions."]}),"\n",(0,o.jsx)(n.p,{children:"The optional Firecracker groundwork is a separate product and does not make a\nformal or isolated execution profile available."}),"\n",(0,o.jsx)(n.h2,{id:"next",children:"Next"}),"\n",(0,o.jsxs)(n.ul,{children:["\n",(0,o.jsx)(n.li,{children:(0,o.jsx)(n.a,{href:"/EvoPolicyGym/docs/policy",children:"Policy ABI \u2192"})}),"\n",(0,o.jsx)(n.li,{children:(0,o.jsx)(n.a,{href:"/EvoPolicyGym/docs/evaluation",children:"Evaluation \u2192"})}),"\n",(0,o.jsx)(n.li,{children:(0,o.jsx)(n.a,{href:"/EvoPolicyGym/docs/runs",children:"Coding Agent Runs \u2192"})}),"\n",(0,o.jsx)(n.li,{children:(0,o.jsx)(n.a,{href:"/EvoPolicyGym/docs/authoring",children:"Benchmark authoring \u2192"})}),"\n",(0,o.jsx)(n.li,{children:(0,o.jsx)(n.a,{href:"/EvoPolicyGym/docs/runtime",children:"Execution and safety \u2192"})}),"\n"]})]})}function h(e={}){const{wrapper:n}={...(0,t.R)(),...e.components};return n?(0,o.jsx)(n,{...e,children:(0,o.jsx)(l,{...e})}):l(e)}},8453(e,n,i){i.d(n,{R:()=>r,x:()=>d});var s=i(6540);const o={},t=s.createContext(o);function r(e){const n=s.useContext(t);return s.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function d(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(o):e.components||o:r(e.components),s.createElement(t.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.