1import{emptyPreview as O,textBlob as T,replaceExtension as C}from"./util-CkZ8KZK2.js";import{l as N}from"./load-Wi9neac3.js";import{m as U,a as D,b as S,p as E,t as p,q as v,r as F,c as I,e as h}from"./describe-DxiM0Ieo.js";import"./vite-preload-ckwbz45p.js";import"./duckdbLite-BZJanxnx.js";import"./duckdb-wasm-Dy5MCI5t.js";import"./vendor-arrow-A8Yy6zVL.js";import"./vendor-BeeS_Sk0.js";import"./cjs-helpers-D6-XlEtG.js";import"./format-E5uFfApC.js";import"./xlsxSheets-BENGaTjw.js";import"./vendor-xlsx-DCwg6k6L.js";import"./columnRoles-C34GwJD6.js";async function L(t,a){var g,$,b;const e=U(t.columns),n=D(t.columns),o=S(t.columns),s=(g=e[0])==null?void 0:g.name,r=($=n[0])==null?void 0:$.name,m=await a(E(t)),i=await a(`SELECT COUNT(*) AS "rows" FROM ${t.table}`),d=p((b=i.rows[0])==null?void 0:b[0])??0,c=[],u=async(x,k)=>{try{c.push(await F(x,k,t,a))}catch{}};s&&(await u(`What is the total ${s}?`,{kind:"aggregate",aggregate:"sum",metric:s,filters:[]}),await u(`What is the average ${s}?`,{kind:"aggregate",aggregate:"avg",metric:s,filters:[]})),s&&r&&await u(`Total ${s} by ${r}`,{kind:"aggregate",aggregate:"sum",metric:s,groupBy:r,direction:"desc",share:!0,limit:10,filters:[]}),s&&o[0]&&await u(`${s} over time`,{kind:"trend",aggregate:"sum",metric:s,dateColumn:o[0].name,grain:(o[0].spanDays??400)<=90?"day":(o[0].spanDays??400)<=1460?"month":"year",grainInferred:!0,filters:[]}),await u("Show missing values",{kind:"missing",filters:[]}),await u("Are there duplicate rows?",{kind:"duplicates",filters:[]});const w=[];for(const x of n.slice(0,3))try{const k=await a(`SELECT ${v(x.name)} AS ${v(x.name)}, COUNT(*) AS "rows", ROUND(100.0 * COUNT(*) / NULLIF(SUM(COUNT(*)) OVER (), 0), 2) AS "share_pct" FROM ${t.table} GROUP BY 1 ORDER BY "rows" DESC NULLS LAST LIMIT 8`);w.push({column:x.name,table:k})}catch{}const l=M(t,d,c,m,s);return{rows:d,columns:t.columns.length,kpis:l,insights:A(c,s,r),distributions:w,profile:m,quality:q(m,c,d),answers:c}}function f(t,a){return t.find(a)}function M(t,a,e,n,o){var w;const s=[{label:"Rows",value:a.toLocaleString("en-US")},{label:"Columns",value:String(t.columns.length)}],r=f(e,l=>l.intent.kind==="aggregate"&&l.intent.aggregate==="sum"&&!l.intent.groupBy);r&&o&&s.push({label:`Total ${o}`,value:r.headline,note:r.exact});const m=f(e,l=>l.intent.kind==="aggregate"&&l.intent.aggregate==="avg");m&&o&&s.push({label:`Average ${o}`,value:m.headline,note:m.exact});const i=n.columns.indexOf("missing"),d=n.rows.reduce((l,g)=>l+(p(g[i])??0),0);s.push({label:"Missing values",value:I(d),note:d===0?"every cell filled":`${(d/Math.max(1,a*t.columns.length)*100).toFixed(2)}% of cells`});const c=f(e,l=>l.intent.kind==="duplicates");if(c){const l=c.table.columns.indexOf("extra_rows"),g=p((w=c.table.rows[0])==null?void 0:w[l])??0;s.push({label:"Duplicate rows",value:h(g),note:g===0?"no exact repeats":"extra copies of an identical row"})}const u=S(t.columns)[0];if(u){const l=n.rows.find(b=>b[0]===u.name),g=(l==null?void 0:l[n.columns.indexOf("min")])??"",$=(l==null?void 0:l[n.columns.indexOf("max")])??"";g&&$&&s.push({label:"Date range",value:`${g.slice(0,10)} to ${$.slice(0,10)}`,note:u.name})}return s}function A(t,a,e){const n=[],o=f(t,i=>i.intent.kind==="aggregate"&&!!i.intent.groupBy);if(o&&a&&e){const i=o.table.columns.length-(o.intent.kind==="aggregate"&&o.intent.share?2:1),d=o.table.columns.indexOf("share_pct"),c=o.table.rows[0];if(c){const u=d>=0?c[d]:void 0;n.push(`${c[0]||"(blank)"} leads ${e} with ${h(p(c[i])??0)} of ${a}${u?`, ${u}% of the total`:""}.`)}if(o.table.rows.length>1){const u=o.table.rows[o.table.rows.length-1];n.push(`The gap between the leading and trailing ${e} is ${h(Math.abs((p(c==null?void 0:c[i])??0)-(p(u==null?void 0:u[i])??0)))} of ${a}.`)}}const s=f(t,i=>i.intent.kind==="trend");s&&n.push(s.detail);const r=f(t,i=>i.intent.kind==="duplicates");r&&n.length<3&&n.push(r.detail);const m=f(t,i=>i.intent.kind==="missing");return m&&n.length<3&&n.push(m.detail),n.slice(0,3)}function q(t,a,e){var m;const n=[],o=t.columns.indexOf("missing"),s=t.columns.indexOf("distinct");for(const i of t.rows){const d=p(i[o])??0;d>0&&n.push(`${i[0]} is missing ${h(d)} of ${h(e)} values (${(d/Math.max(1,e)*100).toFixed(1)}%).`)}for(const i of t.rows){const d=p(i[s])??0;e>1&&d===1&&n.push(`${i[0]} holds one value for every row, so it cannot separate anything.`)}const r=f(a,i=>i.intent.kind==="duplicates");if(r){const i=p((m=r.table.rows[0])==null?void 0:m[r.table.columns.indexOf("extra_rows")])??0;i>0&&n.push(`${h(i)} rows are exact copies of another row.`)}return n.length===0&&n.push("No missing values, no constant columns and no repeated rows."),n.slice(0,8)}function y(t,a=30){if(t.columns.length===0||t.rows.length===0)return"";const e=`| ${t.columns.join(" | ")} |`,n=`| ${t.columns.map(()=>"---").join(" | ")} |`,o=t.rows.slice(0,a).map(s=>`| ${t.columns.map((r,m)=>(s[m]??"").replace(/\|/g,"\\|")).join(" | ")} |`).join(` 2`);return[e,n,o].join(` 3`)}function R(t,a){const e=[];e.push(`# Summary of ${a}
3`,""),e.push(`${t.rows.toLocaleString("en-US")} rows, ${t.columns} columns. Every figure below was calculated locally with DuckDB; nothing was uploaded and nothing was estimated.`,""),e.push("## Headline figures","");for(const n of t.kpis)e.push(`- **${n.label}:** ${n.value}${n.note?` (${n.note})`:""}`);if(e.push(""),t.insights.length>0){e.push("## What the numbers say","");for(const n of t.insights)e.push(`- ${n}`);e.push("")}e.push("## Data quality","");for(const n of t.quality)e.push(`- ${n}`);e.push(""),e.push("## Columns",""),e.push(y(t.profile,100),"");for(const n of t.distributions)e.push(`## Values of ${n.column}`,""),e.push(y(n.table),"");return e.push("---",""),e.push("Produced by the deterministic Ask tier in ExploreMyData. No model wrote these numbers. To ask a question of your own against the same file, open /tools/ask-csv."),e.join(` 4`)}const X=async t=>{const a=t.file??(t.text!==void 0?new File([t.text],t.filename||"data.csv",{type:"text/csv"}):void 0);if(!a)throw new Error("Pick a file to summarize.");const e=await N(a,t.ctx.onProgress);t.ctx.onProgress("Calculating the summaryâ¦");const n=await L(e.schema,e.run),o=R(n,a.name),s=[`${n.rows.toLocaleString("en-US")} rows`,`${n.columns} columns`,...n.kpis.slice(2).map(r=>`${r.label}: ${r.value}`),...n.insights];return{status:"ok",output:{filename:C(t.filename,"md"),blob:T(o,"text/markdown"),text:o,copyKind:"text",preview:O(),previewText:o,summary:s,summaryStyle:"list",warnings:[]}}};export{X as default,R as summaryToMarkdown};
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.