PageSourceSearch

https://erpl.io/assets/js/d4713f34.8e7ae464.js

js erpl.io collected 2026-09-25 20:08:21 UTC 18,259 bytes, 1 lines download raw bytes

1"use strict";(self.webpackChunkerpl_docusaurus=self.webpackChunkerpl_docusaurus||[]).push([[9265],{28453:(e,a,t)=>{t.d(a,{R:()=>r,x:()=>i});var n=t(96540);const s={},o=n.createContext(s);function r(e){const a=n.useContext(o);return n.useMemo(function(){return"function"==typeof e?e(a):{...a,...e}},[a,e])}function i(e){let a;return a=e.disableParentContext?"function"==typeof e.components?e.components(s):e.components||s:r(e.components),n.createElement(o.Provider,{value:a},e.children)}},33596:(e,a,t)=>{t.d(a,{A:()=>n});const n=t.p+"assets/images/hero-e13182962dc6fb7e3ddc3897b2ddef60.jpg"},38960:(e,a,t)=>{t.d(a,{A:()=>n});const n=t.p+"assets/images/architecture-data-out-7522c92f2baed3f010dcff60a50f5aaf.svg"},48637:(e,a,t)=>{t.r(a),t.d(a,{assets:()=>l,contentTitle:()=>i,default:()=>h,frontMatter:()=>r,metadata:()=>n,toc:()=>d});var n=t(87619),s=t(74848),o=t(28453);const r={slug:"real-time-sap-for-ai-part-1",title:"Your SAP Data Is Already Real-Time \u2014 You Just Need to Stop Caching It",authors:["joachim-rosskopf"],tags:["erpl","erpl-web","anofox-tabular","sap","real-time","ai","data-quality","duckdb"],date:new Date("2026-05-22T00:00:00.000Z"),image:"./hero.jpg",description:"Most SAP analytics pipelines pay a 14-hour latency tax for no good reason. Here is how to pull live SAP data into a DuckDB-native AI stack \u2014 no warehouse required."},i=void 0,l={image:t(78787).A,authorsImageUrls:[void 0]},d=[{value:"Why "real-time" usually means "20 minutes ago"",id:"why-real-time-usually-means-20-minutes-ago",level:2},{value:"Layer 1: ERPL pulls SAP ECC / S/4HANA live",id:"layer-1-erpl-pulls-sap-ecc--s4hana-live",level:2},{value:"Layer 2: ERPL-Web closes the cloud gap",id:"layer-2-erpl-web-closes-the-cloud-gap",level:2},{value:"Layer 3: anofox-tabular gates the stream",id:"layer-3-anofox-tabular-gates-the-stream",level:2},{value:"What this looks like in production",id:"what-this-looks-like-in-production",level:2},{value:"Next week, in Part 2",id:"next-week-in-part-2",level:2}];function c(e){const a={a:"a",code:"code",em:"em",h2:"h2",img:"img",p:"p",pre:"pre",strong:"strong",...(0,o.R)(),...e.components};return(0,s.jsxs)(s.Fragment,{children:[(0,s.jsx)(a.p,{children:"Picture a forecasting team at an industrial-controls Mittelstand. Their job for the next quarter: launch an AI-driven inventory and demand-forecasting tool that the operations team will actually trust enough to act on. They have the model. They have the GPUs. They cannot ship."}),"\n",(0,s.jsx)(a.p,{children:"The reason: their nightly ETL job runs at 02:00 and finishes around 04:00, by which point the model trains on data that's already half a day old. By the time the forecast reaches the procurement system at 08:00, it's reasoning about a world that existed fourteen hours ago. When the CFO asks why the model recommended ordering more of an item that just sold out at lunchtime, nobody has a good answer \u2014 because the data the model saw never knew the lunchtime existed."}),"\n",(0,s.jsx)(a.p,{children:"The fix isn't a better ETL. The fix is removing the ETL."}),"\n",(0,s.jsx)(a.p,{children:(0,s.jsx)(a.img,{alt:"Data flowing out of SAP into a DuckDB-native AI stack",src:t(33596).A+"",width:"1376",height:"768"})}),"\n",(0,s.jsx)(a.h2,{id:"why-real-time-usually-means-20-minutes-ago",children:'Why "real-time" usually means "20 minutes ago"'}),"\n",(0,s.jsx)(a.p,{children:"The default SAP analytics architecture is three boxes: SAP, a warehouse, a BI layer on top of the warehouse. Each box exists because someone, at some point, decided the next box couldn't talk to the previous one directly. That assumption was true in 2008. It's been getting less true every year since DuckDB shipped."}),"\n",(0,s.jsxs)(a.p,{children:["When you actually trace the latency, the SAP system isn't the slow part. RFC reads on a healthy ECC system come back in seconds. ODP delta extracts on a properly subscribed data source return whatever changed since the last cursor position, often in under a minute. The latency lives in the ",(0,s.jsx)(a.em,{children:"middle box"})," \u2014 the warehouse you assumed you needed because everyone else has one."]}),"\n",(0,s.jsx)(a.p,{children:"What if you didn't have one? What if your AI agents and your forecasting models queried SAP through a thin SQL layer that fans out to RFC, ODP, Datasphere, and Business Central \u2014 and only materialized when it actually helped? That's the stack we built ERPL for."}),"\n",(0,s.jsx)(a.p,{children:(0,s.jsx)(a.img,{alt:"Real-time SAP data architecture: ERPL and ERPL-Web stream into DuckDB, anofox-tabular gates the stream",src:t(38960).A+"",width:"1276",height:"396"})}),"\n",(0,s.jsx)(a.h2,{id:"layer-1-erpl-pulls-sap-ecc--s4hana-live",children:"Layer 1: ERPL pulls SAP ECC / S/4HANA live"}),"\n",(0,s.jsx)(a.p,{children:"The starting point is a DuckDB secret. ERPL stores SAP credentials the same way DuckDB stores S3 or Postgres credentials \u2014 once, declaratively, with the connection metadata attached. From then on, every ERPL function uses the secret implicitly."}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"INSTALL 'erpl' FROM 'http://get.erpl.io';\nLOAD 'erpl';\n\nCREATE SECRET sap (\n  TYPE sap_rfc,\n  ASHOST 'sap-prod.example.com',\n  SYSNR  '00',\n  CLIENT '100',\n  USER   'erpl_reader',\n  PASSWD 'redacted',\n  LANG   'EN'\n);\n\nPRAGMA sap_rfc_ping;\n"})}),"\n",(0,s.jsxs)(a.p,{children:[(0,s.jsx)(a.code,{children:"PRAGMA sap_rfc_ping"})," raises immediately if anything is misconfigured, so the credentials check is one statement, not a 200-line connection class."]}),"\n",(0,s.jsxs)(a.p,{children:["With the secret in place, you read SAP tables as if they were DuckDB tables \u2014 including pushdown for ",(0,s.jsx)(a.code,{children:"WHERE"})," and column selection. The forecasting team needs recent sales-order headers (",(0,s.jsx)(a.code,{children:"VBAK"}),") for one of their sales organizations. ERPL's ",(0,s.jsx)(a.code,{children:"sap_read_table"})," has predicate pushdown enabled, so a regular SQL ",(0,s.jsx)(a.code,{children:"WHERE"})," clause is pushed all the way down into the RFC call. You don't write a special filter parameter; you write SQL:"]}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"SELECT VBELN, ERDAT, KUNNR, NETWR, WAERK\nFROM sap_read_table('VBAK', MAX_ROWS => 5000)\nWHERE ERDAT >= DATE '2026-05-01'\n  AND VKORG = '1000';\n"})}),"\n",(0,s.jsxs)(a.p,{children:["That executes in seconds against a live ECC instance. The ",(0,s.jsx)(a.code,{children:"MAX_ROWS"})," is a guardrail; pushdown narrows the read further. For column-heavy tables you can also pass ",(0,s.jsx)(a.code,{children:"COLUMNS 
1=> ['VBELN', 'ERDAT', 'KUNNR']"})," and let RFC drop the rest before bytes leave SAP."]}),"\n",(0,s.jsxs)(a.p,{children:["For incremental loads, RFC is the wrong tool \u2014 that's what ODP is for. ODP cursors are server-side state: SAP remembers what you've already seen per subscriber. The first call to ",(0,s.jsx)(a.code,{children:"sap_odp_read_full"})," for a given (context, data source) returns a FULL snapshot and opens the cursor. Every subsequent call returns only the DELTA since the last position. There is no per-call mode flag and no manual cursor management \u2014 the server tracks it."]}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"-- First call: SAP returns the full snapshot and opens the cursor.\n-- Every subsequent call: only the delta since the last position.\nSELECT *\nFROM sap_odp_read_full('BW', 'VBAK$F', threads => 4);\n"})}),"\n",(0,s.jsx)(a.p,{children:"Run that on a cron every few minutes and you have a continuously-fresh sales-order stream landing directly in DuckDB, with zero ETL code between SAP and your model."}),"\n",(0,s.jsx)(a.h2,{id:"layer-2-erpl-web-closes-the-cloud-gap",children:"Layer 2: ERPL-Web closes the cloud gap"}),"\n",(0,s.jsx)(a.p,{children:"Not every SAP system is on-prem. Some of the forecasting team's data lives in Microsoft Dynamics 365 Business Central (the spin-off subsidiary uses BC instead of S/4HANA), and the corporate planning numbers are in SAP Datasphere. We built ERPL-Web so the same DuckDB session can reach both without changing the SQL contract."}),"\n",(0,s.jsx)(a.p,{children:"For Business Central, attach the company as if it were a database:"}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"INSTALL 'erpl_web' FROM 'http://get.erpl.io';\nLOAD 'erpl_web';\n\nATTACH 'CRONUS Germany AG' AS bc (TYPE business_central);\n\n-- Now the BC entities look like regular tables\nSELECT no, name, balance_due\nFROM bc.customer\nWHERE balance_due > 0\nORDER BY balance_due DESC\nLIMIT 20;\n"})}),"\n",(0,s.jsxs)(a.p,{children:[(0,s.jsx)(a.code,{children:"ATTACH"})," does the OData V4 metadata discovery once. After that, every entity in the company shows up as a DuckDB table, with predicate pushdown via the OData ",(0,s.jsx)(a.code,{children:"$filter"})," clause."]}),"\n",(0,s.jsx)(a.p,{children:"For Datasphere, the read function speaks the same SQL dialect:"}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"SELECT *\nFROM datasphere_read_relational('PLANNING_SPACE', 'V_DEMAND_FORECAST')\nWHERE month = '2026-05';\n"})}),"\n",(0,s.jsxs)(a.p,{children:["The forecasting team doesn't care that one read goes through SAP-managed OAuth2 and the other goes through RFC over an SSH tunnel. Same ",(0,s.jsx)(a.code,{children:"SELECT *"}),", same DuckDB result set. That uniformity is what lets the data plane disappear."]}),"\n",(0,s.jsx)(a.h2,{id:"layer-3-anofox-tabular-gates-the-stream",children:"Layer 3: anofox-tabular gates the stream"}),"\n",(0,s.jsxs)(a.p,{children:["Live data is fast, but live data is also dirty. The forecasting model needs the dirty rows filtered out ",(0,s.jsx)(a.em,{children:"before"})," it sees them, not flagged in a Monday-morning report. anofox-tabular is a DuckDB extension that runs the validation in SQL, vectorized, in-process \u2014 no external service call, no Python ML pipeline."]}),"\n",(0,s.jsxs)(a.p,{children:["Take the customer master. Two checks the team runs every load: are the customer email addresses syntactically valid, and are the German VAT numbers well-formed for the country they claim? Emails live in ",(0,s.jsx)(a.code,{children:"ADR6"}),", joined to ",(0,s.jsx)(a.code,{children:"KNA1"})," via ",(0,s.jsx)(a.code,{children:"ADRNR"}),", so the pattern is one CTE plus two calls:"]}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"LOAD anofox_tabular;\n\nWITH customers AS (\n  SELECT k.KUNNR, k.NAME1, k.LAND1, k.STCEG, e.SMTP_ADDR\n  FROM sap_read_table('KNA1') AS k\n  LEFT JOIN sap_read_table('ADR6') AS e\n    ON e.ADDRNUMBER = k.ADRNR\n  WHERE k.LAND1 = 'DE'\n)\nSELECT KUNNR, NAME1, SMTP_ADDR, STCEG\nFROM customers\nWHERE NOT email_is_valid(SMTP_ADDR)\n   OR NOT vat_is_valid(STCEG, 'DE');\n"})}),"\n",(0,s.jsxs)(a.p,{children:["That returns the rows that ",(0,s.jsx)(a.em,{children:"failed"})," validation \u2014 perfect for an alert table the data steward gets pinged on."]}),"\n",(0,s.jsx)(a.p,{children:"The more interesting catch is statistical. The forecasting model trains on order line values and quantities. A single rogue row \u2014 wrong net value for the quantity, or vice versa \u2014 can knock the model's RMSE up by an order of magnitude. anofox-tabular's isolation forest works directly against a DuckDB table:"}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"-- Persist the last week of orders so isolation_forest_mv can reference it by name\nCREATE OR REPLACE TABLE live_orders AS\nSELECT VBELN, POSNR, K
1DGRP, NETWR, KWMENG\nFROM sap_odp_read_full('BW', 'VBAP$F')\nWHERE ERDAT >= CURRENT_DATE - INTERVAL '7' DAY;\n\n-- Multi-column isolation forest. Positional args:\n--   (table_name, columns_csv, n_trees, sample_size, contamination, output_mode)\nSELECT row_id, anomaly_score\nFROM isolation_forest_mv(\n  'live_orders', 'NETWR,KWMENG',\n  100, 256, 0.005, 'scores'\n)\nWHERE is_anomaly\nORDER BY anomaly_score DESC\nLIMIT 20;\n"})}),"\n",(0,s.jsxs)(a.p,{children:["The first time we ran this against the Mittelstand's pre-production cutover data, the top anomaly was a single order line whose ",(0,s.jsx)(a.code,{children:"NETWR"})," was a thousand times too large for its ",(0,s.jsx)(a.code,{children:"KWMENG"})," \u2014 a unit-mismatch from a legacy ABAP report. Joining the ",(0,s.jsx)(a.code,{children:"row_id"})," back to ",(0,s.jsx)(a.code,{children:"live_orders"})," revealed the offending row also carried ",(0,s.jsx)(a.code,{children:"KDGRP = '9999'"}),", a placeholder value the same report had been writing for years. The forecasting model would have eaten both surprises without complaint and quietly skewed every prediction for that segment."]}),"\n",(0,s.jsx)(a.h2,{id:"what-this-looks-like-in-production",children:"What this looks like in production"}),"\n",(0,s.jsx)(a.p,{children:"You don't need a Kafka cluster for this. The whole stack runs in one DuckDB process. A typical hourly job:"}),"\n",(0,s.jsx)(a.pre,{children:(0,s.jsx)(a.code,{className:"language-sql",children:"-- 1. Pull the delta (advances the ODP cursor once)\nCREATE OR REPLACE TABLE orders_hourly AS\nSELECT *\nFROM sap_odp_read_full('BW', 'VBAP$F');\n\n-- 2. Score for anomalies (runs against the snapshot, not the cursor)\nCREATE OR REPLACE TABLE orders_flagged AS\nSELECT *\nFROM isolation_forest_mv(\n  'orders_hourly', 'NETWR,KWMENG',\n  100, 256, 0.005, 'scores'\n);\n\n-- 3. Hand the clean rows to the model\nCREATE OR REPLACE VIEW forecast_inputs AS\nSELECT o.*\nFROM orders_hourly AS o\nLEFT JOIN orders_flagged AS f ON f.row_id = o.rowid\nWHERE f.is_anomaly IS NOT TRUE;\n"})}),"\n",(0,s.jsxs)(a.p,{children:["Persist ",(0,s.jsx)(a.code,{children:"orders_hourly"})," and ",(0,s.jsx)(a.code,{children:"orders_flagged"})," through DuckLake snapshots if you want time-travel or audit. Latency budget on this stack against a real ECC system: under 90 seconds for an RFC table read, under 5 minutes for an ODP delta load that captures a quarter-day of order activity. No warehouse landing, no staging tables, no nightly window."]}),"\n",(0,s.jsx)(a.p,{children:(0,s.jsx)(a.img,{alt:"Latency comparison: nightly ETL vs ERPL direct",src:t(96532).A+"",width:"1320",height:"660"})}),"\n",(0,s.jsx)(a.h2,{id:"next-week-in-part-2",children:"Next week, in Part 2"}),"\n",(0,s.jsxs)(a.p,{children:["We have a fresh, validated view of SAP reality. The forecasting model could already train on it, and the dashboards could already read it. But there's a second consumer the team hasn't onboarded yet: the AI agent that's supposed to ",(0,s.jsx)(a.em,{children:"use"})," this data \u2014 and, occasionally, write back to SAP when it notices something wrong. That's where flAPI and erpl-adt come in."]}),"\n",(0,s.jsx)(a.p,{children:(0,s.jsx)(a.a,{href:"/blog/real-time-sap-for-ai-part-2",children:(0,s.jsx)(a.strong,{children:"Part 2 \u2014 The Day Our AI Agent Filed a Transport \u2192"})})})]})}function h(e={}){const{wrapper:a}={...(0,o.R)(),...e.components};return a?(0,s.jsx)(a,{...e,children:(0,s.jsx)(c,{...e})}):c(e)}},78787:(e,a,t)=>{t.d(a,{A:()=>n});const n=t.p+"assets/images/hero-e13182962dc6fb7e3ddc3897b2ddef60.jpg"},87619:e=>{e.exports=JSON.parse('{"permalink":"/blog/real-time-sap-for-ai-part-1","editUrl":"https://github.com/datazoode/erpl-landingpage/tree/main/blog/2026-05-22-real-time-sap-for-ai-part-1/index.md","source":"@site/blog/2026-05-22-real-time-sap-for-ai-part-1/index.md","title":"Your SAP Data Is Already Real-Time \u2014 You Just Need to Stop Caching It","description":"Most SAP analytics pipelines pay a 14-hour latency tax for no good reason. Here is how to pull live SAP data into a DuckDB-native AI stack \u2014 no warehouse required.","date":"2026-05-22T00:00:00.000Z","tags":[{"inline":false,"label":"ERPL","permalink":"/blog/tags/erpl","description":"ERPL extension tutorials"},{"inline":false,"label":"ERPL-Web","permalink":"/blog/tags/erpl-web","description":"ERPL-Web DuckDB extension for SAP cloud and SaaS integration"},{"inline":false,"label":"anofox-tabular","permalink":"/blog/tags/anofox-tabular","description":"anofox-tabular DuckDB extension for data validation, PII, and anomaly detection"}
1,{"inline":false,"label":"SAP","permalink":"/blog/tags/sap","description":"SAP system integration"},{"inline":false,"label":"Real-time","permalink":"/blog/tags/real-time","description":"Real-time data processing"},{"inline":false,"label":"AI","permalink":"/blog/tags/ai","description":"AI-powered development and automation"},{"inline":false,"label":"Data Quality","permalink":"/blog/tags/data-quality","description":"Data validation, anomaly detection, and quality gating"},{"inline":false,"label":"DuckDB","permalink":"/blog/tags/duckdb","description":"DuckDB database tutorials"}],"readingTime":7.72,"hasTruncateMarker":true,"authors":[{"name":"Joachim Rosskopf","title":"Co-Founder & CEO","url":"https://data-zoo.de","socials":{"linkedin":"https://www.linkedin.com/in/joachim-rosskopf/","github":"https://github.com/jrosskopf"},"imageURL":"/images/profiles/profile_jr.png","key":"joachim-rosskopf","page":null}],"frontMatter":{"slug":"real-time-sap-for-ai-part-1","title":"Your SAP Data Is Already Real-Time \u2014 You Just Need to Stop Caching It","authors":["joachim-rosskopf"],"tags":["erpl","erpl-web","anofox-tabular","sap","real-time","ai","data-quality","duckdb"],"date":"2026-05-22T00:00:00.000Z","image":"./hero.jpg","description":"Most SAP analytics pipelines pay a 14-hour latency tax for no good reason. Here is how to pull live SAP data into a DuckDB-native AI stack \u2014 no warehouse required."},"unlisted":false,"prevItem":{"title":"An AI Agent That Calls Live SAP Data \u2014 and Fixes It in ABAP","permalink":"/blog/real-time-sap-for-ai-part-2"},"nextItem":{"title":"I Asked Claude Code to Build a CDS View. It Got the Delta Annotations Right.","permalink":"/blog/claude-code-builds-odp-delta-cds-view"}}')},96532:(e,a,t)=>{t.d(a,{A:()=>n});const n=t.p+"assets/images/latency-comparison-87de2c7f5354b43038cda5134090a939.png"}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.