PageSourceSearch

https://streamdal.com/blog/

html streamdal.com collected 2026-09-25 20:16:41 UTC 135,293 bytes, 70 lines download raw bytes

1<!DOCTYPE html><html lang="en"> <head><!-- Google Tag Manager -->
1<script type="c2906bfa759604a8e90947b7-text/javascript">
2      (function (w, d, s, l, i) {
3        w[l] = w[l] || [];
4        w[l].push({
5          "gtm.start": new Date().getTime(),
6          event: "gtm.js",
7        });
8        var f = d.getElementsByTagName(s)[0],
9          j = d.createElement(s),
10          dl = l != "dataLayer" ? "&l=" + l : "";
11        j.async = true;
12        j.src = "https://www.googletagmanager.com/gtm.js?id=" + i + dl;
13        f.parentNode.insertBefore(j, f);
14      })(window, document, "script", "dataLayer", "GTM-PX732JX");
15    </script>
15<!-- End Google Tag Manager --><!-- Font preloads --><link rel="preload" href="/fonts/inter.woff2" as="font" type="font/woff2" crossorigin><link rel="preload" href="/fonts/space-grotesk.woff2" as="font" type="font/woff2" crossorigin><title>Streamdal Blog</title><meta charset="utf-8"><link rel="canonical" href="https://streamdal.com/blog/"><meta name="description" content="Embed privacy controls in your application code to detect PII as it enters and leaves your systems, preventing it from reaching unintended databases, data streams, or pipelines. Build trust by eliminating the unknown."><meta name="robots" content="index, follow"><meta property="og:title" content="Code-Native Data Privacy"><meta property="og:type" content="website"><meta property="og:image" content="/assets/images/streamdal-opengraph.webp"><meta property="og:url" content="https://streamdal.com/blog/"><meta name="twitter:card" content="summary_large_image"><meta name="twitter:site" content="@streamdal"><meta name="twitter:title" content="Code-Native Data Pipelines"><meta name="twitter:image" content="https://streamdal.com/assets/images/streamdal-opengraph.webp"><meta name="twitter:description" content="Embed privacy controls in your application code to detect PII as it enters and leaves your systems, preventing it from reaching unintended databases, data streams, or pipelines. Build trust by eliminating the unknown."><meta name="twitter:creator" content="@streamdal"><meta name="viewport" content="width=device-width"><link rel="icon" type="image/svg+xml" sizes="32x32" href="/assets/icons/favicon-dark.svg" media="(prefers-color-scheme: dark)"><link rel="icon" type="image/svg+xml" sizes="32x32" href="/assets/icons/favicon-light.svg" media="(prefers-color-scheme: light)"><link><meta name="generator" content="Astro v4.4.9"><link rel="sitemap" href="/sitemap-index.xml"><link rel="stylesheet" href="/_astro/_slug_.Cvq48XcE.css" />
16<style>.wordstream{display:flex;align-items:center;justify-content:center;padding-left:1.75rem;padding-right:1.75rem;padding-bottom:50px;padding-top:1rem;--tw-text-opacity: 1;color:rgb(255 255 255 / var(--tw-text-opacity))}@media (min-width: 475px){.wordstream{gap:1rem}}@media (min-width: 640px){.wordstream{gap:1.5rem}}@media (min-width: 1024px){.wordstream{padding-bottom:75px}}@media (min-width: 1440px){.wordstream{justify-content:flex-start;padding-left:56px}}.header__group{display:flex;flex-direction:column;align-items:flex-start;justify-content:flex-start;gap:.75rem}.wordstream__header{text-align:center;font-family:Space Grotesk,ui-sans-serif,system-ui,sans-serif,"Apple Color Emoji","Segoe UI Emoji",Segoe UI Symbol,"Noto Color Emoji";font-size:48px;font-weight:300;line-height:100%}@media (min-width: 640px){.wordstream__header{font-size:60px}}@media (min-width: 1024px){.wordstream__header{font-size:90px}}@media (min-width: 1440px){.wordstream__header{font-size:140px}}.wordstream__header .header--bold{font-weight:700}.wordstream__sub{font-family:Space Grotesk,ui-sans-serif,system-ui,sans-serif,"Apple Color Emoji","Segoe UI Emoji",Segoe UI Symbol,"Noto Color Emoji";font-size:24px;font-weight:700;line-height:100%}
17.card-blog{position:relative;display:flex;flex-direction:column;gap:1.5rem;overflow:hidden;border-radius:1rem;border-width:1px;border-style:solid;--tw-border-opacity: 1;border-color:rgb(31 23 57 / var(--tw-border-opacity));--tw-bg-opacity: 1;background-color:rgb(17 14 30 / var(--tw-bg-opacity));padding:1.5rem 1rem;--tw-text-opacity: 1;color:rgb(255 255 255 / var(--tw-text-opacity));transition-property:box-shadow;transition-duration:.3s;transition-timing-function:cubic-bezier(.4,0,.2,1)}@media (min-width: 475px){.card-blog{gap:0px;border-radius:.5rem}}@media (min-width: 640px){.card-blog{border-radius:1rem}}@media (min-width: 1024px){.card-blog{padding-left:1.5rem;padding-right:1.5rem}}@media (min-width: 1440px){.card-blog{max-width:780px}}.card-blog *{position:relative;z-index:1}.card-blog:before{content:"";position:absolute;top:0;left:0;width:100%;height:100%;background:radial-gradient(4296.34% 94.5% at 17.56% 50%,#110e1e4d,#956cff4d);opacity:0;transition:opacity .3s ease-in-out;z-index:0}.card-blog:hover{box-shadow:1px 4px 16.8px #6a31ff40}.card-blog:hover:before{opacity:1}.more-articles-section{position:relative;margin-top:4rem;padding:4rem 2rem}@media (min-width: 475px){.more-articles-section{margin-top:3.5rem;padding-top:1.75rem;padding-bottom:2.5rem}}@media (min-width: 640px){.more-articles-section{margin-top:5rem;padding-top:2.25rem;padding-bottom:3.5rem}}@media (min-width: 768px){.more-articles-section{margin-top:4rem;padding-left:104px;padding-right:104px;padding-bottom:5rem}}@media (min-width: 1024px){.more-articles-section{padding-left:72px;padding-right:72px;padding-bottom:7rem}}.more-articles-title{font-size:32px;font-weight:700;line-height:2.25rem}@media (min-width: 640px){.more-articles-title{padding-left:32px;padding-right:32px;font-size:3rem;line-height:1;line-height:52px}}@media (min-width: 768px){.more-articles-title{padding-left:0;padding-right:0}}.more-articles-title:before{margin-left:2rem;margin-right:2rem}@media (min-width: 640px){.more-articles-title:before{margin-left:64px;margin-right:64px}}@media (min-width: 768px){.more-articles-title:before{margin-left:104px;margin-right:104px}}@media (min-width: 1024px){.more-articles-title:before{margin-left:72px;margin-right:72px}}.more-articles-title:before{content:"";position:absolute;
17left:0;right:0;height:1px;background:#948ea4;top:0}.more-articles-container{display:grid;grid-template-columns:repeat(3,1fr);gap:10px;padding-top:16px}@media (max-width: 1440px){.more-articles-container{grid-template-columns:repeat(2,1fr)}}@media (max-width: 474px){.more-articles-container{grid-template-columns:repeat(1,1fr);gap:4px;padding-top:8px}}
18</style>
19<link rel="stylesheet" href="/_astro/index.BhAHZrnC.css" />
20<link rel="stylesheet" href="/_astro/BlogPostCollection.Cv8lQHIx.css" />
20<script type="c2906bfa759604a8e90947b7-module" src="/_astro/hoisted.BbPSJNcy.js"></script>
vendor: 1 bytes, line 20
20
21<script type="c2906bfa759604a8e90947b7-text/javascript">!(function(w,p,f,c){if(!window.crossOriginIsolated && !navigator.serviceWorker) return;c=w[p]=Object.assign(w[p]||{},{"lib":"/~partytown/","debug":false});c[f]=(c[f]||[])})(window,'partytown','forward');/* Partytown 0.8.2 - MIT builder.io */
22!function(t,e,n,i,o,r,a,s,d,c,l,p){function u(){p||(p=1,"/"==(a=(r.lib||"/~partytown/")+(r.debug?"debug/":""))[0]&&(d=e.querySelectorAll('script[type="text/partytown"]'),i!=t?i.dispatchEvent(new CustomEvent("pt1",{detail:t})):(s=setTimeout(f,1e4),e.addEventListener("pt0",w),o?h(1):n.serviceWorker?n.serviceWorker.register(a+(r.swPath||"partytown-sw.js"),{scope:a}).then((function(t){t.active?h():t.installing&&t.installing.addEventListener("statechange",(function(t){"activated"==t.target.state&&h()}))}),console.error):f())))}function h(t){c=e.createElement(t?"script":"iframe"),t||(c.style.display="block",c.style.width="0",c.style.height="0",c.style.border="0",c.style.visibility="hidden",c.setAttribute("aria-hidden",!0)),c.src=a+"partytown-"+(t?"atomics.js?v=0.8.2":"sandbox-sw.html?"+Date.now()),e.querySelector(r.sandboxParent||"body").appendChild(c)}function f(n,o){for(w(),i==t&&(r.forward||[]).map((function(e){delete t[e.split(".")[0]]})),n=0;n<d.length;n++)(o=e.createElement("script")).innerHTML=d[n].innerHTML,o.nonce=r.nonce,e.head.appendChild(o);c&&c.parentNode.removeChild(c)}function w(){clearTimeout(s)}r=t.partytown||{},i==t&&(r.forward||[]).map((function(e){l=t,e.split(".").map((function(e,n,i){l=l[i[n]]=n+1<i.length?"push"==i[n+1]?[]:l[i[n]]||{}:function(){(t._ptf=t._ptf||[]).push(i,arguments)}}))})),"complete"==e.readyState?u():(t.addEventListener("DOMContentLoaded",u),t.addEventListener("load",u))}(window,document,navigator,top,window.crossOriginIsolated);;((d,s)=>(s=d.currentScript,d.addEventListener('astro:before-swap',()=>s.remove(),{once:true})))(document);</script>
22</head> <body> <!-- Google Tag Manager (noscript) -->
vendor: 69 bytes, line 22
22 <noscript> <iframe src="https://www.googletagmanager.com/ns.html?id=
22GTM-PX732JX
vendor: 84 bytes, line 22
22" height="0" width="0" style="display:none;visibility:hidden"></iframe> </noscript> 
22<!-- End Google Tag Manager (noscript) --> <!--<Header type="website" />--> <header> <nav class="mx-auto flex h-[81px] max-w-[1440px] items-center justify-between"> <a class="logo-link" href="/"> <img src="/assets/logos/streamdal.svg" alt="Streamdal logo" width="198" height="28" loading="lazy" decoding="async"> </a> <ul class="desktop-nav-links"> <li> <a class="desktop-link" href="/" title="Company">Company</a> </li> <li> <a class="desktop-link" href="https://docs.streamdal.com" title="Docs">Docs</a> </li> <li> <a class="desktop-link" href="#integrations" title="Integrations">Integrations</a> </li> <li> <a class="desktop-link" href="/blog" title="Blog">Blog</a> </li> </ul> <div class="community-links"> <a class="demo-btn" href="https://demo.streamdal.com" data-astro-cid-d4j6s75h>
23Watch our demo
24<img src="/assets/nav/triangle.svg" alt="Play icon" data-astro-cid-d4j6s75h width="12" height="14" loading="lazy" decoding="async"> </a>  <a class="star-btn undefined" href="https://github.com/streamdal/streamdal" data-astro-cid-hsxdubvx>
25Star us on GitHub
26<img src="/assets/nav/star.svg" alt="Play icon" data-astro-cid-hsxdubvx width="19" height="18" loading="lazy" decoding="async"> </a>  </div> <div id="hamburger-btn"> <span></span> <span></span> <span></span> </div> <div class="menu-nav" id="myNav"> <ul class="content"> <li> <a href="#" title="Company"> Company</a> </li> <li> <a href="https://docs.streamdal.com" title="Docs"> Docs</a> </li> <li> <a href="/#integrations" title="Integrations"> Integrations</a> </li> <li> <a href="/blog" title="Blog">Blog</a> </li> <li> <ul> <li><a class="star-btn undefined" href="https://github.com/streamdal/streamdal" data-astro-cid-hsxdubvx>
27Star us on GitHub
28<img src="/assets/nav/star.svg" alt="Play icon" data-astro-cid-hsxdubvx width="19" height="18" loading="lazy" decoding="async"> </a> </li> <li><a class="demo-btn" href="https://demo.streamdal.com" data-astro-cid-d4j6s75h>
29Watch our demo
30<img src="/assets/nav/triangle.svg" alt="Play icon" data-astro-cid-d4j6s75h width="12" height="14" loading="lazy" decoding="async"> </a> </li> </ul> </li> </ul> </div>  </nav> </header>  <main class="mx-auto max-w-[1440px] md:px-14 lg:px-[70px]"> <section class="wordstream 2xl:pl-[56px]"> <img class="hidden xs:block 2xl:w-[178px]" src="/assets/images/bird.svg" alt="bird"> <div class="header__group"> <h2 class="wordstream__header mx-auto xs:mx-0">
31word<span class="header--bold">stream</span> </h2> <h5 class="wordstream__sub text-center xs:text-start">
32Spillin’ that deep-tech startup tea.
33</h5> </div> </section> <style>astro-island,astro-slot,astro-static-slot{display:contents}</style>
33<script type="c2906bfa759604a8e90947b7-text/javascript">(()=>{var e=async t=>{await(await t())()};(self.Astro||(self.Astro={})).only=e;window.dispatchEvent(new Event("astro:only"));})();;(()=>{var v=Object.defineProperty;var A=(c,s,a)=>s in c?v(c,s,{enumerable:!0,configurable:!0,writable:!0,value:a}):c[s]=a;var d=(c,s,a)=>(A(c,typeof s!="symbol"?s+"":s,a),a);var u;{let c={0:t=>m(t),1:t=>a(t),2:t=>new RegExp(t),3:t=>new Date(t),4:t=>new Map(a(t)),5:t=>new Set(a(t)),6:t=>BigInt(t),7:t=>new URL(t),8:t=>new Uint8Array(t),9:t=>new Uint16Array(t),10:t=>new Uint32Array(t)},s=t=>{let[e,n]=t;return e in c?c[e](n):void 0},a=t=>t.map(s),m=t=>typeof t!="object"||t===null?t:Object.fromEntries(Object.entries(t).map(([e,n])=>[e,s(n)]));customElements.get("astro-island")||customElements.define("astro-island",(u=class extends HTMLElement{constructor(){super(...arguments);d(this,"Component");d(this,"hydrator");d(this,"hydrate",async()=>{var f;if(!this.hydrator||!this.isConnected)return;let e=(f=this.parentElement)==null?void 0:f.closest("astro-island[ssr]");if(e){e.addEventListener("astro:hydrate",this.hydrate,{once:!0});return}let n=this.querySelectorAll("astro-slot"),r={},l=this.querySelectorAll("template[data-astro-template]");for(let o of l){let i=o.closest(this.tagName);i!=null&&i.isSameNode(this)&&(r[o.getAttribute("data-astro-template")||"default"]=o.innerHTML,o.remove())}for(let o of n){let i=o.closest(this.tagName);i!=null&&i.isSameNode(this)&&(r[o.getAttribute("name")||"default"]=o.innerHTML)}let h;try{h=this.hasAttribute("props")?m(JSON.parse(this.getAttribute("props"))):{}}catch(o){let i=this.getAttribute("component-url")||"<unknown>",b=this.getAttribute("component-export");throw b&&(i+=` (export ${b})`),console.error(`[hydrate] Error parsing props for component ${i}`,this.getAttribute("props"),o),o}let p;await this.hydrator(this)(this.Component,h,r,{client:this.getAttribute("client")}),this.removeAttribute("ssr"),this.dispatchEvent(new CustomEvent("astro:hydrate"))});d(this,"unmount",()=>{this.isConnected||this.dispatchEvent(new CustomEvent("astro:unmount"))})}disconnectedCallback(){document.removeEventListener("astro:after-swap",this.unmount),document.addEventListener("astro:after-swap",this.unmount,{once:!0})}connectedCallback(){if(!this.hasAttribute("await-children")||document.readyState==="interactive"||document.readyState==="complete")this.childrenConnectedCallback();else{let e=()=>{document.removeEventListener("DOMContentLoaded",e),n.disconnect(),this.childrenConnectedCallback()},n=new MutationObserver(()=>{var r;((r=this.lastChild)==null?void 0:r.nodeType)===Node.COMMENT_NODE&&this.lastChild.nodeValue==="astro:end"&&(this.lastChild.remove(),e())});n.observe(this,{childList:!0}),document.addEventListener("DOMContentLoaded",e)}}async childrenConnectedCallback(){let e=this.getAttribute("before-hydration-url");e&&await import(e),this.start()}async start(){let e=JSON.parse(this.getAttribute("opts")),n=this.getAttribute("client");if(Astro[n]===void 0){window.addEventListener(`astro:${n}`,()=>this.start(),{once:!0});return}try{await Astro[n](async()=>{let r=this.getAttribute("renderer-url"),[l,{default:h}]=await Promise.all([import(this.getAttribute("component-url")),r?import(r):()=>()=>{}]),p=this.getAttribute("component-export")||"default";if(!p.includes("."))this.Component=l[p];else{this.Component=l;for(let y of p.split("."))this.Component=this.Component[y]}return this.hydrator=h,this.hydrate},e,this)}catch(r){console.error(`[astro-island] Error hydrating ${this.getAttribute("component-url")}`,r)}}attributeChangedCallback(){this.hydrate()}},d(u,"observedAttributes",["props"]),u))}})();</script>
33<astro-island uid="ZWs61Q" component-url="/_astro/BlogPostCollection.Cyd1qRJW.js" component-export="default" renderer-url="/_astro/client.CWJTIpmb.js" props="{&quot;blogPosts&quot;:[1,[[0,{&quot;id&quot;:[0,&quot;blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance.mdx&quot;],&quot;slug&quot;:[0,&quot;blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance&quot;],&quot;body&quot;:[0,&quot;\nimport EngageUs from \&quot;@shared/EngageUs/EngageUs.tsx\&quot;;\nimport ImageWithCaption from &#39;@components/ImageWithCaption.astro&#39;\nimport PostFooter from &#39;@components/PostFooterDiscord.astro&#39;\n\nElasticsearch has become ubiquitous and can be found in organizations of all sizes. However, optimizing Elasticsearch isn’t just a science; it’s an art. This is because performance metrics are uniquely influenced by the type of data being ingested.\n\nI have managed some fairly large Elasticsearch clusters for several organizations. I have employed the following principles to keep Elasticsearch clusters up and running fast.\n\n## Core Principles for Maintaining a Fast ElasticSearch Cluster\n\n### **Heap Size Management**\n\nProper configuration of the JVM heap size is crucial. It’s recommended to allocate no more than 50% of the available memory to the Elasticsearch heap while ensuring that the heap size is enough to manage the cluster’s workload. Do not exceed 32GB, as going too large can lead to slowdowns as well.\n\n### **Leverage Datastreams when possible**\n\nDatastreams are your friends, especially when dealing with time-based data. They simplify index creation and management, making your data handling much more efficient.\n\n### **Monitoring and Performance Tuning**\n\nHere’s a non-negotiable: regular monitoring of your cluster’s health and performance metrics. Tools like Prometheus, coupled with the Elasticsearch exporter, are invaluable in this regard. Neglecting monitoring is a perilous path, as Elasticsearch can be finicky, and performance issues tend to accumulate stealthily over time. I typically use [Prometheus](https://github.com/prometheus/prometheus) with the [Elasticsearch exporter](https://github.com/prometheus-community/elasticsearch_exporter).\n\n&lt;ImageWithCaption width={1000} height={446} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-01.webp&#39; alt=&#39;Prometheus Monitor&#39; caption=&#39;Prometheus Monitor&#39;/&gt;\n\n### **Correctly size shards based on the cluster**\n\nShard sizing is more art than science. There’s no one-size-fits-all answer, but experience suggests that shards in the 10GB to 50GB range generally yield good results for logs and time series data. Play with adjusting these and review your metrics to see how your query/index times are affected.\n\n### **Create lifecycle policies**\n\nAutomate the management of your index lifecycle. This involves setting up phases for the creation, aging, and deletion of indices, thereby optimizing both storage and data accessibility.\n\n&lt;EngageUs /&gt;\n\n## A Deeper Dive: Data Efficiency\n\nOne of the most important strategies involves meticulous data management — specifically, dropping unnecessary data, reducing the number of fields, and minimizing field sizes. This approach is crucial for maintaining a lean and efficient Elasticsearch cluster. It’s akin to decluttering your digital environment, which significantly enhances performance by simplifying the search and retrieval processes.\n\n### **Streamdal Log Processor Integration**\n\nWe will cover using open-source [Streamdal](http://streamdal.com/) and its log processor to accomplish these goals. But, you can probably implement similar functionality by manually using log stash’s log filters. By using [Streamdal’s log processor](https://github.com/streamdal/log-processor) with our LogStash agent, we tap into a suite of functions designed for efficient data analysis and transformation. This integration ensures that the data landing in Elasticsearch is already processed, optimized, and ready for efficient storage and retrieval. See my other blog post [Super LogStash with PII Rules](https://medium.com/streamdal/supercharge-logstash-with-smart-pii-rules-a5db76142017) for more info on how to set up/additional features.\n\n## Data Management: Game Plan\n\nLet&#39;s look at a sample log entry and come up with a plan to keep our Elasticsearch running fast.\n\n```json\n{\n  \&quot;timestamp\&quot;: \&quot;2024-02-12T14:46:25Z\&quot;,\n  \&quot;user_id\&quot;: 77,\n  \&quot;activity\&quot;: \&quot;view\&quot;,\n  \&quot;details\&quot;: {\n    \&quot;ip_address\&quot;: \&quot;192.168.1.166\&quot;,\n    \&quot;location\&quot;: \&quot;Germany\&quot;,\n    \&quot;email\&quot;: \&quot;[email protected]\&quot;,\n    \&quot;phone\&quot;: \&quot;
33+1-967-510-7714\&quot;,\n    \&quot;long_field\&quot;: \&quot;vna6EnpbOajYeUCMjKH5DgngCYYwPJo9MCKCjfd16d0mKuSNzbkWvm0WhAXDYOWkaJcGjQJjz9wbOglX6943d8GbkabSvV7KdGzOadMoEaR4C85eKcOAUTZoYa2MU4wwAh6lP9j5muYLBxoGFfxxhZ4ugw45DoFHVHIqm001NgsNFrMvFkrqkqWjU5pr44JougBqwgud1FF6FVBG12qIncTBPietYG9c6SOtDpPuIoBQD7WpAAmLuJYajsuaXjghGzCJ1PGuop1GorubJ4ihgiYNIpV6KLniNNwA7CL2DbAMRBCDGWEDcnYWtwr1syiwhfSj5KXhl8Xj46AkwDowMUq3NJXLDL7zY34ML15ETZRRPcyURjWCbb8Uq8BoGvt0s2M0TS9Zd7G0L8l7fLi3XcUSaLadVjDWBbH33KJpvOEMisTFugXjIb9k0LysmPovGTSmNOxxjr5mW6gKY7Tshb2z9s8kfvcrwPVzEXwYuUwT2IPwGo8pIVNhdhfQmb6l884wAGWrgzhut2WuXP8xmR2dNXFHBKMjUBApT5YDzgsLq6J8C2NLxAod2YLD7lPG9Jjia6WcJt0fk1MtHQgjKrtehKk0bCtzeOHFiICMoXkLap9I2xrEx8SJ6vA94o5tEL8seFcWh1OxxFyzVizAnIrNIziISSvEFinhH56acYcZoojLarWEH52JycGQdED2bBiNRrWUxQvYYfzuywue09FJD8ZjqZmIrpTuY7zfBVeygS0TWwolmz74JEO8exSGm9mgTUmVSLvczYtFBtFG7F2I30Yjx41AUAXJ54s2GLQfIrGmP5sMS2vmE7DZCXtVDWdL0hdkszHrS5AesndfRGDLj901nlIoe646tkGcxMGEHOMNsTxDLUUzyqm8pabKf191Jeu2AVJaySBL8GVjGwyKY5pPiIafQltZzyvBlkIWlW5dsSXr9ADYENir1EVXlaSwloqjCnbvhUtakShT92VnvldoOgAtHmpH2QtmgllmqBCbdSP502Hy8QA26UHV\&quot;,\n    \&quot;product_color\&quot;: \&quot;red\&quot;,\n    \&quot;product_size\&quot;: \&quot;large\&quot;,\n    \&quot;irrelevant_field\&quot;: \&quot;This field is not relevant to the app\&quot;\n  },\n  \&quot;is_e2e_test_account\&quot;: false\n}\n```\n\nWe will now create a series of pipelines to remove any unnecessary data. When completed we will have a much more sustainable stream of logs.\n\n### **Dropping Test and Development Data**\n\nIn our example log, there is a field called ‘is_e2e_test_account’ with value either true or false. This indicates the message is a test from our internal systems. We have no reason to log these to our Elasticsearch, so let’s find all instances where the value equals true and set metadata that instructs our log-processor to drop them.\n\n&lt;ImageWithCaption width={1000} height={543} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-02.webp&#39; alt=&#39;Drop End to END Test&#39;/&gt;\n\n### **Truncating Large Field Values**\n\nIn the example event, we will now truncate the value under the key ‘long_field’. As you can see it&#39;s a very verbose value and our application does not need the full value.\n\n&lt;ImageWithCaption width={1000} height={515} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-03.webp&#39; alt=&#39;TODO: NEED ALT&#39;/&gt;\n\n### **Keeping Only Necessary Fields**\n\nStreamdal’s *‘Extract fields’* step is a powerful tool for streamlining our data. It allows us to selectively keep only the essential fields in our JSON data, effectively dropping any unnecessary or redundant information.\n\n&lt;ImageWithCaption width={1000} height={515} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-04.webp&#39; alt=&#39;TODO: NEED ALT&#39;/&gt;\n\n### **Game Plan Results:**\n\nBefore we dropped test data, about 20% of all data saved to Elasticsearch was from our e2e test account.\n\n&lt;ImageWithCaption width={1000} height={515} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-05.webp&#39; alt=&#39;TODO: NEED ALT&#39;/&gt;\n\nAfter we applied our pipeline, all test data dropped off.\n\n&lt;ImageWithCaption width={1000} height={515} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-06.webp&#39; alt=&#39;TODO: NEED ALT&#39;/&gt;\n\nOur Elasticsearch had a bunch of extra fields and a large unnecessary field value.\n\n&lt;ImageWithCaption width={1000} height={515} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-07.webp&#39; alt=&#39;TODO: NEED ALT&#39;/&gt;\n\nNow we have extracted the fields we care about, and truncated unnecessarily long field bodies.\n\n&lt;ImageWithCaption width={1000} height={515} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance-08.webp&#39; alt=&#39;TODO: NEED ALT&#39;/&gt;\n\n### **Conclusion: Elevating Elasticsearch Performance**\n\nThe key to a high-performing Elasticsearch cluster lies in a blend of configuration and smart data management. By managing heap sizes, leveraging data streams, conducting regular performance monitoring, sizing shards appropriately, creating lifecycle policies, and optimizing data processing with Streamdal, we not only enhance the speed and efficie
33ncy of our clusters but also ensure their scalability and resilience.\n\nFollowing these principles should keep your Elasticsearch running fast and avoid performance trending down over time.\n\n&lt;PostFooter/&gt;\n\n\n\n\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Blazing Fast Elasticsearch: Optimizing Data Storage for Peak Performance&quot;],&quot;pubDate&quot;:[3,&quot;2024-02-14T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Optimize ElasticSearch performance with smart data management, leveraging Streamdal for efficient processing.&quot;],&quot;author&quot;:[0,&quot;Fritz&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/blazing-fast-elasticsearch-optimizing-data-storage-for-peak-performance.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/fruitz.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;comparing-avro-vs-protobuf-for-data-serialization.mdx&quot;],&quot;slug&quot;:[0,&quot;comparing-avro-vs-protobuf-for-data-serialization&quot;],&quot;body&quot;:[0,&quot;\nimport PostFooter from &#39;@components/PostFooterDiscord.astro&#39;\n\n## Comparing Avro vs Protobuf for Data Serialization\n\nData serialization is a crucial aspect of modern distributed systems because it enables the efficient communication and storage of structured data. In this article, we will discuss two popular serialization formats: Avro and Protocol Buffers, Protobuf for short, and compare their strengths and weaknesses to help you make an informed decision about which one to use in your projects.\n\n## What Is Data Serialization, and Why Do You Need It?\n\nSerialization is the process of converting structured data, such as objects or records, into a format that can be transmitted over the network or stored on disk. This process is essential for enabling communication between distributed systems or microservices, for building efficient event-driven systems, and for persisting data in databases or file systems.\n\n## What is Avro?\n\nAvro is a serialization framework developed by the Apache Software Foundation. It is designed to be language-independent and schema-based, which means that data is serialized and deserialized using a schema that describes the structure of the data. Avro schemas are defined in JSON.\n\n### What is Protobuf?\n\nProtobuf is a serialization format developed by Google. Like Avro, Protobuf is also schema-based and language-independent. However, unlike Avro, Protobuf relies on static typing and code generation to serialize and deserialize data. That means you need to compile your schema into language-specific classes or libraries with the protoc compiler before you can use it in your application.\n\n### Avro vs. Protobuf: Strengths and Weaknesses\n\nBoth Avro and Protobuf have strengths and weaknesses, which make them suitable for different use cases. Let’s take a look at the differences in more detail, so you can make an informed decision about which is right for your use case:\n\n### Avro Strengths\n\n- **Dynamic typing**: Unlike Protobuf, Avro does not require code generation, which enables more flexibility and easier integration with dynamic languages like Python or Ruby.\n- **Self-describing messages**: Serialized data in Avro includes embedded schema information, making it possible to decode the data even if the reader does not have access to the original schema.\n\n### Avro Weaknesses\n\n- **Verbosity of schema definition**: Avro schemas are defined in JSON, which can be more verbose than Protobuf’s .proto format.\n- **Slower serialization/deserialization performance**: Due to its dynamic typing nature and inclusion of embedded schema information, Avro can have slower serialization and deserialization speeds compared to Protobuf.\n- **Limited support for some languages**: While Avro supports multiple programming languages, the level of support or maturity for some languages might not be as polished as others (e.g., Java has better support than Python).\n\n### Protobuf Strengths\n\n- **High performance with a small payload size**: Protobuf is designed for fast serialization and deserialization, as well as compact binary representation of data. This makes it suitable for high-throughput applications like RPCs (Remote Procedure Calls) or real-time streaming systems.\
33n- **Static typing with code generation**: Protobuf requires pre-defined message structures compiled into language-specific classes or libraries, which can result in better type safety and performance compared to Avro’s dynamic typing.\n- **Cross-platform compatibility**: Protobuf supports multiple programming languages and platforms, making it an ideal choice for applications with diverse technology stacks.\n\n### Protobuf Weaknesses\n\n- **Code generation**: Protobuf requires an extra step in the development workflow, as you need to recompile the corresponding language-specific classes/libraries whenever the message structure changes.\n- Less human-readable wire format: Serialized data in Protobuf is purely binary, making it harder to inspect or debug compared to Avro.\n\nWhen choosing between Avro and Protobuf for data serialization, consider factors such as the typing system, performance, and language support. Avro might be a better fit for applications that require flexibility and human-readability of the schema, whereas Protobuf is the better choice for applications that prioritize performance, type safety, and cross-platform compatibility.\n\nRegardless of the encodings used with streaming or event-driven data, being able to observe, monitor, and act on data is paramount for compliance, reliability, and shipping complex features faster.\n\n&lt;PostFooter/&gt;\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Comparing Avro vs Protobuf for Data Serialization&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-05T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Avro and Protobuf are two popular data serialization formats. Learn how they work and their strengths and weaknesses.&quot;],&quot;author&quot;:[0,&quot;Dan Silber&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/comparing-avro-vs-protobuf-for-data-serialization.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/dan-silber.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;configuring-streamdal-with-terraform.mdx&quot;],&quot;slug&quot;:[0,&quot;configuring-streamdal-with-terraform&quot;],&quot;body&quot;:[0,&quot;\nimport ImageWithCaption from &#39;@components/ImageWithCaption.astro&#39;\nimport PostFooter from &#39;../../components/PostFooterDiscord.astro&#39;\n\nWe love our gorgeous UI and we know you will too. However, we understand that manually configuring all your pipelines can be a daunting task at scale. We’ve created a Terraform provider to help. This blog will show you how to quickly get a pipeline up and running with our Terraform provider.\n\n## Why Terraform?\n\nAs the current industry leader, Terraform was an obvious first choice of Infrastructure as Code (IaC) tooling to support configuring the Streamdal server. We do plan to support additional IaC tools in the future.\n\n[The Terraform provider and documentation](https://registry.terraform.io/providers/streamdal/streamdal/latest)\n\n## Getting Started\n\n*Skip this section if you are already familiar with Terraform and simply copy/paste the provider block below.*\n\n1. [Install the Terraform command line tool](https://developer.hashicorp.com/terraform/install)\n2. Create a directory/folder to hold the config file and Terraform data. Let’s call ours `streamdal-config`\n3. Now create an empty text file called `main.tf`. This is where we’ll put the Streamdal pipeline definitions, written in [Hashicorp Configuration Language(HCL).](https://github.com/hashicorp/hcl)\n4. In the `main.tf` file, we’ll configure a Terraform provider, which is used to take our HCL definitions and make the necessary API calls to the Streamdal Server:\n\n```json\n terraform {\n   required_providers {\n     streamdal = {\n       version = \&quot;0.1.2\&quot;\n       source  = \&quot;streamdal/streamdal\&quot;\n     }\n   }\n }\n\n provider \&quot;streamdal\&quot; {\n   token              = \&quot;streamdal-server-token-here\&quot;\n   address            = \&quot;streamdal-server-address-here:8082\&quot;\n   connection_timeout = 10\n }\n```\n\n5. Now run `terraform init` in your terminal to download the required provider. You should see output similar to the following:\n\n&lt;ImageWithCaption width={1000} height={446} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/output-of-terraform-init-command.webp&#39; alt=&#39;Output of Terraform init command&#39; caption=&#39;Output of Terraform init command&#39;/&gt;
33\n\nNow we’re ready to begin configuring our first pipeline and audience.\n\n## Setting up our first Pipeline\n\nIf you skipped the previous section, copy/paste the provider block into your `.tf` file.\n\nWe’ll start off with a basic pipeline definition that detects some PII (an email address) in a JSON payload and masks it.\n\nWe’ll use the [pipeline resource](https://registry.terraform.io/providers/streamdal/streamdal/latest/docs/resources/pipeline) to configure this:\n\n```json\nresource \&quot;streamdal_pipeline\&quot; \&quot;mask_email\&quot; {\n  name = \&quot;Mask Email\&quot;\n\n  step {\n    name = \&quot;Detect Email Field\&quot;\n\n    # We specify abort conditions here since we don&#39;t want\n    # to continue with the second step if there is nothing\n    # to transform.\n    on_false {\n      abort = \&quot;abort_current\&quot; # No email found\n    }\n    on_error {\n      abort = \&quot;abort_current\&quot; # An error occurred\n    }\n    dynamic = false\n    detective {\n      type   = \&quot;pii_email\&quot;\n      args   = [] # no args for this type\n      negate = false\n      path   = \&quot;\&quot; # No path, we will scan the entire payload\n    }\n  }\n\n  step {\n    name    = \&quot;Mask Email Step\&quot;\n    dynamic = true\n    transform {\n      mask_value {\n        # No path needed since dynamic=true\n        # We will use the results from the first detective step\n        path = \&quot;\&quot;\n\n        # Mask the email field(s) we find with asterisks\n        mask = \&quot;*\&quot;\n      }\n    }\n  }\n}\n```\n\nUsing the `streamdal_pipeline` resource, we’ll configure a pipeline named *“Mask Email”*. This pipeline has two-step blocks that define each step of the pipeline.\n\nThe first step uses the “Detective” module (`detective{}`)to look for email addresses (`type=\&quot;pii_email\&quot;`)anywhere in the JSON payload it receives.\n\nThis type of matcher does not require any arguments (`args=[]`), and by omitting a path to a specific field (`path=\&quot;\&quot;`), the entire payload will be scanned for any fields that have a value of an email address.\n\nNow on to the second step…\n\nWe define another `step{}` block to create a second step. The important bit here is `dynamic=true` which indicates that we will use the results from the first step.\n\nWe will mask the email in the JSON payload, using the *“Transform”* module ( `transform{}`). Inside the `transform{}` block, we’ll configure a `mask_value` transformation, with an empty `path=\&quot;\&quot;`.\n\n*Note: We’re not specifying a path here because the previous step will pass any paths it detects as arguments to the second step.*\n\nGreat! We’ve defined our first pipeline, and now we need to set up the *“audience”* to use this pipeline on.\n\n## Defining and Assigning an Audience\n\nAn audience is a definition that the Streamdal SDK uses to determine which pipelines to apply to a given call to `.Process()`. It consists of 4 pieces of data:\n\n- `service_name` — A name indicating the service you’re running the SDK in.\n- `component_name` —This is used to indicate the source or destination of the data that you will be processing. Typically something like `kafka`, `postgres`, `rabbitmq`, etc\n- `operation_name` — Some kind of short descriptor indicating the operation that is processing the data.\n- `operation_type` — **Either** `consumer` **or** `producer`\n\nFor demo purposes, let’s say we’re running the Streamdal SDK in our **billing service**, which **consumes** data from a **Kafka** topic in order to generate a sales **report**.\n\nUsing the `streamdal_audience` Terraform resource, we would define the audience as follows:\n```json\nresource \&quot;streamdal_audience\&quot; \&quot;billing_sales_report\&quot; {\n  service_name   = \&quot;billing-svc\&quot;\n  component_name = \&quot;kafka\&quot;\n  operation_name = \&quot;gen-sales-report\&quot;\n  operation_type = \&quot;consumer\&quot;\n  pipeline_ids   = [resource.streamdal_pipeline.mask_email.id]\n}\n```\n\nFor folks unfamiliar with Terraform, the value of the `pipeline_ids` field is dynamically populated by the result of the `streamdal_pipeline` resource we previously defined.\n\nNow save your `main.tf` file, and we’ll create our pipeline and audience in the Streamdal server with the `terraform apply` command:\n\n&lt;ImageWithCaption width={1000} height={446} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/output-from-terraform-apply-cmd.webp&#39; alt=&#39;Output from Terraform apply command&#39; caption=&#39;Output from Terraform apply command&#39;/&gt;\n\nLet’s open up our Streamdal console to see the results:\n\n&lt;ImageWithCaption width={1000} height={446} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/Streamdal-console-with-our-new-pipeline-and-audience.webp&#39; alt=&#39;Streamdal console with our new pipeline and audience&#39; caption=&#39;Streamdal console with our new pipeline and audience&#39;/&gt;\n\nAs you can see, our *audience* is now defined and the “Mask Email” pipeline has been created and assigned to it.\n\n**We’re now ready to process data through our pipeline and prevent PII leakage!**\n\n## Full Example\n\nWe have created a repository located at https://github.com/streamdal/blog-terraform-demo with the Terraform code from this article and also an example Golang app for you to run.\n\n&lt;PostFooter/&gt;\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Configuring Streamdal with Terraform&quot;],&quot;pubDate&quot;:[3,&quot;2024-03-18T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;
33We love our gorgeous UI and we know you will too. However, we understand that manually configuring all your pipelines can be a daunting…&quot;],&quot;author&quot;:[0,&quot;Mark Gregan&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/Configuring-Streamdal-with-Terraform.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/mark-gregan.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;creating-a-prisma-client-extension.mdx&quot;],&quot;slug&quot;:[0,&quot;creating-a-prisma-client-extension&quot;],&quot;body&quot;:[0,&quot;\n\nimport ImageWithCaption from &#39;@components/ImageWithCaption.astro&#39;\nimport PostFooter from &#39;../../components/PostFooterDiscord.astro&#39;\n\nThe open-source extension code is [here](https://github.com/streamdal/prisma-extension-streamdal), complete with a runnable example.\n\n## Introduction\n\nPrisma, the popular Typescript ORM, has supported extending its client since version 4.1.0. The extension mechanism is straightforward and very flexible. You can use it to alter models, alter queries, add client level methods, and alter result data. Additionally, extensions can be chained so you can use multiple extensions together.\n\nAs you can see from a short list of available extensions [here](https://www.prisma.io/docs/orm/prisma-client/client-extensions/extension-examples), there are myriad use cases for client extensions including data transformations, logging, caching, and additional security. You can even use it to support natural language queries via ChatGPT.\n\nI had to bounce around a bit between their docs, examples, and discord to get our extension fully working, so I thought it would be worth walking through it.\n\n## Background\n\nAt Streamdal, we provide an open-source platform for building and executing Code-Native data pipelines. We already have a variety of SDKs to let our users view and transform data coming in and out of their Prisma client in the language of their choice. However, providing an extension allows our users to add any number of data pipelines to any existing Prisma operation simply by supplying a little extra configuration.\n\n## Getting Started\n\nPrisma provides an extension starter template [here](https://github.com/prisma/prisma-client-extension-starter). Fork or click the **This Template** button to create an extension project skeleton. By default, the starter template provides an example schema and implements a new `existsFn` on all Prisma models. You’ll find it at `src/index.ts`.\n\n```ts\nexport const existsFn = (_extensionArgs: Args) =&gt;\n  Prisma.defineExtension({\n    name: \&quot;prisma-extension-find-or-create\&quot;,\n    model: {\n      $allModels: {\n        async exists&lt;T, A&gt;(\n          this: T,\n          args: Prisma.Exact&lt;A, Prisma.Args&lt;T, &#39;findFirst&#39;&gt;&gt;\n        ): Promise&lt;boolean&gt; {\n\n          const ctx = Prisma.getExtensionContext(this)\n          const result = await (ctx as any).findFirst(args)\n          return result !== null\n        },\n      },\n    },\n  })\n```\n\nIn addition to extending models by adding new methods, you can add new client level methods or alter existing client operations such as **find** and create. We’ll be using this last technique as we want to allow our users to add pipelines to existing Prisma operations by adding a bit of configuration instrumentation and without having to rewrite a lot of code. Let’s try extending Prisma’s **create**. For a complete example see [here](https://github.com/streamdal/prisma-extension-streamdal/blob/main/src/index.ts).\n\n```ts\nimport { Prisma } from \&quot;@prisma/client/extension\&quot;;\nimport {\n  Audience,\n  OperationType,\n  SDKResponse,\n  Streamdal,\n  StreamdalConfigs,\n} from \&quot;@streamdal/node-sdk\&quot;;\n\nexport { Audience, OperationType, StreamdalConfigs };\n\nexport type StreamdalArgs = {\n  streamdalAudience?: Audience;\n};\n\nexport const streamdal = (streamdalConfigs: StreamdalConfigs) =&gt; {\n  const streamdal = new Streamdal(streamdalConfigs);\n  const decoder = new TextDecoder();\n  const encoder = new TextEncoder();\n\n  return Prisma.defineExtension({\n    name: \&quot;prisma-extension-streamdal\&quot;
33,\n    model: {\n      $allModels: {\n        async create&lt;T, A&gt;(\n          this: T,\n          args?: Prisma.Exact&lt;A, Prisma.Args&lt;T, \&quot;create\&quot;&gt;&gt; &amp; StreamdalArgs,\n        ): Promise&lt;Prisma.Result&lt;T, A, \&quot;create\&quot;&gt;&gt; {\n          const { streamdalAudience, ...rest }: any &amp; StreamdalArgs =\n            args || {};\n          const ctx = Prisma.getExtensionContext(this);\n\n          if (streamdalAudience) {\n            const streamdalResult: SDKResponse = await streamdal.process({\n              audience: streamdalAudience,\n              data: encoder.encode(JSON.stringify(rest.data)),\n            });\n            const data = decoder.decode(streamdalResult.data);\n            return (ctx as any).$parent[ctx.$name as any].create({\n              data: JSON.parse(data),\n            });\n          }\n\n          return (ctx as any).$parent[ctx.$name as any].create(rest);\n        },\n      },\n    },\n  });\n};\n```\n\nThere’s quite a bit going on here so I’ll walk through it a bit at a time. I’ve included the imports from our Typescript Node SDK so you can see that adding libraries to a Prisma client extension is as simple as doing an `npm install` as you would in any node app.\n\nAfter the imports, I re-export some library constructs so our users don’t have to care or know about any upstream libraries. Those combined with our main extension method signature mean the user can easily see all available configuration options by inspecting the type signature in their IDE.\n\n```ts\nexport { Audience, OperationType, StreamdalConfigs };\n\nexport type StreamdalArgs = {\n  streamdalAudience?: Audience;\n};\n\nexport const streamdal = (streamdalConfigs: StreamdalConfigs) =&gt; {\n...\n}\n```\n\nWith that our users can now simply wrap their existing Prisma client with this extension to get access to the added functionality and without breaking any existing use cases. The extension wrapper looks like this.\n\n```ts\nconst prisma = new PrismaClient().$extends(streamdal(streamdalConfigs));\n```\n\nWithin the extension, we immediately construct a few persistent things we need and name our extension using the convention Prisma recommends:\n\n```ts\nexport const streamdal = (streamdalConfigs: StreamdalConfigs) =&gt; {\n  const streamdal = new Streamdal(streamdalConfigs);\n  const decoder = new TextDecoder();\n  const encoder = new TextEncoder();\n\n  return Prisma.defineExtension({\n    name: \&quot;prisma-extension-streamdal\&quot;,\n    ...\n  });\n};\n```\n\nNext we extend the **create** operation on all models. Here we add the **StreamdalArgs** type we declared above to the existing Prisma **create** type so users can again inspect the **create** type signature to see the newly available options.\n\n```ts\nreturn Prisma.defineExtension({\n    name: \&quot;prisma-extension-streamdal\&quot;,\n    model: {\n      $allModels: {\n        async create&lt;T, A&gt;(\n          this: T,\n          args?: Prisma.Exact&lt;A, Prisma.Args&lt;T, \&quot;create\&quot;&gt;&gt; &amp; StreamdalArgs,\n        ): Promise&lt;Prisma.Result&lt;T, A, \&quot;create\&quot;&gt;&gt; {\n          ...\n        },\n      },\n    },\n  });\n```\n\nNote that the new args are optional. This allows us to add new functionality without breaking existing use cases. If the new args are present, we will wrap the operation with the new functionality, if not we simply invoke the existing operation. We do this by getting access to the parent operation via the extension context like so:\n\n```ts\nconst ctx = Prisma.getExtensionContext(this);\n...\nreturn (ctx as any).$parent[ctx.$name as any].create(rest);\n```\n\nThe exact details of the extended operation logic might vary depending on the operation. For example in our case, for read operations we execute our pipelines on the data *after* it fetched and for mutations we execute pipelines on the data *before* it is sent to the database. The idea here is that our users might want to do something like detect PII and mask it before it enters the database or leaves the system.\n\nYou can see that intervention in the example **create** we implemented.\n\n```ts\nconst { streamdalAudience, ...rest }: any &amp; StreamdalArgs = args || {};\nconst ctx = Prisma.getExtensionContext(this);\n\nif (streamdalAudience) {\n  const streamdalResult: SDKResponse = await streamdal.process({\n    audience: streamdalAudience,\n    data: encoder.encode(JSON.stringify(rest.data)),\n  });\n  const data = decoder.decode(streamdalResult.data);\n  return (ctx as any).$parent[ctx.$name as any].create({\n    data: JSON.parse(data),\n  });\n}\n\nreturn (ctx as any).$parent[ctx.$name as any].create(rest);\n```\n\nNext, we add extended model methods for all the operations we care about with the appropriate pipeline logic before or after the underlying operation. That means method wrappers for all the creates, finds, and updates Prisma provides. See them all [here](https://github.com/streamdal/prisma-extension-streamdal/blob/main/src/index.ts).\n\nNow that we’ve done that, we publish to NPM so all our users need to do to get started is: `npm install @streamdal/prisma-extension-streamdal`.\n\nIf you are curious about our Code-Native pipeline platform you can read more about it at streamdal.com or get started [here](https://github.com/streamdal/streamdal).\n\n&lt;PostFooter/&gt;\n\n\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Creating a Prisma Client Extension&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-10T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Prisma, the popular Typescript ORM, has supported extending its client since version 4.1.0.&quot;],&quot;author&quot;:[0,&quot;Jacob&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/prisma-client.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/jacob.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;data-management-simplifying-your-data-pipelines.mdx&quot;],&quot;slug&quot;:[0,&quot;data-management-simplifying-your-data-pipelines&quot;],&quot;body&quot;:[0,&quot;\nimport PostFooter from &#39;../../components/PostFooterDiscord.astro&#39;\nimport ImageWithCaption from &#39;@components/ImageWithCaption.astro&#39;\n\nThe complexities and costs of data management are a huge challenge. Traditional setups involve a combination of data ingestion, storage, ETL processes, and analytics platforms, leading to increased operational costs, complexity, [85% project failure rate](https://www.techrepublic.com/article/85-of-big-data-projects-fail-but-your-developers-can-help-yours-succeed/), many failure domains, and high turnover.\n\nI can tell you from real-world experience that developer turnover is a real issue. As soon as one complex component gets built out someone on the team inevitably leaves. This is usually because the idea of maintaining such a project is very unappealing.\n\nData processing systems can create data silos, complicate compliance with security standards, and require extensive resources to manage effectively.\n\n&lt;ImageWithCaption width={665} height={664} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/data-managment-simplyfying-your-data-01.webp&#39; alt=&#39;data processing system&#39;/&gt;\n\nThe diagram above illustrates a conventional data pipeline architecture. It starts with various sources like Data Stores, Data Streams, and Applications feeding into an Event Queue, commonly implemented using systems like Kafka.\n\nFrom there, data flows into a Data Lake, such as Snowflake or Athena, where it often undergoes formatting and wrangling to fit the necessary structure. Subsequent stages involve numerous ETL (Extract, Transform, Load) jobs to process large data batches for specific use cases like Data Science, Business Intelligence, Analytics, and Machine Learning. Each of these stages tends to be resource-intensive and time-consuming.\n\nThis brings us to [Streamdal](https://streamdal.com/) which reimagines this workflow by integrating data processing directly into the codebase (Code-Native), streamlining and simplifying the entire data pipeline. This approach aims to bypass the expensive and time-consuming stages of traditional systems, offering a more efficient path for processing large swaths of data.\n\n## What is Streamdal?\n\n[Streamdal](https://streamdal.com/) is an open-source code-native data pipeline platform that leverages WebAssembly ([Wasm](https://webassembly.org/)) to execute data transformations directly within the end user’s code via the [Streamdal SDK](https://docs.streamdal.com/en/core-components/sdk/). This approach enables developers to integrate pipelines directly into their code base. The diagram below illustrates a typical Streamdal workflow.\n\n&lt;ImageWithCaption width={700} height={411} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/data-managment-simplyfying-your-data-02.webp&#39; alt=&#39;typical workflow diagram&#39; caption=&#39;Streamdal Logical Flow&#39;/&gt;\n\nThe [Streamdal console](https://demo.streamdal.com/) pictured below allows developers to delegate the task of creating specific data transformations and rules to various teams. This approach spares developers from repeatedly diving into the codebase to update how data is managed. The UI is a massive time-saver, allowing teams to quickly visualize data flow across the organization and to create and attach pipelines to their chosen data streams with ease.\n\n&lt;ImageWithCaption width={1000} height={505} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/data-managment-simplyfying-your-data-03.webp&#39; alt=&#39;streamdal console&#39; caption=&#39;Streamdal Console&#39;/&gt;\n\nBy moving data pipelines into the end-user’s application via Streamdal SDK, the process of creating and managing pipelines becomes simpler and more efficient. This enhances collaboration among developers, data teams, security experts, and BI professionals, leading to improved outcomes across the board.\n\n&lt;ImageWithCaption width={630} height={874} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/data-managment-simplyfying-your-data-04.webp&#39; alt=&#39;code-native pipeline workflow&#39;/&gt;\n\nThe diagram above illustrates the [code-native pipeline workflow](https://docs.streamdal.com/en/getting-started/how-streamdal-works/) — at the core is the Streamdal platform, comprising the Streamdal Server and UI. These components work in tandem to manage and deploy WebAssembly (Wasm) rules, which are crafted within the Streamdal UI. Once defined, these rules are translated into Wasm and communicated to the end-user application through Streamdal’s SDK.\n\nThis setup is lightweight since the data processing workload is offloaded to the end-user application, minimizing the need for scaling the Streamdal platform itself.\n\nThe table below shows Streamdal [transformation benchmarks](https://github.com/streamdal/streamdal/blob/main/sdks/go/benchmarks_test.go). Wasm performs the transformation via the Streamdal SDK at sub-millisecond speeds. The performance that Wasm provides is a key feature that allows us to put data pipelines inline into the application itself.\n\n&lt;ImageWithCaption width={389} height={188} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/data-managment-simplyfying-your-data-05.webp&#39; alt=&#39;performance&#39;/&gt;\n\nBy processing data in real-time we immediately handle noise and complexity, which streamlines the flow of data into the data lakes and subsequent utilization for Data Science, Business Intelligence, Analytics, and Machine Learning.\n\n## Key Be
33nefits of Streamdal for Data Management\n\n- **Direct Execution of User-Defined Transformations:** The direct execution of Wasm transformations, provides a highly efficient method for data processing. Reducing the need for additional infra.\n- **Scalability and Low Overhead:** The platform’s code-native approach allows for easy scaling by simply increasing instances of the end-user code, without the need for managing separate transformation services.\n- **Simplified Data Transformation:** Streamdal reduces complexity by eliminating traditional ETL processes, streamlining the entire data transformation workflow.\n- **Empowerment of Teams:** Developers, data teams, security, and BI teams can directly implement their transformations and rules, enabling greater control and customization.\n- **Efficient Resource Utilization:** The reduction in dedicated resources for large ETL jobs leads to lower operational costs and more efficient use of infrastructure.\n- **Real-Time Data Monitoring:** For applications where immediate data visibility is crucial, Streamdal enables real-time monitoring of I/O operations.\n- **Reduce Data Lake Cost:** By transforming data in flight we can clean up any unnecessary data and transform the data before it hits our Data Lake. This greatly reduces the volume of data we are moving around in our Data Lake.\n\n### Conclusion\n\nCode-native pipelines present a paradigm shift in data management by streamlining traditional resource-intensive data pipelines. In a data-driven world where agility and responsiveness are key, Code-Native pipelines stand out as an essential tool for any enterprise looking to optimize its data operations and drive meaningful insights.\n\n&lt;PostFooter/&gt;\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Data Management: Simplifying Your Data Pipelines&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-17T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Streamdal reimagines data management with code-native pipelines, reducing ETL complexity, and direct execution of WASM transformations.&quot;],&quot;author&quot;:[0,&quot;Fritz&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/data-management-simplifying-your-data-pipelines-thubmnail.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/fruitz.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;data-pipelines-in-kafkajs.mdx&quot;],&quot;slug&quot;:[0,&quot;data-pipelines-in-kafkajs&quot;],&quot;body&quot;:[0,&quot;\nimport {Image} from \&quot;astro:assets\&quot;;\nimport PostFooter from &#39;../../components/PostFooterDiscord.astro&#39;\nimport VideoFrame from &#39;@components/VideoFrame/VideoFrame.astro&#39;\n\n**tl;dr:** Shim source is [here](https://github.com/streamdal/kafkajs) and a fully working example is [here](https://github.com/streamdal/streamdal-examples/tree/main/kafka/kafkajs-shim). And there is a video walkthrough that you can watch here:\n\n&lt;VideoFrame width=\&quot;560\&quot; height=\&quot;315\&quot; src=\&quot;https://www.youtube.com/embed/11KLh4DbHbo?si=uF3eGNWkVstyR9BT\&quot; title=\&quot;YouTube video player\&quot;/&gt;\n\n\n&lt;div class=&#39;flex items-center justify-center pb-6 mb-3.5 mt-8&#39;&gt;\n    &lt;span  class=&#39;w-1 h-1 bg-white rounded-full mx-2.5&#39;&gt;&lt;/span&gt;\n    &lt;span  class=&#39;w-1 h-1 bg-white rounded-full mx-2.5&#39;&gt;&lt;/span&gt;\n    &lt;span  class=&#39;w-1 h-1 bg-white rounded-full mx-2.5&#39;&gt;&lt;/span&gt;\n&lt;/div&gt;\n\n## Introduction\n\nStreamdal offers an open-source shim for the popular [KafkaJS](https://github.com/tulios/kafkajs) library that allows any application that uses KafkaJS to add sophisticated data pipelines with as little as a few configuration parameters. The shim automatically maps your Kafka-related services such that you can visually create pipelines and attach them to any Kafka operation in seconds. The pipelines are executed in your application code so your data can be transformed inline without being sent to an external platform.\n\n## Backgroun
33d\n\nAt Streamdal we’re on a mission to democratize real-time data pipelines. To that end, we’ve built a lightweight open-source platform with a variety of [language SDKs](https://docs.streamdal.com/en/core-components/sdk/) and [shims](https://docs.streamdal.com/en/core-components/libraries-shims/) to bring this powerful functionality to the widest possible audience.\n\n## Getting Started\n\nLet’s assume you already have a Node Typescript or Javascript application that leverages KafkaJs, if not feel free to clone our example repo [here](https://github.com/streamdal/streamdal-examples) so you can follow along. The example we’ll be using for this walkthrough is under **_kafka/kafkjs-shim._**\n\nThis:\n\n```\n\&quot;dependencies\&quot;: {\n  \&quot;kafkajs\&quot;: \&quot;^2.2.4\&quot;\n}\n```\n\nBecomes this:\n\n```\n\&quot;dependencies\&quot;: {\n  \&quot;@streamdal/kafkajs\&quot;: \&quot;^2.2.6\&quot;\n}\n```\n\nRun **`npm install`** to install the updated dependency.\n\n_It’s important to note that if you run your app now without making any other changes, it will run exactly as before. The Streamdal shim is **unobtrusive** and **will not add any additional behavior** unless you configure it to do so._\n\n## Turn It On\n\nThere are a few ways to enable Streamdal’s data pipeline functionality. The quickest is to add environment variables. If your Node application supports `.env` files, simply add these three environment variables to your `.env` file:\n\n```\nSTREAMDAL_URL=\&quot;localhost:8082\&quot;\nSTREAMDAL_TOKEN=\&quot;1234\&quot;\nSTREAMDAL_SERVICE_NAME=\&quot;localhost\&quot;\n```\n\nAlternatively, you can simply export these environment variables in the shell where you run your Node application, or you can configure Streamdal entirely via code. See the commented [_streamdalConfigs_ code](https://github.com/streamdal/streamdal-examples/blob/main/kafka/kafkajs-shim/src/index.ts).\n\n&gt;You’ll need to set the env vars above to reference the Streamdal platform running in your environment. If you don’t have Streamdal installed you can follow the [quickstart guide](https://docs.streamdal.com/en/getting-started/quickstart/). Or if are using **kafkjs-shim** example, simply run **docker compose up** and that will bring up everything you need, including Kafka and the Streamdal platform_\n\nFire up your app as normal and it will register with your Streamdal instance. Streamdal will automatically map your Kafka operations. If you are using the **_kafkjs-shim_** example, run `npm start`.\n\nHead on over to the Streamdal Console to see an interactive map of your Kafka-related operations. For the **_kafkjs-shim_** example, we’ll open `http://localhost:8080`.\n\n&lt;Image\n    width={800}\n    height={467}\n    src=\&quot;/assets/images/blog/data-pipelines-in-kafkajs.webp\&quot;\n    alt=\&quot;Data Pipelines in KafkaJS\&quot;\n/&gt;\n\nYou’ll see all of your Kafka operations represented along with some useful metrics about activity on each operation. You can click **Start Tail** to take a real-time peek at the data flowing through any given operation:\n\n&lt;Image\n    width={800}\n    height={467}\n    src=\&quot;/assets/images/blog/user-onboard-user-data.webp\&quot;\n    alt=\&quot;Data Pipelines in KafkaJS\&quot;\n/&gt;\n\n&gt;Note that data is sampled and sent to your Streamdal instance **only** when you invoke a Tail operation. Data does **not** leave your application for the purpose of running pipelines. Streamdal ships pipelines to your application and executes them inline using WASM!\n\n## Add A Pipeline\n\nYou can see from the screenshot above that some minimum user registration information is flowing through our Kafka consumers and producers. Let&#39;s click on the **user-onboard-external-user-data** operation and then select **Create new pipeline** in the right-side info drawer. We’ll name the pipeline and add a **Detective** with an empty path and a type of **PII ANY**. This step will locate any kind of Personally Identifiable Information anyw
33here in the operation’s payload.\n\n\n&lt;Image\n    width={800}\n    height={467}\n    src=\&quot;/assets/images/blog/pii-any.webp\&quot;\n    alt=\&quot;pii any\&quot;\n/&gt;\n\nNext let&#39;s add another step to mask the PII we’ve found in the previous step. Set the step type to **Transform**, the transform type to **Mask**, and then choose **Use output of the previous detective step as the path**.\n\n&lt;Image\n    width={800}\n    height={467}\n    src=\&quot;/assets/images/blog/pii-any-step-2.webp\&quot;\n    alt=\&quot;pii any 2\&quot;\n/&gt;\n\nSave the pipeline and then attach it to our **user-onboard-external-user-data** operation by checking the attach option in the side drawer. Next, click **Start Tail** in the side drawer and we’ll see how the data flowing through this operation looks now:\n\n&lt;Image\n    width={800}\n    height={467}\n    src=\&quot;/assets/images/blog/user-onboard-user-data-attached.webp\&quot;\n    alt=\&quot;pii any 2\&quot;\n/&gt;\n\n**_Voila_**, you can see that Streamdal has shipped the attached pipeline to our application and the shim begins executing it immediately! Now any code downstream of this consumer pipeline will receive the transformed data. In this case, it will no longer contain PII.\n\n## Conclusion\n\nThat’s just the tip of the iceberg of the sorts of pipelines you can configure and use with the Streamdal platform. Click around the pipeline creation screen and you’ll see a large variety of pre-loaded **Detective** and **Transformer** types as well as automatic schema inference, schema change detection, arbitrary HTTP requests, and Key/Value lookups.\n\n&lt;PostFooter/&gt;\n\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Data Pipelines in KafkaJS&quot;],&quot;pubDate&quot;:[3,&quot;2024-03-18T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Shim source is here and a fully working example is here. And there is a video walkthrough that you can watch here&quot;],&quot;author&quot;:[0,&quot;Jacob&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/data-pipelines-in-kafkajs.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/jacob.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;how-to-migrating-github-repos-to-a-monorepo.mdx&quot;],&quot;slug&quot;:[0,&quot;how-to-migrating-github-repos-to-a-monorepo&quot;],&quot;body&quot;:[0,&quot;\nimport PostFooter from &#39;@components/PostFooterDiscord.astro&#39;\n\n&gt;To avoid peppering too many opinions in this “how to” guide, I’ve written a separate “opinion-piece” article on monorepos — you can read it here: [Mostly Terrible: The Monorepo](/blog/mostly-terrible-the-monorepo).\n\nSometime in early Dec 2023, our team decided to migrate ~10 public repositories for an OSS project to a monorepo.\n\nThis article provides a “rough” outline for the meatiest parts of the migration.\n\nGood luck! 🤞\n\n### Goals\n\nWe’ve got _**two**_ goals for the migration:\n\n1. Put the contents of the individual repos under a subdir in the monorepo\n2. Inject the original commit log of the individual repos into the monorepo\n\n\n### Repos\n\nWe will be migrating the following repositories to a monorepo `github.com/streamdal/mono`:\n\n1. [`streamdal/server `](https://github.com/streamdal/server)\n    - Golang\n\n2. [`streamdal/console`](https://github.com/streamdal/console)\n    - TypeScript + Deno + React\n\n3. [`streamdal/cli`](https://github.com/streamdal/cli)\n    - Golang\n\n4. [`streamdal/docs`](https://github.com/streamdal/docs)\n   - Astro\n\n5. [`streamdal/wasm`](https://github.com/streamdal/wasm)\n    - Rust + Wasm\n\n6. [`streamdal/protos`](https://github.com/streamdal/protos)\n    - Protobuf schemas for Go, Python, Rust, TS\n\n7. [`streamdal/wasm-detective`](https://github.com/streamdal/wasm-detective)\n    - Rust lib\n\n8. [`streamdal/wasm-transform`](https://github.com/streamdal/wasm-transform)\n    - Rust lib\n\n### Requirements\n\n- We’ll be operating from `/Users/dselans/Code`, referred to as the _“work-dir”_\n- You will need to have `git` and `zsh` (for native `chdir()`) installed locally\n- Most of the migration will be handled by a [migration script](https://raw.githubusercontent.com/streamdal/streamdal/main/scripts/monorepo/migrate.sh)\n- Last, I performed this migration on MacOS (Sonoma) — if you’re using something else, you might need to do some tweaking 🤷‍♂\n\n## Step 1: Layout\n\nYou need to figure out and come up with a directory structure/layout for your new mono repo. This is an extremely important step and if you fuck this up now, it’ll be twice as painful to unfuck this later on.\n\nThis is the layout I used for `streamdal/streamdal` - it is fairly common and non-controversial - it might work for you.\n\n```\n┌── assets               &lt;---- static assets used in monorepo\n│   ├── img\n│   └── ...\n├── apps                 &lt;---- target dir that will contain apps\n│   ├── cli\n│   ├── console                \n│   ├── docs\n│   ├── server\n│   └── ...\n├── docs\n│   ├── install\n│   │\t  ├── docker\n│   │   └── ...\n│\t  ├── instrument\n│   └── ...\n├── libs                 &lt;--- target dir for app dependencies, common/forked libs\n│   ├── protos\n│   ├── wasm\n│   ├── wasm-detective\n│   ├── wasm-transform\n│   └── ...\n├── scripts\n│   ├── install\n│   │\t  ├── streamdal.sh\n│   │   └── ...\n│   └── ...\n├── LICENSE\n├── Makefile\n└── README.md\n```\n\n## Step 2: Prep Work\n\nYou should _probably_ do this during off-hours when folks aren’t updating repos often. If that’s not possible, no big deal, you’ll just have to do some syncing post-migration.\n\nGo through the list of repos and clone them to your _work dir_:\n\n```\n# Change into the work-dir\n$ cd /Users/dselans/Code\n```\n\n```\n# Grab the migration script\n$ curl -o migrate.sh &lt;https://raw.githubusercontent.com/streamdal/streamdal/main/scripts/monorepo/migrate.sh&gt;\n# Clone the \&quot;to-be-migrated\&quot; repos\n$ git clone [email protected]:streamdal/server\n$ ...\n```\n\nThe `migrate.sh` script expects repo dirs to exist\n\nOpen `migrate.sh` in your editor and update the following bits:\n\n1. `MONO_DIR` - specify the target monorepo dir (`mono`)\n2. `BASE_DIRS` - specify the dirs that the script should create (can leave as-is, if the layout above makes sense)\n3. `FILES` - specify what files the script should create / `touch`\n4. Update `REPOS` with a space-separated list of the “to-be-migrated” repos you cloned\n5. Update SUB_DIR with the dir you want the migrated repos to live under\n— ie. If you have `REPOS=\&quot;foo bar\&quot;` , `MONO_DIR=\&quot;mono\&quot;` and `SUB_DIR=\&quot;apps\&quot;` - the “foo” and “bar” repos will be migrated to `./mono/apps/foo` and `./mono/apps/bar`.\n6. Save and exit editor\n\n## Step 3: Migrate!\n\nWe are ready to begin the migration.\n\n```\n# From work-dir\n$ zsh migrate.sh\n```\n\nThe script will attempt to do the following:\n\n1. Create $MONO_DIR and initialize a git repo in it\n2. Make a copy of `../$REPO` as ../$REPO.clone\n3. Perform all further work from `../$REPO.clone` dir\n4. Move `../$REPO.clone/*` into `../$REPO/$SUB_DIR`\n5. Commit and merge changes in `../$REPO.clone/*`\n6. Chdir to `$MONO_DIR` and set `../$REPO.clone` as a remote\n7. Merge `../$REPO`’s main into `$MONO_DIR`’s\n8. Commit and move on to the next $REPO specified in `REPOS=`\n\nThe “meaty” part of the migration is complete.\n\n## Post Monorepo Migration Hot Tipz\n\nPerforming the `git` part of migration is the _first_ step in your “journey”. There will be a handful of other things you’ll need to do to get things into decent shape and ready for production.\n\nHere are some tips to get you started:\n\n1. Repo size is a legitimate concern now, so start by identifying large, dupe, or garbage assets. `du -sh * | grep M` in your monorepo dir. Look for log files, build dirs, `node_modules`, Rust’s target dirs, accidentally checked in Docker-compose volumes, etc.\n2. When you find garbage, don’t forget to add the paths to a `.gitignore`\n3. Unless you have a _large_ number of repos (20+), I would do one repo migration at a time to catch any potential issues and fix them on the spot.\n4. You will probably not like the structure or redecide something and ultimately have to rerun the migration again. `rm -rf $MONO_DIR &amp;&amp; rm -rf *.clone` to start fresh.\n5. Don’t forget to update `README.md`&#39;s in the migrated repos to indicate that the main repo has moved to $XYZ. For good measure, “archive” the repo as well (via repo settings in GitHub).\n6. Attempting to update _everything_ in one go is a huge undertaking and will take longer than you anticipated. Migrate the repos first, finish that, then tackle CI, then tackle READMEs, docs, and so on.\n7. Updating CI will probably be the heaviest lift. You first want to gate the workflows so that they run only when the PR contains changes for `/apps/some-app/*`\n— You can accomplish this by using GitHub’s path filters on a `pull_request` trigger (and on `push` to `main`). Take a look at our [workflows](https://github.com/streamdal/streamdal/tree/main/.github/workflows) for reference.\n\n_**UPDATE 01.2024**: Gotta admit, having everything in one place is pretty nice. Intellij appears to be smart enough to understand that diff subdirs have diff languages — I would’ve imagined it would have problems, at least with indexing. Not bad._\n\n&lt;PostFooter/&gt;&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;How To: Migrating GitHub Repos to a Monorepo&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-03T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Sometime in early Dec 2023, our team decided to migrate ~10 public repositories for an OSS project to a monorepo.&quot;],&quot;author&quot;:[0,&quot;Daniel Selans&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/how-to-migrating-github-repos-to-a-monorepo.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/daniel-selans.jpg&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;lets-rolll.mdx&quot;],&quot;slug&quot;:[0,&quot;lets-rolll&quot;],&quot;body&quot;:[0,&quot;\nimport {Image} from \&quot;astro:assets\&quot;;\n\n## Hi. I am Dan and we are Streamdal.\n\nWe are a [YCombinator (S20)](https://www.ycombinator.com/companies/streamdal) company made up of nerds who enjoy working on complicated, fairly low-level things like using web assembly to run client-side data pipelines. Neat!\n\nJoin us on Discord: [discord.gg/streamdal](https://discord.gg/streamdal)\n\nOver the past three years, we have worked in some seriously dark corners of bleeding-edge tech that often involves figuring out creative solutions for problems that aren’t discussed on StackOverflow, Medium, or in a GitHub issue.\n\nAside from engineering, we’ve also been heavily plugged into the early-stage startup world. We’ve pitched, raised money, held board meetings, and have stayed up countless nights trying to figure out optimal messaging, go-to-market strategies, and whatnot else.\n\n&gt; And this entire time, we’ve mostly kept this stuff to ourselves…\n\n… partly because writing quality content is hard 🤷‍♂\n\nAll of this is to say that we’ve probably been doing a disservice to ourselves and the world by not sharing some of these findings. We’ve benef
33ited greatly from the thousands of articles, posts (and shitposts), we’ve read over our lifetime. **So new year, new us.**\n\nFrom here on out, we’re going to try and share a bit more about the stuff we work on, our AHA moments, our thoughts, and as a result, maybe leave a positive mark on the INTERNET. 😘\n\nTo set expectations, here are some of the topics you’ll see us writing about:\n\n- (Code-native) Pipelines\n- Open source\n- Data eng stuff\n- Web assembly\n- Protobuf\n- gRPC\n- Go\n- Most other “semi-mainstream+ programming languages”\n- Large-scale and/or high-throughput systems\n- Architecture (systems and software)\n- Observability\n- Incidents / outages\n- Data center stuff\n- Startup thoughts\n\nAlright! Let’s do this thing 🤙\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Let’s Rolllllll&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-01T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Hi. I am Dan and we are Streamdal.&quot;],&quot;author&quot;:[0,&quot;Daniel Selans&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/lets-roll.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/daniel-selans.jpg&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;monorepos-version-tag-and-release-strategy.mdx&quot;],&quot;slug&quot;:[0,&quot;monorepos-version-tag-and-release-strategy&quot;],&quot;body&quot;:[0,&quot;\n\nimport PostFooter from &#39;../../components/PostFooterDiscord.astro&#39;\n\nIf additional CI work is the #1 (initial) side-effect of moving to a monorepo, then dealing with versioning, tagging, and releasing is a close #2.\n\nTalking about versioning or tagging strategies is not exciting stuff but seeing as we *just* dealt with it and it’s fresh in my head, I might as well document the solution and help someone retain some sanity as they navigate versioning, tagging and release strategies for their monorepo.\n\n## Some Context\n\nIn late December 2023, we migrated ~10 public repositories to a single monorepo. This effort was documented in these two posts:\n\n- [Mostly Terrible: The Monorepo](/blog/mostly-terrible-the-monorepo)\n- [How To: Migrating GitHub Repos to a Monorepo](/blog/how-to-migrating-github-repos-to-a-monorepo)\n\nAll of the expected stuff happened — CI, docs, scripts, Makefiles, and whatnot had to be updated to work with the new monorepo directory structure. However, one thing we did not anticipate is the amount of time we’ll need to spend on figuring out and fixing our version, tag, and release process.\n\nThis was our list of requirements:\n\n1. Each individual component needs to have its own version.\n2. Each individual component can have its own release process.\n3. It should be possible to have a version for the “whole” monorepo.\n– As in, a single version that can be used as a point-in-time reference for the state of the entire repository.\n\nThis list of demands is fairly normal but as it turns out, it can be a serious pain in the ass. So here follows a strategy that did work for us + the problems you’re likely to hit + workarounds for those problems.\n\n&gt; ***⚠️ The version/tag/release headaches can be avoided if you can use a single, unified version. ⚠️***\n\nBut… we decided against this for a few reasons:\n\n1. We would have to “build the world” every time there is an update to the repo. This is not feasible as some of the components take &gt;15min to build.\n2. We did not want to have a “tag explosion” — the repos are updated fairly often and we would end up with thousands of tags in a very short period.\n3. It would be difficult to “spot” what is considered a “good” tag that represents a fully working project.\n\nUnified versions seem simple at first, however they come with a bunch of caveats and gotchas.\n\n## Our Strategy\n\nThis is the versioning, tagging, and release strategy we settled on:\n\n1. Sub-component CI **PR** workflow is gated for:\n– Creation of a pull request AND\n– Detecting changes ([path filter](https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#onpushpull_requestpull_request_targetpathspaths-ignore)) in `apps/$project/**`\n\n2. Sub-component CI **release** workflow is gated for:\n– Push to `main` (there is no “merge” trigger) AND\n– Detecting changes ([path filter](https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#onpushpull_requestpull_request_targetpathspaths-ignore)) in `apps/$project/**`\n\n3. Release CI workflow will create a tag using the format:\n`apps/$project/vX.Y.Z`\n– This is not val
33id semantic versioning but it is a common approach to tagging multiple components that live within the same repository.\n– Alternatively, using `X.Y.Z-foo` or `X.Y.Z+bar` is indeed a valid semver but it also implies something entirely different than what you intended (for example, `-` is used for signifying a pre-release; `+` can be used to specify build metadata).\n\nThere are a few gotchas.\n\n## Gotcha: Invalid Semver\n\n`some/$project/vX.Y.Z` is NOT valid semver - if your release consists of creating artifacts and pushing them to other registries (such as to crates.io, NPM, or PyPI), you will `DEFINITELY` need to strip the prefix `some/$project/v` before you submit the artifact.\n\nAnd this brings us to **why using pre-release and build tags in semver is NOT going to work** — stripping prefixes is common enough; stripping suffixes — not so much.\n\nFor example, we use [mathieudutour/github-tag-action](https://github.com/mathieudutour/github-tag-action) Github Action to automatically create and/or increase semver based tags. Neat, but the important part is that it has support for `tag_prefix` - meaning if it’s specified, it will look for tags with the given prefix and strip it when interpreting and bumping the semver.\n\nSecondly, the `github-tag-action` also exposes outputs that will contain just the new version, without the prefix. This way you can avoid doing hacky `sed` calls to replace the prefix etc.\n\n## Gotcha: Go Pkg Versions\n\nIf your monorepo contains Go packages, they are likely stored in some subdir, such as `libs/my-go-pkg/*`. The problem here is that `go mod` expects `go.mod` and `go.sum` to be at the root of the repo… which is almost definitely not going to be the case with a monorepo.\n\nYou can get around this by having the path to the go module specified in the tag in the format of `path/to/[email protected]` - and this is another reason why choosing the “prefix” approach is a good idea - it works with Go.\n\nFor an example of how we got this to work, take a look at the [streamdal/streamdal](https://github.com/streamdal/streamdal/) repo and specifically the `lib/protos` and `libs/protos/build/go` subdirs.\n\n&gt; ***⚠️ This functionality is known as a “multi-module repository”.⚠️***\n&gt;\n&gt; *You can read more about it on the [Go Wiki](https://go.dev/wiki/Modules#faqs--multi-module-repositories).*\n\n## Gotcha: Publishing to Deno\n\nOne of the release destinations for our [protos lib](https://github.com/streamdal/streamdal/tree/main/libs/protos) is to [deno.land](https://deno.land/x).\n\nThe usual workflow involves setting up your Github repo to fire a webhook for deno.land whenever a branch or a tag is created.\n\nThe URL for the webhook is: `https://api.deno.land/webhook/gh/streamdal_protos`\n\nBut this won’t work for two reasons:\n\n1. The module you are trying to publish lives in a subdir\n2. The tags in your repo now have a prefix such as `libs/[email protected]`\n\nThankfully, Deno thought ahead of this and the webhook endpoint supports two query params: `subdir` and `tag_prefix`.\n\n&gt; ***⚠️ Docs for deno.land module publishing:***\n&gt; *https://github.com/denoland/deno_registry2/blob/main/API.md*\n\n## Takeaway\n\nOnce you overcome the usual CI-related and organizational hurdles of migrating to a monorepo, you will almost certainly have to deal with versions, tags, or releases.\n\nAnd as illustrated by the “gotchas” — they will almost certainly involve the tag prefix in some way.\n\nBefore looking at other solutions, always look into the possibility of “prefix stripping”.\n\nGood luck!\n\n&lt;PostFooter/&gt;\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Monorepos: Version, Tag, and Release Strategy&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-10T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;If additional CI work is the #1 (initial) side-effect of moving to a monorepo, then dealing with versioning, tagging, and releasing is a close #2.&quot;],&quot;author&quot;:[0,&quot;Daniel Selans&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/monorepos.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/daniel-selans.jpg&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;mostly-terrible-the-monorepo.mdx&quot;],&quot;slug&quot;:[0,&quot;mostly-terrible-the-monorepo&quot;],&quot;body&quot;:[0,&quot;\nimport PostFooter from &#39;../../components/PostFooterDiscord.astro&#39;\nimport {Image} from \&quot;astro:assets\&quot;;\n\n## Pre(-r)amble\n\n_This is an “opinion piece”. For the actual “how-to” guide, read [Migrating Repos on Github to a Monorepo](/blog/how-to-migrating-github-repos-to-a-monorepo)._\n\nI think monorepos are a waste of time, at best, and idiotic, at worst. Usually, it’s both.\n\nOf course, there are exceptions but they are few and far between (ie. node, well-established stack, and/or single lang) and even in those cases, the PROs of the monorepo are up for debate.\n\nAnd when it comes to CONs, there is not much I can add that hasn’t already been said in various articles, posts, and whatnot else, warning people about using monorepos… so I’ll leave you with my top 3 concerns 😄 : CI, access-controls, and versioning.\n\n_SIDENOTE: I get it. Writing hate posts about monorepos is pretty low-effort. But I promise, there’s more to this post than just “hurr-durr monorepos bad”!_\n\nWith that out of the way, let’s get to the meat of this post:\n\n## Reasons for Moving to a Monorepo\n\nLike clockwork, someone in your org will bring up monorepos and how they will solve all of your organizational problems.\n\nIt happens every year, in an architecture Slack channel, over lunch with platform folks, or worse, brought up by the VP of engineering. This will then be followed up by you writing up a three-page doc explaining why it is a really bad idea and what the actual cost of doing this will be. Rinse and repeat every X years.\n\nBut here’s the thing. After &gt;20 years in tech, I think I have finally found the **FIRST** instance where the pain of migrating to a monorepo, is actually worth it _(spoiler alert: it’s the open-source angle)_.\n\n- **Poor optics**\n- **Poor UX / Poor DevEx**\n\nTo understand why those are important to us, you first need to understand the softw
33are our company works on (to some extent).\n\nStreamdal is an open-source, code-native pipeline engine — it allows you to do data wrangling application-side, without the need for traditional pipelines. The toolkit consists of many components and each one has its own repo. There’s a [server](https://github.com/streamdal/server), [console](https://github.com/streamdal/console), [protobuf schemas](https://github.com/streamdal/protos), [wasm artifacts](https://github.com/streamdal/wasm), a [Go SDK](https://github.com/streamdal/go-sdk), [Node SDK](https://github.com/streamdal/node-sdk), [Python SDK](https://github.com/streamdal/python-sdk), [docs](https://github.com/streamdal/docs), _AND_ the main, [\&quot;helper\&quot; repo](https://github.com/streamdal/streamdal).\n\nMost importantly, many of the repos are interdependent — the `console` won&#39;t work without a `server`, the server won&#39;t work without `wasm artifacts` and the whole thing is useless without SDKs.\n\nFinally, the “helper” repo serves as the main “entry point” repo — it is the main repo that we “advertise” and the main place that folks use to interact with the project. It contains install scripts, docs, some examples, and images and that’s about it.\n\nAnd that is the first problem:\n\n### Poor Optics\n\nWhat I mean by “poor optics” is that the “helper” repo does not change often. The only time it gets updated is to either fix the install script or add some docs. It doesn’t have any “code” in it, it’s just some shell scripts, maybe template and markdown files, maybe a `Makefile` and some assets.\n\nThis means that the “helper” repo will ALWAYS have fewer commits than the other repos.\n\n_A comparison of “activity &amp; commits” on GitHub. On the left, is what the “helper” repo looked like before migration. On the right, is what the repo looked like post-migration to a monorepo._\n\n&lt;Image width={800} height={94} src=&#39;/assets/images/blog/mostly-terrible-monorepo-commits.webp&#39; alt=&#39;analyze&#39;/&gt;\n\nAnd I don’t know about you but if I see a `Last updated: 6 months ago` on an Open Source project, I will already form an opinion about the \&quot;liveness\&quot; of it.\n\nOn top of that, because the repo only contains scripts, templates, and assets — Github will display the following in the “Languages” section for your flagship repo.\n\n&lt;Image width={800} height={330} src=&#39;/assets/images/blog/mostly-terrible-monorepo-prev-languages.webp&#39; alt=&#39;languages before monorepo&#39;/&gt;\n\nYeah. Pretty lame. And not only from a vanity standpoint but I personally use the “Languages” section to quickly determine whether the project’s tech is “compatible” with my stack. As in, if my stack does not have any Java apps, should I start now?\n\nFor comparison, this is how this section looks like now that we have migrated to a monorepo:\n\n&lt;Image width={800} height={400} src=&#39;/assets/images/blog/mostly-terrible-monorepo-now-languages.webp&#39; alt=&#39;languages in monorepo&#39;/&gt;\n\nBut besides optics, there is also the problem of:\n\n### Poor UX and DevEx\n\nWhen you’ve got 10+ repos:\n\n- _How do you ensure that your users know which repo to interact with?_\n- _If you want to report a bug and you don’t know if it’s caused by the `server`, `wasm` , `console`, or some other component, where do you submit the issue?_\n- _How do folks contribute?_\n\nCommunity, collaboration, and support are all at the heart of an open-source project.\n\nIn our case, having 10+ inter-dependant repos makes it really difficult for users to figure out where they should submit issues, where (and how) to contribute to the project, or just figure out where to seek support.\n\nOf course, you could keep contribution guidelines in all your README.md&#39;s… hopefully they are all up to date! 😓\n\nFor this reason, it is crucial for an open-source project to provide users with a clear path for how to interact with the project.\n\nA monorepo that contains ALL of the components that your project relies on, is a really good fit for a monorepo. I would argue to say that this single point alone might be worth it for some folks to migrate, but you do you.\n\n### Before Pulling the Trigger\n\nEven if it seems like a monorepo might make sense for you, chances are, it still does not.\n\nThe pain involved in migrating to a monorepo is directly proportional to the maturity, size, and number of traditional repos that are monorepo candidates.\
33n\nIn other words, the more mature your project is, the more languages you have, the more repos you have, the more custom work you have done per repo… the more difficult it will be to perform the migration.\n\nHere are some things to keep in mind before you pull the “migration trigger”:\n\n- **You will need to figure out the structure of the repo**\n— Does the usual `/apps`, `/libs`, `/docs`, .. layout make sense?\n- **You will need to figure out your versioning story**\n— How are individual components versioned? Can you pull off a single unified version?\n- **You will need to update 👏 every 👏 single 👏 CI 👏 job**\n— Among many other pieces, you will need to “gate” the CI jobs — ie. if you are `/apps/server` and a PR was opened for `/apps/client` - you shouldn’t exec server CI tests and so on.\n- **You will need to update every single `README.md`**\n— This is a “duh” but is a serious time-sink. All of the badges, links and in some cases the copy itself will need to be updated.\n- **You will need to figure out if a larger repo size will pose a problem**\n— Will your CI choke having to check out a 500MB+ repo every PR? Do you use multiple CI platforms? Is one massively slower?\n\nOf course, this is not an exhaustive list but these are the immediate issues that I had to deal with when performing our monorepo migration. Most of these have been documented in the [“How-to: Migrating to a Monorepo”](/blog/how-to-migrating-github-repos-to-a-monorepo/) article.\n\n### Conclusion\n\nThe monorepo addresses both of these problems **perfectly**. So with that, on Dec 20th, 2023, we (mostly) completed the migration from multiple repos to a [single (sexy) repo](https://github.com/streamdal/streamdal).\n\nNone of it was particularly challenging — mostly annoying, time-sucking busy work.\n\nThe migration took ~1 week of actual hands-on time. It resulted in writing a [migration script](https://github.com/streamdal/streamdal/blob/main/scripts/monorepo/migrate.sh), running lots of `sed` and `awk`, and getting uncomfortably familiar with GitHub Actions.\n\nAnd still, I think the PROs outweigh the CONs (at least for our **very specific use case**).\n\n_Here are the “before” and “after” pictures of the original repos VS monorepo 🥲_\n\n&lt;Image width={1000} height={492} src=&#39;/assets/images/blog/mostly-terrible-monorepo-vs-repositories.webp&#39; alt=&#39;versus repositories&#39;/&gt;\n\n### Rating\n\n“Mostly Terrible”\n\n…because it turns out monorepos can be useful, just very rarely.\n\n&lt;PostFooter/&gt;\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Mostly Terrible: The Monorepo&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-01T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Like clockwork, someone in your org will bring up monorepos and how they will solve all of your organizational problems.&quot;],&quot;author&quot;:[0,&quot;Daniel Selans&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/mostly-terrible-the-monorepo.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/daniel-selans.jpg&quot;],&quot;highlighted&quot;:[0,true]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;ptotobuf-vs-json-for-your-event-driven-architecture.mdx&quot;],&quot;slug&quot;:[0,&quot;ptotobuf-vs-json-for-your-event-driven-architecture&quot;],&quot;body&quot;:[0,&quot;\nimport ImageWithCaption from &#39;@components/ImageWithCaption.astro&#39;\nimport EngageUs from \&quot;@shared/EngageUs/EngageUs.tsx\&quot;;\nimport PostFooter from &#39;@components/PostFooterDiscord.astro&#39;\n\nChoosing a serialization method for your data can be a confusing task considering the multitude of options out there, such as JSON, MessagePack, AVRO, Thrift, Protobuf, Flat Buffers, etc. If you’re using gRPC or Thrift RPC to communicate between your services, the decision has already been made for you. But for event-driven, you’ll more than likely want to prioritize performance over other considerations.\n\nToday we’ll evaluate two of the most common serialization methods for event-driven systems: Protocol Buffers and JSON. Let’s briefly overview both options:\n\n## Protobuf vs JSON: A Quick Overview\n\n### JSON\n\nJSON is probably the most obvious candidate for most:\n\n- Supported in the standard library of almost any language.\n- Human-readable.\n- No strict schema to deal with.\n\nDue to its ubiquity, dealing with JSON is going to be the easiest option. Even better, you don’t need any tooling to see what’s in the messages. Just pop open kafdrop web UI and view your messages in plain text, easy-peasy! Depen
33ding on your needs and/or point-of-view, the lack of a strictly defined schema can be a positive or a negative. We’ll discuss that further later on.\n\n### Protobuf\n\nProtobuf is the less obvious choice of the two. But it definitely has its strengths:\n\n- Schemas for free\n- Support for binary fields without the need to base64 back and forth\n- Built-in API support with gRPC\n- And most importantly: Speed!\n\nThere is a bit of initial setup work when dealing with protobuf:\n\n1. First, you have to define your events and their schemas ahead of time.\n2. Then you have to use protoc, which isn&#39;t the world&#39;s most intuitive tool, to generate the code needed to create/manipulate protobuf events.\n3. And finally, you have to include that generated code in all of your projects that are handling your events.\n\nThat’s a few more steps than just calling `[json.Marshal()]` and `[json.Unmarshal()]`!\n\n&lt;EngageUs/&gt;\n\n## Protobuf vs JSON: Schemas\n\n### Do I even need a Schema?\n\nPlain JSON is definitely going to be easier and quicker to iterate with. You won’t have to update any definitions, regenerate any code, or re-pull any changes into your code bases. Need a new field? Just add it in service A, and update service B to read the new field.\n\nWe can see that this is probably not going to scale well, however. Having multiple engineers on multiple teams adding, removing, or editing fields at will is going to result in things breaking at some point. It would be much better to have a single source of truth, a clearly defined schema.\n\n### Schema Options for JSON\n\nThe two big players in codifying the schema of a JSON message are [AVRO](https://avro.apache.org/) and [JSONSchema](https://json-schema.org/). The specifics/benefits of these two are outside the scope of this discussion, but we’re including them to let you know that there are options out there.\n\n### Schema Options for Protobuf\n\nHere’s a key benefit of using protobuf, the definitions already define the schemas. No need to layer any other solution on top!\n\nThere are some caveats to be aware of though. Changing the type on a field will break backward compatibility, preventing you from replaying or reading messages generated before the change. It’s best to treat protobuf definitions as additive only. If your field’s data type needs to change, add a new field with the new data type, and “remove” the old field by marking it as deprecated (by appending `[deprecated = true]` to the end of the field definition), or reserving the field number.\n\n## Protobuf vs JSON: Performance\n\nIf you’re going the event-driven architecture route, you’re probably interested in scale and want the best performance you can get out of a serialization method.\n\nLet’s write some benchmarks to get an idea of the speed of both. At Streamdal, we’re a Golang shop, so we’ll use that for our benchmark. If you’re using a different language, its implementations of JSON and protobuf and the resulting performance will be different.\n\nFirst, we’ll define both the JSON and protobuf events. Our JSON event will look like this:\n\n```json\ntype MyJSONMessage struct {\n\tId      string `json:\&quot;id\&quot;`\n\tMessage string `json:\&quot;message\&quot;`\n\tNum     int    `json:\&quot;num\&quot;`\n}\n```\n\nAnd our protobuf event will look like this:\n\n_events/mypbmessage.proto:_\n\n```\nsyntax = \&quot;proto3\&quot;;\n```\n\n```\npackage events;\n\noption go_package = \&quot;github.com/streamdal/pbvsjson/events\&quot;;\n\nmessageMyPBMessage {\n  string id = 1;\n  stringmessage = 2;\nint32 num =\n```\n\nWe’ll compile that proto definition into Go code:\n\n```\n$ protoc -I ./events --go_out=paths=source_relative:events events/*.proto\n```\n\nNow let’s benchmark serializing and deserializing our JSON and protobuf events:\n\n_main_test.go:_\n\n```\npackage main\n```\n\n```go\nimport (\n    \&quot;encoding/json\&quot;\n    \&quot;math\&quot;\n    \&quot;testing\&quot;\n    \&quot;github.com/golang/protobuf/proto\&quot;\n    \&quot;github.com/streamdal/pbvsjson/events\&quot;
33\n)\n\ntype MyJSONMessage struct {\n    Id      string `json:\&quot;id\&quot;`\n    Message string `json:\&quot;message\&quot;`\n    Num     int    `json:\&quot;num\&quot;`\n}\n\nvar msg *MyJSONMessage\nvar pbMsg *events.MyPBMessage\n\nfunc init() {\n    msg = &amp;MyJSONMessage{\n        Id:      \&quot;97ec560f-2b14-4930-9a27-f0427f08951c\&quot;,\n        Message: \&quot;Let&#39;s benchmark!\&quot;,\n        Num:     math.MaxInt32,\n    }\n    pbMsg = &amp;events.MyPBMessage{\n        Id:      \&quot;97ec560f-2b14-4930-9a27-f0427f08951c\&quot;,\n        Message: \&quot;Let&#39;s benchmark!\&quot;,\n        Num:     math.MaxInt32,\n    }\n}\n\nfunc BenchmarkJSON(b *testing.B) {\n    for i := 0; i &lt; b.N; i++ {\n        // First marshal to JSON\n        data, err := json.Marshal(msg)\n        if err != nil {\n            b.Error(err)\n        }\n        // Now unmarshal back into usable struct\n        decoded := &amp;MyJSONMessage{}\n        if err := json.Unmarshal(data, decoded); err != nil {\n            b.Error(err)\n        }\n    }\n}\n\nfunc BenchmarkProtobuf(b *testing.B) {\n    for i := 0; i &lt; b.N; i++ {\n        // First marshal to wire format\n        data, err := proto.Marshal(pbMsg)\n        if err != nil {\n            b.Error(err)\n        }\n        // Now unmarshal back into usable struct\n        tmpMsg := &amp;events.MyPBMessage{}\n        if err := proto.Unmarshal(data, tmpMsg); err != nil {\n            b.Error(err)\n        }\n    }\n}\n```\n\nAnd let’s see what those benchmarks look like:\n\n```\n$ go test -bench=.\ngoos: darwin\ngoarch: amd64\npkg: github.com/streamdal/pbvsjson\ncpu: Intel(R) Core(TM) i5-1038NG7 CPU @ 2.00GHz\nBenchmarkJSON-8                 \t  612400\t      1731 ns/op\nBenchmarkProtobuf-8             \t 3088689\t       385.9 ns/op\n```\n\n_**Protobuf is the clear winner here.**_\n\nLet’s dig a little deeper and see why with some more benchmarks:\n\n_main_test.go:_\n\n```go\nfuncBenchmarkJSON_marshal(b *testing.B) {\nfor i := 0; i &lt; b.N; i++ {\n\t\t// First marshal to JSON\n\t\t_, err := json.Marshal(msg)\nif err != nil {\n\t\t\tb.Error(err)\n\t\t}\n\t}\n}\n```\n\n```go\nfuncBenchmarkProtobuf_marshal(b *testing.B) {\nfor i := 0; i &lt; b.N; i++ {\n\t\t// First marshal to wire format\n\t\t_, err := proto.Marshal(pbMsg)\nif err != nil {\n\t\t\tb.Error(err)\n\t\t}\n\t}\n}\nfuncBenchmarkJSON_unmarshal(b *testing.B) {\n\tdata, err := json.Marshal(msg)\nif err != nil {\n\t\tb.Error(err)\n\t}\nfor i := 0; i &lt; b.N; i++ {\n\t\ttmpMsg := &amp;MyJSONMessage{}\nif err := json.Unmarshal(data, tmpMsg); err != nil {\n\t\t\tb.Error(err)\n\t\t}\n\t}\n}\nfuncBenchmarkProtobuf_unmarshal(b *testing.B) {\n\tdata, err := proto.Marshal(pbMsg)\nif err != nil {\n\t\tb.Error(err)\n\t}\nfor i := 0; i &lt; b.N; i++ {\n\t\ttmpMsg := &amp;events.MyPBMessage{}\nif err := proto.Unmarshal(data, tmpMsg); err != nil {\n\t\t\tb.Error(err)\n\t\t}\n    }\n}\n```\n\nThese additional benchmarks give us:\n\n```\n$ go test -bench=.\ngoos: darwin\ngoarch: amd64\npkg: github.com/streamdal/pbvsjson\ncpu: Intel(R) Core(TM) i5-1038NG7 CPU @ 2.00GHz\nBenchmarkJSON-8                 \t  612400\t      1731 ns/op\nBenchmarkProtobuf-8             \t 3088689\t       385.9 ns/op\nBenchmarkJSON_marshal-8         \t 3975855\t       302.7 ns/op\nBenchmarkProtobuf_marshal-8     \t 8163854\t       145.1 ns/op\nBenchmarkJSON_unmarshal-8       \t  886795\t      1290 ns/op\nBenchmarkProtobuf_unmarshal-8   \t 5829130\t       217.1 ns/op\n```\n\nA 2x advantage when serializing Protobuf vs JSON. Not bad! But the real win here is when unserializing. The overhead of having to parse the text-based JSON format is too insignificant not to ignore. At a 6x speed advantage, protobuf is the way to go for the performance-minded.\n\n## Protobuf vs JSON: Message Size\n\nThe JSON message clocks in at 91 bytes, and the protobuf message at 62 bytes. Our example events are trivial in structure for the sake of grok-ability, but your events will grow along with the number transmitted through your infrastructure. Size should be a consideration.\n\n## Protobuf vs JSON: Observability &amp; Readability\n\nAs a text-based format, JSON wins the readability challenge. That’s not to say Protobuf can’t be made readable though. Whether you build a tool internally or use an open-source tool like [Plumber](https://github.com/streamdal/plumber).\n\n## Conclusions\n\nIn the world of scale, protobuf is definitely the way to go.\n\nSchemas are a great benefit to microservice architecture by giving you a single source of truth for how your events should be structured. It may seem like a bunch of busy work in the beginning, but the benefits lay at scale.\n\nObservability takes a hit when it comes to protobuf, but there are solutions.\n\nFor hobby projects, JSON is a solid choice for readability and development velocity. But in the world of scale, go binary, go protobuf.\n\n&lt;PostFooter/&gt;&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Protobuf vs JSON for Your Event-Driven Architecture&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-08T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Protobuf vs JSON for Your Event-Driven Architecture.&quot;],&quot;author&quot;:[0,&quot;Mark Gregan&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/ptotobuf-vs-json-for-your-event-driven-architecture.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/mark-gregan.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;supercharge-logstash-with-smart-pii-rules.mdx&quot;],&quot;slug&quot;:[0,&quot;supercharge-logstash-with-smart-pii-rules&quot;],&quot;body&quot;:[0,&quot;\nimport PostFooter from &#39;@components/PostFooterDiscord.astro&#39;\nimport ImageWithCaption from &#39;@components/ImageWithCaption.astro&#39;\nimport VideoFrame from &#39;@components/VideoFrame/VideoFrame.astro&#39;\n\nThe sheer volume of logs, data lakes, and real-time information presents a major challenge, especially when managing Personally Identifiable Information (PII).\n\nIn this blog, we explore how integrating Streamdal with Logstash can up your log processing game, with a focus on PII redaction.\n\n## Streamdal: Centralizing PII Redaction Rules\n\nThe Opensource platform Streamdal enables us to centralize the management of PII redaction rules and transformations. By defining these rules in the Streamdal console UI, we can uniformly apply them across our entire log processing pipeline, ensuring consistency and compliance with privacy standards.\n\n&lt;ImageWithCaption width={521} height={797} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/supercharge-logstash-with-smart-pii-rules-01.webp&#39; alt=&#39;Streamdal dashboard&#39; caption=&#39;Streamdal dashboard&#39;/&gt;\n\n## Log Processing Agent\n\nWe chose Go (Golang) to develop our own log processing agent. Go is well-suited for implementing complex, real-time data processing tasks. These agents are designed to identify and redact PII data from logs, ensuring that sensitive information is handled and stored securely.\n\nThe Log Processing Agent is an additional service you deploy alongside your Logstash agents. Which will receive processing rules from the Streamdal console UI, process events sent from Logstash, and then forward the processed logs back to Elasticsearch.\n\n&lt;ImageWithCaption width={700} height={177} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/supercharge-logstash-with-smart-pii-rules-02.
33webp&#39; alt=&#39;Architecture&#39; caption=&#39;Architecture&#39;/&gt;\n\n## Logstash and Elasticsearch\n\nOnce processed by our log-processor, these PII-sanitized logs are forwarded back to Logstash. From there, they enter Elasticsearch, our chosen engine for log storage and analysis. Elasticsearch provides the capability to store, search, and analyze large volumes of data, now with the added assurance that PII elements have been securely redacted.\n\n&lt;ImageWithCaption width={1000} height={416} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/supercharge-logstash-with-smart-pii-rules-03.webp&#39; alt=&#39;Masked PII&#39; caption=&#39;Masked PII&#39;/&gt;\n\n### Deployment Overview\n\nWe briefly cover the deployment in the steps below. I have the full demo environment with Elasticsearch, Logstash, log-processor, and Kibana [here](https://github.com/streamdal/log-processor/tree/main).\n\n1. Deploy [Streamdal](https://github.com/streamdal/streamdal/tree/main/docs/install) in either [Docker](https://github.com/streamdal/streamdal/tree/main/docs/install/docker) or [Kubernetes](https://github.com/streamdal/streamdal/tree/main/docs/install/helm).\n2. Deploy the Streamdal log processor. The example below is what your docker-compose might look like. See the log-processor [repo](https://github.com/streamdal/log-processor) for more details\n\n```dockerfile\n log-processor:\n    container_name: go-app\n    image: streamdal/log-processor\n    environment:\n     - SERVER=streamdal-server:8082\n     - LISTEN_PORT=6000\n     - LOGSTASH_OUTPUT_PORT=logstash-server:7002\n     - STREAMDAL_TOKEN=1234\n    ports:\n      - \&quot;6000:6000\&quot;\n      - \&quot;7002:7002\&quot;\n```\n\n3. Configure Logstash to pass JSON data to the Streamdal log processor.\n\n```\n# Ingest data\ninput {\n  tcp {\n    port =&gt; 5044\n    codec =&gt; json_lines\n  }\n}\n# Pass to streamdal log-processor\noutput {\n  tcp {\n    host =&gt; \&quot;streamdal-log-processor\&quot;\n    port =&gt; 6000\n    codec =&gt; json_lines\n  }\n}\n# Pass back to logstash\ninput {\n  tcp {\n    port =&gt; 7002\n    codec =&gt; json_lines\n  }\n}\n# Store in Elasticsearch\noutput {\n  elasticsearch {\n    hosts =&gt; [\&quot;elasticsearch:9200\&quot;] # Assumes Elasticsearch service is named &#39;elasticsearch&#39; in docker-compose\n    index =&gt; \&quot;processed-logs-%{+YYYY.MM.dd}\&quot; # Customize the index name as needed\n  }\n}\n```\n\n4. Confirm the service is up and running in Streamdal console UI in your browser http://127.0.0.1:8080.\n5. Create a pipeline and attach it via the console UI. In this case, we are using the preexisting IPV4_ADDRESS rule to mask the output.\n\n&lt;ImageWithCaption width={1000} height={508} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/supercharge-logstash-with-smart-pii-rules-04.webp&#39; alt=&#39;Create a Detective Rule To Mask PII IPV4 addresses&#39; caption=&#39;Create a Detective Rule To Mask PII IPV4 addresses&#39;/&gt;\n\n6. Attach the Smart PII pipeline to your Logstash workflow.\n\n&lt;ImageWithCaption width={1000} height={518} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/supercharge-logstash-with-smart-pii-rules-05.webp&#39; alt=&#39;Attach Smart PII to workflow&#39; /&gt;\n\n7. Use the tail function to confirm data is being masked.\n\n&lt;ImageWithCaption width={1000} height={404} loading=&#39;lazy&#39; src=&#39;/assets/images/blog/supercharge-logstash-with-smart-pii-rules-06.webp&#39; alt=&#39;Tail function logs&#39; /&gt;\n\n8. Confirm that the masked data is coming into your Logstash.\n\n### See It In Action\n\nWe’ve put together a video tutorial demonstrating automatic detection and masking of PII in action. Check out the video below!\n\n&lt;VideoFrame width=\&quot;560\&quot; height=\&quot;315\&quot; src=\&quot;https://www.youtube.com/embed/bU7P6-4ppuY?si=-FZRaYZE-mPLKx0c\&quot; title=\&quot;YouTube video player\&quot;/&gt;\n\n#### Conclusion\n\nWith the log processor we developed, we’ve achieved a level of flexibility and control that was previously unattainable. Here are the key takeaways from our journey:\n\n1. **Smart PII Rules:** While Logstash has the concept of filters, they are functionally dumb rules that require handcrafting regex. We can now attach Detective rules to any workflow which will automatically scan for all matching PII and replace, remove, or notify on them.\n2. **Dynamic Rule Deployment:** Our solution allows us to distribute rules across all log-processing agents in real-time. This dynamic approach means we can adapt to new data patterns or regulatory requirements swiftly, without delays.\n3. **Real-Time Monitoring and Metrics:** The system provides the ability to monitor log metrics continuously. This feature ensures that we have our finger on the pulse of our data streams, allowing for proactive responses to emerging trends or anomalies.\n4. **Creating and Managing Real-Time Pipelines:** We now can construct and modify real-time pipelines. This functionality ensures that data processing and transformations occur before the data reaches its final destination in Elasticsearch. This is extremely important when it comes to meeting security requirements.\n5. **Flexible Rule Management:** The ability to attach or detach rules in real-time, without the need for restarts or manual reconfiguration, is a game-changer. It offers unparalleled adaptability in our log processing, ensuring that our systems remain agile and responsive to change.\n6. **Seamless Integration and Scalability:** Integrating Streamdal with our existing Logstash and Elasticsearch infrastru
33cture was seamless. The scalability of this solution means it can grow with our needs, handling increasing volumes of data without sacrificing performance.\n\n&lt;PostFooter/&gt;\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;Supercharge Logstash with Smart PII Rules&quot;],&quot;pubDate&quot;:[3,&quot;2024-01-23T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;Explore Streamdal’s integration with Logstash for PII redaction in logs, ensuring privacy compliance and real-time data processing.&quot;],&quot;author&quot;:[0,&quot;Fritz&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/supercharge-logstash-with-smart-pii-rules.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/fruitz.png&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;what-is-code-native-data-privacy.mdx&quot;],&quot;slug&quot;:[0,&quot;what-is-code-native-data-privacy&quot;],&quot;body&quot;:[0,&quot;\n\nimport ArticleSeparator from &#39;@components/ArticleSeparator.astro&#39;\nimport ImageWithCaption from &#39;@components/ImageWithCaption.astro&#39;\n\nIn this post, I will introduce you to a new concept called &lt;em&gt;“Code-Native Data Privacy”&lt;/em&gt;.\nI will show you what it is, why it matters and most importantly, how it can simplify dealing with most data privacy, compliance and regulation-related issues.\n\n## What is it?\n&lt;em&gt;“Code-Native Data Privacy”&lt;/em&gt; is a modern approach to dealing with data privacy related issues.\n\nRather than standing up separate data infrastructure that is responsible for scanning, cataloging and masking data, the *“code-native”* way would instead have you embed the “scanning, cataloging and masking” parts, directly in your application code.\n\n&lt;ImageWithCaption width={700} height={371} src=&#39;/assets/images/blog/what-is-code-native-data-privacy-01.webp&#39; alt=&#39;Traditional scrub approach VS scrubbing with “code-native data privacy”&#39; caption=&#39;Traditional scrub approach VS scrubbing with “code-native data privacy”&#39; /&gt;\n\n## What’s the point?\n\nUp until now, organizations have been dealing with data privacy the same way for the last 15 years. And it’s inefficient, slow and really costly.\n\nThe traditional approach would look something like this:\n\n1. Begin by **ingesting data** from somewhere\n2. Then **store it** somewhere\n3. Then setup a process to **scan** the ingested data\n4. And then, if the scan finds something bad, **transform** the bad data\n\n&gt; *There are many things wrong here…*\n\n1. For one, it is **complicated** — there are many moving parts and that usually means it’s hard to debug when things don’t work right.\n2. Besides that, it is going to be **slow** — “how slow” depends on how much effort you put into this system but it could be anywhere from 1 minute behind “real-time” to days or weeks.\n3. And, let’s not forget that until the data is properly cleansed, there’s a risk of **violating customer privacy agreements** or, even worse, **falling out of compliance**.\n4. .. and many more like source of truth issues, specialized engineering requirements, potential for network and/or infra-related problems or even organizational issues (who owns what).\n\nThat’s the gist of it and that’s basically how most companies deal with sensitive data. *And there’s no shame in it.* Trying to build something more sophisticated is a painful process — at the very least, you will need to level-up your engineering force, maybe even put together a dedicated data team and *definitely* need to setup additional infrastructure for this solution.\n\nThe point of *“code-native data privacy”* is to **eliminate** every single one of these CONs.\n\nWith *“Code-native”*, you get:\n1. **10x simpler data engineering effort**\n2. **100x lower cost for data infrastructure**\n3. **1000x faster data operations**\n\n.. and, most importantly ..\n\n4. Unlock new **business enablement** avenues\n\n### &gt;&gt;
33A Note on Business Enablement\n\n*“Code-native data privacy”* has several really impressive benefits but the one I think that has a lot business impact is **”[business enablement](https://www.highspot.com/blog/business-enablement/)”**.\n\nWhat I mean by that is that *“code-native”* enables other parts of the business to make immediate, effective changes, without having to involve other parts of the organization.\n\nFor example, before *“code-native”*, whenever your security team received word from compliance about needing to scrub or mask data with certain details, the process of getting the change implemented could take weeks or months.\n\nThe security lead would have to deliberate with a project manager, they would find the appropriate engineering manager, the engineering manager would then determine where to slot this work in and at some point, a TODO ticket would get created … and then a week or month later, compliance requirements would finally land.\n\nWith *“code-native”*, the compliance department has been enabled to implement the changes via the provided Console WebApp, without having to involve anyone else.\n\nThe result — compliance is empowered, developers get time back which results in faster and higher quality deliverables.\n\n*To illustrate — this is how it usually looks like when security asks for a feature from devs:*\n\n&lt;ImageWithCaption width={700} height={534} src=&#39;/assets/images/blog/what-is-code-native-data-privacy-02.webp&#39; alt=&#39;Security asks old way&#39; /&gt;\n\n*And this is what it looks like with “code-native data privacy”:*\n\n&lt;ImageWithCaption width={700} height={439} src=&#39;/assets/images/blog/what-is-code-native-data-privacy-03.webp&#39; alt=&#39;Security asks new way&#39; /&gt;\n\n## What does it actually look like?\nSo far, I’ve been talking about these concepts mostly in the abstract so let’s get to the brass tacks — what does *“code-native data privacy”* actually look like?\n\n*“Code-native data privacy”* is made up of three, distinct parts:\n\n***Library / SDK***\n\nAt the heart of *“code-native”* is a light-weight **software library / SDK** *(software development kit)* that you inject into an application that performs data operations like reading user records from a database or writing invoices to some third-party service.\n\nData operations are referred to as “rules” and they can be anything as simple as “data must contain field `XYZ`” or scanning for PII and masking any detected fields in the data.\n\n&gt; *Think of this **library** as a “pre” and “post”-processor that your application will run **after** it has read data and **before** it writes data.*\n\n&lt;ImageWithCaption width={497} height={541} src=&#39;/assets/images/blog/what-is-code-native-data-privacy-04.webp&#39; alt=&#39;Example app instrumented with SDK validates (and redacts) data before it’s written to MySQL&#39; caption=&#39;Example app instrumented with SDK validates (and redacts) data before it’s written to MySQL&#39; /&gt;\n\n***Console / UI***\n\n&gt; But, business requirements change, compliance laws evolve and eventually, you will need to update your data scrubbing rules.\n\nRather than having to update the application itself with new rules, Streamdal comes with a *Web Application* we call the **Console** from which any team can configure new rules and push them to the libraries in *real-time*.\n\n&lt;ImageWithCaption width={700} height={436} src=&#39;/assets/images/blog/what-is-code-native-data-privacy-05.webp&#39; alt=&#39;Console showing live view of instrumented “kafkacat” application&#39; caption=&#39;Console showing live view of instrumented “kafkacat” application&#39; /&gt;\n\n**Server**\n\nAnd lastly, there is a **Server** component. The *server* is responsible for facilitating communications between the **Console** and applications using the **Libraries/SDKs**.\n\nWhile the *server* is needed for sending updates to SDKs, it is **not** a critical component — it can experience outages and that won’t have any effect on the SDKs.\n\n&gt; Most users would not have to ever interact with the server directly — it is intended to operate as a “set-it-and-forget-it” and requires very little baby-sitting.\n&gt;\n&gt; You can read more about the server’s responsibilities [here](https://docs.streamdal.com/en/core-components/server/#servers-role--functionality).\n\n## Zero-Network, 100% Local\nSaving the best for last — *“code-native data privacy”* operates **entirely locally**.\n\nAs in, data operations executed by the libraries do not have to perform any network calls — everything is done locally, in real-time.\n\nThe result is 100x speed improvements, a dramatically simplified security story and less compliance-related headaches.\n&lt;ArticleSeparator/&gt;
33\n“Code-native data privacy” is completely **free**.\n\nThere’s no weird licensing, no “gotchas”. You can run it yourself right now, fork it or do whatever else you like with it, no strings attached.\n\nTo try it out for yourself, head to the [“Getting Started”](https://github.com/streamdal/streamdal?tab=readme-ov-file#getting-started) section in the main [Streamdal repo](https://github.com/streamdal/streamdal) or play around with the [live demo](https://demo.streamdal.com/).\n&lt;ArticleSeparator/&gt;\n\n## Resources\n- Main repository: https://github.com/streamdal/streamdal\n- Documentation: https://docs.streamdal.com\n- Live (read-only) demo: https://demo.streamdal.com\n- Walkthrough: https://www.youtube.com/watch?v=gjHpSMWoWPs\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;What is Code-Native Data Privacy?&quot;],&quot;pubDate&quot;:[3,&quot;2024-03-08T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;In this post, I will introduce you to a new concept called “Code-Native Data Privacy”.&quot;],&quot;author&quot;:[0,&quot;Daniel Selans&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/what-is-code-native-data-privacy.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/daniel-selans.jpg&quot;],&quot;highlighted&quot;:[0,false]}],&quot;render&quot;:[0,null]}],[0,{&quot;id&quot;:[0,&quot;what-is-upstream-data-security-and-why-is-it-important.mdx&quot;],&quot;slug&quot;:[0,&quot;what-is-upstream-data-security-and-why-is-it-important&quot;],&quot;body&quot;:[0,&quot;\nTo provide adequate data security, businesses need to know which sensitive data is in their system, how sensitive it is, and how it got there. If they don’t know, they can’t guarantee the data is stored and processed appropriately. And if they can’t guarantee that, they face a dizzying array of regulatory, reputational, and financial risks.\n\nTraditional approaches to data security use an “after the fact” model. They scrutinize data stores and databases in search of anomalous data. But by the time they find it, the damage may already have been done. It’s good to know you’ve been storing PII or security tokens in an insecure cloud bucket, but it’d be even better to avoid doing it in the first place.\n\nUpstream data security is an alternative approach that gives businesses a fighting chance to spot sensitive data AND act on it before it reaches its final destination and becomes an issue.\n\nIn this article, we’ll explain what upstream data security is and explore some of the limitations of traditional approaches to data security.\n\n## The Benefits of using Upstream Data Security for an Enterprise\n\nUpstream data security offers companies a proactive and preventive approach to data protection. By identifying and acting on sensitive data at its source, it mitigates the risk of data security failures, reducing the risk of regulatory fines and reputational damage.\n\nUpstream data security also enables real-time visibility into data streams, which can help businesses make informed decisions, better understand the real-world functioning of the application, and automate data security processes.\n\nAdditionally, discovering potential security issues upstream is typically more operationally cost-effective than the alternative. Early detection and action save businesses considerable time and resources that would otherwise be spent identifying and rectifying data security issues further downstream.\n\nIn conclusion, upstream data security offers a proactive, efficient, and cost-effective approach to protecting sensitive data. It enables organizations to identify and manage sensitive data at the source, reducing the risks of data security and compliance failures.\n\n## Upstream Data Security Solutions\n\nTraditionally, upstream data security has been a domain largely managed in-house, driven by the necessity to tailor security measures to the unique data handling requirements of each organization.\n\nThis custom approach explains the absence of vendor-specific solutions in the list, as businesses often opt for internally developed mechanisms to meet their specific security needs.\n\nHere is a list of the most common approaches currently employed by companies to navigate upstream data security challenges:\n\n1. **Home-Grown Solutions (Data Validation Libraries &amp; Custom Middleware)**\n- **Pros:** Tailored to specific business needs;
33 data validation ensures adherence to schemas.\n- **Cons:** Often lack real-time processing; can be rigid and increase latency; significant maintenance required.\n\n2. **API Gateways**\n- **Pros:** Centralize security policies and validate data at the entry point.\n- **Cons:** May not provide deep, real-time data inspection; can become performance bottlenecks; specific to HTTP-like traffic.\n\n3. **Event Bus Architecture**\n- **Pros:** Filters and redacts sensitive data in transit, routing clean data to secure channels.\n- **Cons:** Event-driven architectures dramatically increase architectural and operational complexity; difficult to maintain &amp; debug.\n\n4. **Streamdal**\n- **Pros:** Real-time data observability and transformation; minimal latency with client-side Wasm execution; centralized UI for immediate insights; highly adaptable to dynamic data.\n- **Cons:** Newer technology might require initial learning for integration.\n\n### Conclusion\n\nIn the current landscape of data management, ensuring upstream data security is paramount for organizations across industries.\n\nThe choice of solution hinges on a delicate balance between robust security measures, operational efficiency, and the agility to adapt to rapidly evolving data ecosystems.\n\nAs technology advances, the ability to monitor, validate, and transform data in real-time becomes increasingly critical. Businesses must carefully evaluate their specific needs and constraints to select a security strategy that not only protects their data assets but also supports their growth and innovation objectives.\n\nIn this dynamic environment, the optimal approach is one that offers flexibility, scalability, and the capacity to respond swiftly to new challenges.\n\n&quot;],&quot;collection&quot;:[0,&quot;posts&quot;],&quot;data&quot;:[0,{&quot;title&quot;:[0,&quot;What Is Upstream Data Security, and Why Is It Important?&quot;],&quot;pubDate&quot;:[3,&quot;2024-03-08T00:00:00.000Z&quot;],&quot;description&quot;:[0,&quot;To provide adequate data security, businesses need to know which sensitive data is in their system, how sensitive it is, and how it got…&quot;],&quot;author&quot;:[0,&quot;Daniel Selans&quot;],&quot;thumbnailSrc&quot;:[0,&quot;/assets/images/blog/what-is-upstream-data-security-and-why-is-it-important.webp&quot;],&quot;authorSrc&quot;:[0,&quot;/assets/avatars/daniel-selans.jpg&quot;],&quot;highlighted&quot;:[0,true]}],&quot;render&quot;:[0,null]}]]]}" ssr="" client="only" opts="{&quot;name&quot;:&quot;BlogPostCollection&quot;,&quot;value&quot;:&quot;preact&quot;}"></astro-island> </main>  
33<script type="c2906bfa759604a8e90947b7-text/javascript">(()=>{var e=async t=>{await(await t())()};(self.Astro||(self.Astro={})).load=e;window.dispatchEvent(new Event("astro:load"));})();</script>
33<astro-island uid="ZSQsSg" component-url="/_astro/Footer.CyDbZkFE.js" component-export="default" renderer-url="/_astro/client.CWJTIpmb.js" props="{}" ssr="" client="load" opts="{&quot;name&quot;:&quot;Footer&quot;,&quot;value&quot;:true}" await-children=""><footer><ul class="footer-navigation lg:[grid-area:navigation]"><li class="footer-item footer-nav-link"><a href="/">Company</a></li><li class="footer-item footer-nav-link"><a href="https://docs.streamdal.com">Docs</a></li></ul><ul class="footer-social-container footer-item pb-6 pt-5 lg:basis-1/2 lg:[grid-area:community-links]"><li class="h-fit"><a href="https://github.com/streamdal/streamdal" class="footer-social-button github-button"><svg aria-hidden="true" xmlns="http://www.w3.org/2000/svg" fill="currentColor" viewBox="0 0 24 24" class="h-6 w-6 text-gray-800 dark:text-white"><path fill-rule="evenodd" d="M12 2c-2.4 0-4.7.9-6.5 2.4a10.5 10.5 0 0 0-2 13.1A10 10 0 0 0 8.7 22c.5 0 .7-.2.7-.5v-2c-2.8.7-3.4-1.1-3.4-1.1-.1-.6-.5-1.2-1-1.5-1-.7 0-.7 0-.7a2 2 0 0 1 1.5 1.1 2.2 2.2 0 0 0 1.3 1 2 2 0 0 0 1.6-.1c0-.6.3-1 .7-1.4-2.2-.3-4.6-1.2-4.6-5 0-1.1.4-2 1-2.8a4 4 0 0 1 .2-2.7s.8-.3 2.7 1c1.6-.5 3.4-.5 5 0 2-1.3 2.8-1 2.8-1 .3.8.4 1.8 0 2.7a4 4 0 0 1 1 2.7c0 4-2.3 4.8-4.5 5a2.5 2.5 0 0 1 .7 2v2.8c0 .3.2.6.7.5a10 10 0 0 0 5.4-4.4 10.5 10.5 0 0 0-2.1-13.2A9.8 9.8 0 0 0 12 2Z" clip-rule="evenodd"></path></svg>Read our repo</a></li><li class="h-fit"><a href="https://twitter.com/streamdal" class="footer-social-button x-button text-black"><svg aria-hidden="true" xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 24 24" class="h-6 w-6"><path fill="currentColor" d="M13.8 10.5 20.7 2h-3l-5.3 6.5L7.7 2H1l7.8 11-7.3 9h3l5.7-7 5.1 7H22l-8.2-11.5Zm-2.4 3-1.4-2-5.6-7.9h2.3l4.5 6.3 1.4 2 6 8.5h-2.3l-4.9-7Z"></path></svg>Follow us on X</a></li><li class="h-fit"><a href="https://discord.gg/streamdal" class="footer-social-button discord-button text-black"><svg aria-hidden="true" xmlns="http://www.w3.org/2000/svg" fill="currentColor" viewBox="0 0 24 24" class="h-6 w-6"><path d="M19 5.6c-1.4-.7-2.8-1.1-4.2-1.3l-.5 1c-1.5-.2-3-.2-4.6 0l-.5-1c-1.4.2-2.8.6-4.1 1.3a17.4 17.4 0 0 0-3 11.6 18 18 0 0 0 5 2.5c.5-.5.8-1.1 1.1-1.7l-1.7-1c.2 0 .3-.2.4-.3a11.7 11.7 0 0 0 10.2 0l.4.3-1.7.9 1 1.7c1.9-.5 3.6-1.4 5.1-2.6.4-4-.6-8.2-3-11.5ZM8.6 14.8a2 2 0 0 1-1.8-2 2 2 0 0 1 1.8-2 2 2 0 0 1 1.8 2 2 2 0 0 1-1.8 2Zm6.6 0a2 2 0 0 1-1.8-2 2 2 0 0 1 1.8-2 2 2 0 0 1 1.8 2 2 2 0 0 1-1.8 2Z"></path></svg>Discuss on Discord</a></li><li class="h-fit"><a href="https://www.linkedin.com/company/streamdal" class="footer-social-button linkedin-button text-black"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 18 17" fill="none" class="h-6 w-6"><path fill-rule="evenodd" clip-rule="evenodd" d="M4.38354 2.25507C4.38354 3.21187 3.65502 3.98618 2.51574 3.98618C1.42084 3.98618 0.692318 3.21187 0.714864 2.25507C0.692318 1.25177 1.42082 0.5 2.53756 0.5C3.655 0.5 4.3617 1.25177 4.38354 2.25507ZM0.806462 16.4986V5.35374H4.2701V16.4978H0.806462V16.4986Z" fill="#28203F"></path><path fill-rule="evenodd" clip-rule="evenodd" d="M7.0457 8.90985C7.0457 7.51974 6.9999 6.33466 6.9541 5.35461H9.96259L10.1225 6.88141H10.1909C10.6467 6.17473 11.786 5.10449 13.632 5.10449C15.9106 5.10449 17.6198 6.60874 17.6198 9.88919V16.4994H14.1562V10.3232C14.1562 8.88659 13.6552 7.90725 12.4018 7.90725C11.4443 7.90725 10.875 8.56813 10.6474 9.20576C10.5559 9.43404 10.5108 9.7525 10.5108 10.0724V16.4994H7.0471V8.90985H7.0457Z" fill="#28203F"></path></svg>Connect on LinkedIn</a></li></ul><div class="footer-item px-0 pb-8 pt-8 lg:basis-1/2 lg:py-28 lg:pl-14 lg:pr-16 lg:[grid-area:logo]"><img loading="lazy" src="/assets/logos/streamdal.svg"/><p class="pt-6 font-inter font-medium">The open-source platform for managing code-native data pipelines, offering serverless, real-time, and efficient data handling directly within code.</p></div><div class="footer-item lg:[grid-area:cat]"><div class="flex w-[100%] items-center justify-center xs:items-start  xs:justify-start md:items-center md:justify-center"><img loading="lazy" src="/assets/images/layout/footer-cat.svg" alt="cat image" class="footer-image-cat "/></div></div><div class="footer-item lg:[grid-area:copyright] 2xl:flex 2xl:justify-between 2xl:pt-[17px]"><ul class="flex justify-start gap-14 pt-3.5 lg:gap-7 2xl:order-2 2xl
33:items-center  2xl:p-0"><li class="font-inter text-xs leading-[18px] 2xl:text-base"><a href>Privacy Policy</a></li><li class="font-inter text-xs leading-[18px] 2xl:text-base"><a href>Terms and conditions</a></li></ul><p class="pt-6 font-inter leading-[21px] 2xl:order-1 2xl:p-0">© 2024, Streamdal LLC.</p></div></footer><!--astro:end--></astro-island> 
33<script src="https://cdn.jsdelivr.net/npm/@widgetbot/crate@3" async defer type="c2906bfa759604a8e90947b7-text/javascript">
34      new Crate({
35        server: "1121896696801132636", // Streamdal Community
36        channel: "1166813344011919411", // #⬩general-chat
37        glyph: ["/assets/logos/discord.svg", "50%"],
38        shard: "https://emerald.widgetbot.io",
39        css: `
40      @keyframes pulse {
41        0%, 100% {
42          transform: scale(1);
43        }
44        50% {
45          transform: scale(1.05);
46        }
47      }
48
49      .button {
50        animation: pulse 2s infinite;
51        background-color: #FFD260;
52        margin-bottom: 5px;
53        margin-right: 5px;
54        box-shadow: 0 0 40px #FFD260;
55      }
56      &.open .button {
57        background-color: transparent;
58        box-shadow: none;
59      }
60
61      @media screen and (max-width: 768px) {
62        .button {
63          margin-bottom: 10px;
64          margin-right: 10px;
65        }
66      }
67    `,
68      });
69    </script>
vendor: 143 bytes, line 69
69 <script src="/cdn-cgi/scripts/7d0fa10a/cloudflare-static/rocket-loader.min.js" data-cf-settings="c2906bfa759604a8e90947b7-|49" defer></script>
69<script type="module" src="https://static.cloudflareinsights.com/beacon.min.js/v31edd6df95cf4e85bb4c19e7a9bdbcba1788362987495" integrity="sha512-iIg7k2xntmwu6/uSb5tpc/hySgZc4eoL31yB29W6tJFo2akwjPWcEqnCEdJvGexCL0KEQwVYv5BlowfhVz26hg==" data-cf-beacon='{"version":"2024.11.0","token":"809c58bfcb6149aa9a09d6f11b367de9","r":1,"spa":2}' crossorigin="anonymous"></script>
69
70</body> </html>

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.