1"use strict";(globalThis.webpackChunkv2_docs=globalThis.webpackChunkv2_docs||[]).push([[2799],{49758(e,n,i){i.r(n),i.d(n,{assets:()=>l,contentTitle:()=>o,default:()=>h,frontMatter:()=>t,metadata:()=>s,toc:()=>d});const s=JSON.parse('{"id":"streams/real-time-indexer-with-kafka-stream","title":"Real Time Indexer With Kafka Stream","description":"Real Time Indexer With Kafka Stream with Bitquery Kafka and protobuf streams for low-latency blockchain ingestion in trading systems.","source":"@site/docs/streams/real-time-indexer-with-kafka-stream.md","sourceDirName":"streams","slug":"/streams/real-time-indexer-with-kafka-stream","permalink":"/docs/streams/real-time-indexer-with-kafka-stream","draft":false,"unlisted":false,"editUrl":"https://github.com/bitquery/streaming-data-platform-docs/tree/main/docs/streams/real-time-indexer-with-kafka-stream.md","tags":[],"version":"current","sidebarPosition":2,"frontMatter":{"sidebar_position":2,"title":"Real Time Indexer With Kafka Stream","description":"Real Time Indexer With Kafka Stream with Bitquery Kafka and protobuf streams for low-latency blockchain ingestion in trading systems."},"sidebar":"tutorialSidebar","previous":{"title":"Real Time Solana Data","permalink":"/docs/streams/real-time-solana-data"},"next":{"title":"Building a Trading Bot Using Bitquery Kafka Streams","permalink":"/docs/streams/sniper-trade-using-bitquery-kafka-stream"}}');var r=i(74848),a=i(28453);const t={sidebar_position:2,title:"Real Time Indexer With Kafka Stream",description:"Real Time Indexer With Kafka Stream with Bitquery Kafka and protobuf streams for low-latency blockchain ingestion in trading systems."},o="Real-Time Blockchain Indexer: Build Reliable Indexers with Kafka Streams Instead of Archive Nodes, gRPC, or Webhook Services",l={},d=[{value:"The Challenge of Blockchain Indexing",id:"the-challenge-of-blockchain-indexing",level:2},{value:"The Archive Node Management Problem",id:"the-archive-node-management-problem",level:3},{value:"gRPC Indexers and Webhook-Based Services Limitations",id:"grpc-indexers-and-webhook-based-services-limitations",level:3},{value:"Why Bitquery Kafka Streams Excel for Blockchain Indexing",id:"why-bitquery-kafka-streams-excel-for-blockchain-indexing",level:2},{value:"1. Built-in Data Retention and Replay",id:"1-built-in-data-retention-and-replay",level:3},{value:"2. More Data Than Raw Nodes or Archive Nodes",id:"2-more-data-than-raw-nodes-or-archive-nodes",level:3},{value:"3. Zero Infrastructure Management",id:"3-zero-infrastructure-management",level:3},{value:"4. Enterprise-Grade Reliability",id:"4-enterprise-grade-reliability",level:3},{value:"5. Direct Access, No Limitations",id:"5-direct-access-no-limitations",level:3},{value:"6. Both Raw and Decoded Data",id:"6-both-raw-and-decoded-data",level:3},{value:"Getting Started with Bitquery Kafka-Based Indexing",id:"getting-started-with-bitquery-kafka-based-indexing",level:2},{value:"Tutorial Tidbits: Building Real-Time Indexers with Bitquery Kafka",id:"tutorial-tidbits-building-real-time-indexers-with-bitquery-kafka",level:2},{value:"Why Bitquery Kafka is Ideal for Blockchain Indexing",id:"why-bitquery-kafka-is-ideal-for-blockchain-indexing",level:3},{value:"Quick Tips for Indexer Development",id:"quick-tips-for-indexer-development",level:3}];function c(e){const n={a:"a",code:"code",h1:"h1",h2:"h2",h3:"h3",header:"header",li:"li",ol:"ol",p:"p",strong:"strong",ul:"ul",...(0,a.R)(),...e.components};return(0,r.jsxs)(r.Fragment,{children:[(0,r.jsx)(n.header,{children:(0,r.jsx)(n.h1,{id:"real-time-blockchain-indexer-build-reliable-indexers-with-kafka-streams-instead-of-archive-nodes-grpc-or-webhook-services",children:"Real-Time Blockchain Indexer: Build Reliable Indexers with Kafka Streams Instead of Archive Nodes, gRPC, or Webhook Services"})}),"\n",(0,r.jsx)(n.p,{children:"Building a real-time blockchain indexer is one of the most challenging infrastructure tasks in the process of tracking trades, monitoring token transfers, parsing internal calls, or building a comprehensive on-chain analytics platform. If you decide to run your own archive node, you need to plan for big SSD storage (multiple TBs), fast disks / high IOPS, and robust backup/monitoring. If not, you might be looking for need reliable, low-latency access to blockchain data at scale."}),"\n",(0,r.jsx)(n.p,{children:"We know about popular approaches like running your own archive nodes, setting up gRPC indexers with Geyser plugins (for Solana), using webhook-based services like Helius, relying on third-party RPC providers, or using graph-based indexers."}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"So why do we need another option?"})}),"\n",(0,r.jsx)(n.h2,{id:"the-challenge-of-blockchain-indexing",children:"The Challenge of Blockchain Indexing"}),"\n",(0,r.jsx)(n.p,{children:"Blockchain indexing requires processing massive volumes of data in real-time. Whether you're building a custom indexer for transaction traces, internal transactions, or on-chain data extraction, consider these requirements:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"High Throughput"}),": Ethereum processes thousands of transactions per block, while Solana can handle hundreds of thousands per second"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Zero Data Loss"}),": Missing a single transaction can break your indexer's consistency"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Low Latency"}),": For trading bots, MEV applications, and real-time dashboards, every millisecond counts"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Data Completeness"}),": You need both raw blockchain data and enriched, decoded information\u2014including internal transactions that don't emit events"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Reliability"}),": Your indexer must handle network issues, node failures, and data gaps gracefully"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Historical Backfilling"}),": You need to process historical blocks while maintaining live subscription to new blocks"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"Traditional indexing approaches struggle with these requirements:"}),"\n",(0,r.jsx)(n.h3,{id:"the-archive-node-management-problem",children:"The Archive Node Management Problem"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"What is an Archive Node?"})}),"\n",(0,r.jsx)(n.p,{children:"An archive node requires enabling all historical state data:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.code,{children:"--pruning=archive"}),": Maintains all states in the state-trie (not just recent blocks)"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.code,{children:"--fat-db=on"}),": Roughly doubles storage by storing additional information to enumerate all accounts and storage keys"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.code,{children:"--tracing=on"}),": Enables transaction tracing by default for EVM traces"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:'This trades massive disk space for expensive computation\u2014essentially a full node with a "super heavy cache" enabled.'}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"The Infrastru
1cture Reality"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:"Storage size growth is massive: Running an archive node often requires many terabytes (for Ethereum, often > 10 TB), which grows over time."}),"\n",(0,r.jsx)(n.li,{children:"Disk performance / IOPS bottlenecks: As chain history grows, read/write performance becomes critical; archive nodes tend to be much slower unless powerful SSDs or optimized storage are used."}),"\n",(0,r.jsx)(n.li,{children:"Synchronization time is huge / resource-intensive: Bootstrapping (full sync) can take days or weeks; replaying chain history is compute-heavy."}),"\n",(0,r.jsx)(n.li,{children:"Maintenance overhead & cost: Archive nodes often require dedicated hardware, monitoring, careful storage planning. This makes them costly and hard to manage for small teams or projects."}),"\n",(0,r.jsx)(n.li,{children:"Operational complexity / configuration risk: Proper config (e.g. pruning/\u201cgcmode=archive\u201d, snapshot management, backups, disk planning) is necessary \u2014 misconfig can lead to data loss or unusable node."}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"The Bottom Line"})}),"\n",(0,r.jsxs)(n.p,{children:["Running an archive node is not a matter of hours or days\u2014it's a matter of ",(0,r.jsx)(n.strong,{children:"weeks"})," even with enterprise hardware. The infrastructure requirements are substantial:"]}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Storage"}),": Nearly 2TB of fast SSD storage (and growing)"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Time"}),": Weeks of continuous syncing"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Performance"}),": Degrades significantly as the database grows"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Maintenance"}),": Constant monitoring and intervention required"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"Bitquery Kafka streams eliminate all of these challenges by providing pre-synced, maintained archive node data through a managed streaming service."}),"\n",(0,r.jsx)(n.h3,{id:"grpc-indexers-and-webhook-based-services-limitations",children:"gRPC Indexers and Webhook-Based Services Limitations"}),"\n",(0,r.jsx)(n.p,{children:"Relying on gRPC indexers (like Solana's Geyser plugin approach), webhook-based services (like Helius), or third-party RPC providers introduces different problems that Bitquery Kafka streams solve:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"gRPC Complexity"}),": Setting up gRPC indexers requires running validators with plugins (like Geyser for Solana), which is resource-intensive and complex\u2014Bitquery Kafka eliminates this need"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Webhook Reliability"}),": Webhook-based services can miss events during downtime, have delivery failures, and lack replay capabilities\u2014Bitquery Kafka's retention solves this"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Rate Limiting"}),": Most providers enforce strict rate limits that can throttle your indexing speed\u2014Bitquery Kafka has no rate limits"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Bandwidth Costs"}),": Many providers charge based on data transfer, making high-volume indexing expensive\u2014Bitquery Kafka offers predictable pricing without bandwidth charges"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Reliability Issues"}),": RPC endpoints, gRPC streams, and webhooks can go down, rate-limit you, or provide inconsistent data\u2014Bitquery Kafka provides enterprise-grade reliability"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Data Gaps"}),": If your indexer crashes or loses connection, you may miss transactions with no way to replay\u2014Bitquery Kafka's 24-hour retention allows you to replay missed data"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Transaction Trace Limitations"}),": Many services don't provide full transaction traces or internal transaction data\u2014Bitquery Kafka includes comprehensive transaction data"]}),"\n"]}),"\n",(0,r.jsx)(n.h2,{id:"why-bitquery-kafka-streams-excel-for-blockchain-indexing",children:"Why Bitquery Kafka Streams Excel for Blockchain Indexing"}),"\n",(0,r.jsx)(n.p,{children:"Bitquery's Kafka streams are designed as an alternative to running your own archive nodes, setting up gRPC indexers, using webhook-based services, or relying on RPC providers for blockchain indexing. Unlike traditional indexing approaches (self-hosted indexers, archive node-based indexing, gRPC indexers with Geyser plugins, or webhook services), Bitquery's Kafka streams provide several critical advantages:"}),"\n",(0,r.jsx)(n.h3,{id:"1-built-in-data-retention-and-replay",children:"1. Built-in Data Retention and Replay"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Bitquery Kafka streams' retention mechanism is a game-changer for blockchain indexing."})}),"\n",(0,r.jsx)(n.p,{children:"Unlike RPC providers or WebSocket subscriptions that lose data on disconnect, Bitquery's Kafka streams retain messages for 24 hours. This means:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Data Loss"}),": If your indexer crashes or needs to restart, you can resume from where you left off"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Gap Recovery"}),": You can replay messages from any point within the retention window"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Testing and Debugging"}),": You can reprocess historical data to test your indexing logic"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Checkpoint Management"}
1),": Bitquery's Kafka consumer groups track your position, ensuring you never miss a message"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"This retention capability is especially valuable when:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:"Your indexer needs maintenance or updates"}),"\n",(0,r.jsx)(n.li,{children:"You discover a bug and need to reprocess data"}),"\n",(0,r.jsx)(n.li,{children:"Network issues cause temporary disconnections"}),"\n",(0,r.jsx)(n.li,{children:"You want to test new indexing logic against recent historical data"}),"\n"]}),"\n",(0,r.jsx)(n.h3,{id:"2-more-data-than-raw-nodes-or-archive-nodes",children:"2. More Data Than Raw Nodes or Archive Nodes"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Bitquery's Kafka streams provide more enriched data than raw blockchain nodes or archive nodes."})}),"\n",(0,r.jsx)(n.p,{children:"Raw nodes and archive nodes give you basic transaction data, but Bitquery's streams include:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Decoded Data"}),": Smart contract calls are already decoded using ABI information"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Transaction Traces"}),": Full transaction traces and internal transaction data without needing debug_traceBlockByNumber"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Enriched Metadata"}),": Token names, symbols, decimals, and USD values are included"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Protocol-Specific Parsing"}),": DEX trades, real-time balances, liquidity pool changes, and protocol events are pre-parsed"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Internal Transactions"}),": Native ETH transfers and internal calls that don't emit events are included"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Cross-Chain Consistency"}),": Same data structure across all supported blockchains"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Both Raw and Decoded"}),": Access to both raw blockchain data and enriched, structured formats"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"This means you spend less time on block parsing, transaction trace extraction, and data processing, and more time building features. Instead of:"}),"\n",(0,r.jsxs)(n.ol,{children:["\n",(0,r.jsx)(n.li,{children:"Fetching raw transaction data from archive nodes"}),"\n",(0,r.jsx)(n.li,{children:"Decoding function calls and parsing calldata"}),"\n",(0,r.jsx)(n.li,{children:"Parsing event logs"}),"\n",(0,r.jsx)(n.li,{children:"Extracting internal transactions that don't emit events"}),"\n",(0,r.jsx)(n.li,{children:"Looking up token metadata"}),"\n",(0,r.jsx)(n.li,{children:"Calculating USD values"}),"\n",(0,r.jsx)(n.li,{children:"Building historical backfilling pipelines"}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"You receive all of this pre-processed and ready to use in a unified data feed."}),"\n",(0,r.jsx)(n.h3,{id:"3-zero-infrastructure-management",children:"3. Zero Infrastructure Management"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"You don't need to manage any nodes or infrastru
1cture."})}),"\n",(0,r.jsx)(n.p,{children:"With Bitquery's Kafka streams:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Archive Node Setup"}),": No need to sync, maintain, or upgrade archive nodes (which require significantly more resources than full nodes)"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No gRPC Indexer Configuration"}),": No need to set up Geyser plugins or validator-level indexing for Solana"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Webhook Infrastructure"}),": No need to build webhook endpoints or handle webhook delivery failures"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Bandwidth Management"}),": All data transfer happens through Bitquery Kafka, with no per-request bandwidth limits"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Scaling Headaches"}),": Bitquery Kafka handles the scaling automatically"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Maintenance Windows"}),": Bitquery manages uptime, redundancy, and failover"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"This is particularly important for indexing because:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Bandwidth Efficiency"}),": Traditional RPC-based indexing can consume massive bandwidth. With Bitquery Kafka, you consume data once and process it efficiently"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Cost Predictability"}),": Direct Kafka access pricing means no surprise bandwidth bills"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Focus on Logic"}),": Spend your time building indexing logic, not managing infrastructure"]}),"\n"]}),"\n",(0,r.jsx)(n.h3,{id:"4-enterprise-grade-reliability",children:"4. Enterprise-Grade Reliability"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Bitquery's Kafka infrastructure is built for mission-critical blockchain indexing."})}),"\n",(0,r.jsx)(n.p,{children:"Unlike RPC providers that can go down or rate-limit you, Bitquery's Kafka streams provide:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"At-Least-Once Delivery"}),": Guarantees that every message is delivered at least once"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Automatic Failover"}),": If one broker fails, others take over seamlessly"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Consumer Groups"}),": Multiple consumers can share the load, with automatic rebalancing"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Partitioning"}),": Data is distributed across partitions for parallel processing"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"For blockchain indexing, this means:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Lost Transactions"}),": Even if your consumer crashes, messages are retained and can be replayed"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Horizontal Scaling"}),": Add more consumer instances to process data faster"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Fault Tolerance"}),": Your indexing system can survive individual component failures"]}),"\n"]}),"\n",(0,r.jsx)(n.h3,{id:"5-direct-access-no-limitations",children:"5. Direct Access, No Limitations"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Bitquery's Kafka pricing model is designed for high-volume indexing."})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Bandwidth Limits"}),": Consume as much data as you need without worrying about rate limits"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Request Limits"}),": Unlike RPC providers, there are no per-second request caps"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Predictable Pricing"}),": Direct Kafka access means predictable costs, not variable bandwidth charges"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"High Throughput"}),": Process millions of transactions per day without throttling"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"This is crucial for indexing because:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Full Chain Coverage"}),": Index every transaction, not just a sample"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Real-Time Processing"}),": Keep up with blockchain transaction rates"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No Throttling"}),": Process data at your own pace without artificial limits"]}),"\n"]}),"\n",(0,r.jsx)(n.h3,{id:"6-both-raw-and-decoded-data",children:"6. Both Raw and Decoded Data"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Access to both raw blockchain data and enriched, decoded formats."})}),"\n",(0,r.jsx)(n.p,{children:"Bitquery provides:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Raw Data Streams"}),": Access to raw blocks, transactions, and logs for custom processing"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Decoded Data Streams"}),": Pre-parsed transactions with decoded function calls and events"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Protocol-Specific Topics"}),": Separate topics for DEX trades, token transfers, and transactions"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Flexible Consumption"}),": Choose the level of data processing that fits your needs"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"This flexibility allows you to:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Start Simple"}),": Use decoded data for quick prototyping"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Go Deep"}),": Switch to raw data when you need custom parsing"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Mix and Match"}),": Use different topics for different parts of your indexer"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:"Build robust indexing pipelines that:"}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:"Never lose data (thanks to retention)"}),"\n",(0,r.jsx)(n.li,{children:"Can replay and reprocess (for data quality)"}),"\n",(0,r.jsx)(n.li,{children:"Handle historical backfilling while maintaining live subscription"}),"\n",(0,r.jsx)(n.li,{children:"Scale horizontally (with consumer groups)"}),"\n",(0,r.jsx)(n.li,{children:"Extract on-chain data without running archive nodes"}),"\n"]}),"\n",(0,r.jsx)(n.h2,{id:"getting-started-with-bitquery-kafka-based-indexing",children:"Getting Started with Bitquery Kafka-Based Indexing"}),"\n",(0,r.jsx)(n.p,{children:"Building a real-time indexer with Bitquery's Kafka streams is straightforward:"}),"\n",(0,r.jsxs)(n.ol,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Get Kafka Access"}),": Contact Bitquery sales by filling the ",(0,r.jsx)(n.a,{href:"https://bitquery.io/forms/api",children:"form on the website"})," for Kafka credentials"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Choose Your Topics"}),": Select the topics that match your indexing needs. List is available ",(0,r.jsx)(n.a,{href:"/docs/streams/kafka-streaming-concepts/#complete-list-of-topics",children:"here"})]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Set Up Consumers"}),": Create Kafka consumers with proper offset management"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Process Messages"}),": Parse protobuf messages and update your index"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Handle Failures"}),": Use Bitquery Kafka's 24-hour retention to recover from crashes"]}),"\n"]}),"\n",(0,r.jsx)(n.h2,{id:"tutorial-tidbits-building-real-time-indexers-with-bitquery-kafka",children:"Tutorial Tidbits: Building Real-Time Indexers with Bitquery Kafka"}),"\n",(0,r.jsx)(n.h3,{id:"why-bitquery-kafka-is-ideal-for-blockchain-indexing",children:"Why Bitquery Kafka is Ideal for Blockchain Indexing"}),"\n",(0,r.jsx)(n.p,{children:"Bitquery Kafka streams provide several advantages over archive node-based indexing, gRPC indexers, webhook-based services, or RPC-based indexing that make them perfect for building real-time blockchain indexers:"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"1. Data Retention for Reliability"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:["Bitquery Kafka streams retain messages for ",(0,r.jsx)(n.strong,{children:"24 hours"}),", allowing you to recover from crashes or restarts"]}),"\n",(0,r.jsx)(n.li,{children:"If your indexer goes down, you can resume from the last processed offset"}),"\n",(0,r.jsx)(n.li,{children:"Unlike RPC providers, you don't need to worry about missing transactions during downtime"}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"2. More Data Than Raw Nodes or Archive Nodes"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:["Bitquery's Kafka streams include ",(0,r.jsx)(n.strong,{children:"decoded smart contract calls"})," and ",(0,r.jsx)(n.strong,{children:"enriched metadata"})]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Transaction traces and internal transactions"})," included without needing debug_traceBlockByNumber"]}),"\n",(0,r.jsx)(n.li,{children:"Pre-parsed DEX trades, token transfers, and protocol events"}),"\n",(0,r.jsxs)(n.li,{children:["Both ",(0,r.jsx)(n.strong,{children:"raw and decoded data"})," available in separate topics"]}),"\n",(0,r.jsx)(n.li,{children:"USD values and token metadata included automatically"}),"\n",(0,r.jsx)(n.li,{children:"Native ETH transfers and internal
1calls that don't emit events are captured"}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"3. Zero Infrastructure Management"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No archive node setup required"}),": Bitquery manages all blockchain nodes (including archive nodes)"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No gRPC indexer configuration"}),": No need for Geyser plugins or validator-level indexing"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No webhook infrastructure"}),": No webhook endpoints or delivery handling needed"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No bandwidth limits"}),": Direct Kafka access means no per-request throttling"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No scaling headaches"}),": Kafka handles horizontal scaling automatically"]}),"\n",(0,r.jsx)(n.li,{children:"Focus on your indexing logic and on-chain data extraction, not infrastru
1cture maintenance"}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"4. Cost-Effective for High-Volume Indexing"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Predictable pricing"}),": Direct Kafka access, not variable bandwidth charges"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"No rate limits"}),": Process millions of transactions per day without throttling"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Efficient consumption"}),": Consume data once and process it multiple times if needed"]}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"5. Enterprise-Grade Reliability"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"At-least-once delivery"}),": Guarantees no message loss"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Automatic failover"}),": Seamless handling of broker failures"]}),"\n",(0,r.jsxs)(n.li,{children:[(0,r.jsx)(n.strong,{children:"Consumer groups"}),": Share load across multiple indexer instances"]}),"\n"]}),"\n",(0,r.jsx)(n.h3,{id:"quick-tips-for-indexer-development",children:"Quick Tips for Indexer Development"}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Handling Duplicates:"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:"Messages may have duplicates in Kafka topics"}),"\n",(0,r.jsx)(n.li,{children:"Implement idempotent processing: track processed transaction hashes"}),"\n",(0,r.jsx)(n.li,{children:"Use a fast lookup store to check if already processed"}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Recovery After Crashes:"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsx)(n.li,{children:"Bitquery Kafka's retention window (24 hours) allows you to replay recent data"}),"\n",(0,r.jsx)(n.li,{children:"Store your processing state aka message offset and partition details."}),"\n",(0,r.jsx)(n.li,{children:"On restart, you can seek to a specific offset if needed"}),"\n",(0,r.jsx)(n.li,{children:"This is a major advantage over RPC providers, gRPC indexers, and webhook-based services, which don't offer replay capabilities"}),"\n"]}),"\n",(0,r.jsx)(n.p,{children:(0,r.jsx)(n.strong,{children:"Choosing the Right Topic:"})}),"\n",(0,r.jsxs)(n.ul,{children:["\n",(0,r.jsxs)(n.li,{children:["Use ",(0,r.jsx)(n.code,{children:"*.transactions.proto"})," for comprehensive transaction indexing"]}),"\n",(0,r.jsxs)(n.li,{children:["Use ",(0,r.jsx)(n.code,{children:"*.dextrades.proto"})," for DEX-specific indexing (faster, less data)"]}),"\n",(0,r.jsxs)(n.li,{children:["Use ",(0,r.jsx)(n.code,{children:"*.tokens.proto"})," for token transfer indexing"]}),"\n",(0,r.jsxs)(n.li,{children:["Use ",(0,r.jsx)(n.code,{children:"*.broadcasted.*"})," topics for mempool-level data (lower latency)"]}),"\n"]})]})}function h(e={}){const{wrapper:n}={...(0,a.R)(),...e.components};return n?(0,r.jsx)(n,{...e,children:(0,r.jsx)(c,{...e})}):c(e)}},28453(e,n,i){i.d(n,{R:()=>t,x:()=>o});var s=i(96540);const r={},a=s.createContext(r);function t(e){const n=s.useContext(a);return s.useMemo(function(){return"function"==typeof e?e(n):{...n,...e}},[n,e])}function o(e){let n;return n=e.disableParentContext?"function"==typeof e.components?e.components(r):e.components||r:t(e.components),s.createElement(a.Provider,{value:n},e.children)}}}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.