PageSourceSearch

https://picovoice.ai/component---src-pages-guide-voice-agents-index-js-a7b2015634746fd6f4d5.js

js picovoice.ai collected 2026-10-01 11:44:34 UTC 15,401 bytes, 1 lines download raw bytes

1"use strict";(self.webpackChunkpicovoice=self.webpackChunkpicovoice||[]).push([[1814],{15696:function(e,t,n){n.d(t,{A:function(){return c}});var a=n(96540),i=n(85902),o="faq-module--is_open--afc91",r=n(17437);const s=e=>"string"==typeof e||"number"==typeof e?e.toString():e&&"boolean"!=typeof e?Array.isArray(e)?e.map(s).join(""):e.props&&e.props.children?s(e.props.children):"":"";var c=e=>{const{0:t,1:n}=(0,a.useState)(null),c=(0,a.useMemo)(()=>({"@context":"https://schema.org","@type":"FAQPage",mainEntity:e.questions.map(e=>({"@type":"Question",name:e.question,acceptedAnswer:{"@type":"Answer",text:s(e.answer)}}))}),[]);return(0,r.Y)("div",{className:"faq-module--faq_list--ed475"},(0,r.Y)(i.Helmet,null,(0,r.Y)("script",{type:"application/ld+json"},JSON.stringify(c))),e.questions.map((e,a)=>{let i=e.question.toLowerCase().replace(/\s+/g,"-").replace(/\//g,"-");i.endsWith("?")&&(i=i.slice(0,-1));const s=t===a;return(0,r.Y)("div",{key:a,className:"faq-module--qa_item--bafd3",id:i},(0,r.Y)("div",{className:"faq-module--question--2a76c",onClick:()=>n(e=>e===a?null:a)},(0,r.Y)("div",{className:"faq-module--faq_plus--b1bbb"+(s?` ${o}`:"")},"+"),(0,r.Y)("div",null,e.question)),(0,r.Y)("div",{className:"faq-module--answer--e86b1"+(s?` ${o}`:"")},e.answer))}))}},68083:function(e,t,n){n.d(t,{A:function(){return o}});var a=n(17437);const i={view:"Picovoice's View",example:"Picovoice Example"};var o=e=>(0,a.Y)("div",{className:"guide_callout-module--guide_callout--48027"+("example"===e.variant?" guide_callout-module--example--a9e05":"")},(0,a.Y)("div",{className:"guide_callout-module--label--93ec5"},i[e.variant]),(0,a.Y)("div",{className:"guide_callout-module--text--f0d22"},e.children))},98914:function(e,t,n){n.d(t,{cN:function(){return o},j3:function(){return r},mv:function(){return i},z$:function(){return a}});const a="/guide/voice-agents/",i=[{number:1,slug:"how-voice-agents-work",shortTitle:"How Voice Agents Work",title:"How Voice Agents Work",description:"The anatomy of a voice agent, the Listen-Understand-Reason-Respond-Speak loop, and why real-time coordination is the hard problem. Start here.",part:"Part I: Foundations"},{number:2,slug:"build-approaches",shortTitle:"Build Approaches",title:"Build Approaches: API, Framework, or Components",description:"Managed voice-agent APIs vs orchestration frameworks vs component-level builds, c
1ompared on control, latency floor, cost at scale, and lock-in.",part:"Part I: Foundations"},{number:3,slug:"voice-ux-latency-turn-taking",shortTitle:"Latency and Turn-Taking",title:"Voice UX: Latency, Turn-Taking, and Barge-In",description:"Per-stage latency budgets, end-of-turn detection, interruption handling, and acknowledgment strategies that keep conversations natural.",part:"Part I: Foundations"},{number:4,slug:"reasoning-orchestration",shortTitle:"Reasoning and Orchestration",title:"Reasoning and Orchestration",description:"Intent handling, dialog state, function calling, retrieval grounding, and when speech-to-intent beats a full LLM path.",part:"Part I: Foundations"},{number:5,slug:"telephony-transport",shortTitle:"Telephony and Transport",title:"Telephony and Real-Time Transport",description:"PSTN, SIP, and WebRTC paths, codecs, media gateways, and session lifecycle for phone-based agents.",part:"Part II: Getting to Production"},{number:6,slug:"on-device-voice-agents",shortTitle:"On-Device Voice Agents",title:"On-Device Voice Agents",description:"Why inference location is an architecture decision, the full on-device stack from wake word to synthesis, and the hybrid patterns in between.",part:"Part II: Getting to Production"},{number:7,slug:"scaling-reliability-observability",shortTitle:"Scaling and Reliability",title:"Scaling, Reliability, and Observability",description:"Testing and evaluation harnesses, telemetry, graceful degradation, and capacity planning for concurrent conversations.",part:"Part II: Getting to Production"},{number:8,slug:"compliance-privacy-security",shortTitle:"Compliance and Privacy",title:"Compliance, Privacy, and Security",description:"Data residency, retention and redaction, consent and disclosure, and why audio that never leaves the device has no transit problem to mitigate.",part:"Part II: Getting to Production"},{number:9,slug:"reference-architectures",shortTitle:"Reference Architectures",title:"Reference Architectures",description:"Four architecture tiers from a single cloud agent to embedded edge deployments, with guidance on when each fits.",part:"Part III: Architecture and What's Next"},{number:10,slug:"future-of-voice-agents",shortTitle:"The Future of Voice Agents",title:"The Future: Speech-to-Speech and the Edge",description:"Speech-to-speech models vs the cascade, what survives the shift, and why model efficiency pushes inference toward the device.",part:"Part III: Architecture and What's Next"},{number:11,slug:"getting-started",shortTitle:"Getting Started",title:"Getting Started",description:"A five-question decision framework, a first-week build plan, and the fastest path from zero to a working agent.",part:"Part IV: Putting It Into Practice"},{number:12,slug:"common-failure-modes",shortTitle:"Common Failure Modes",title:"Common Failure Modes in Voice Agents",description:"The recurring production failures, from silent stalls and stale barge-in to cross-call audio leaks, and the design pattern that prevents each.",part:"Part IV: Putting It Into Practice"}],o=e=>i.find(t=>t.slug===e),r=e=>`${a}${e.slug}/`},81202:function(e,t,n){n.r(t);var a=n(96540),i=n(85902),o=n(28007),r=n(43048),s=n(15696),c=n(68083),l=n(31630),d=n(4052),u=n(98914),h=n(17437);const p=[{question:"What are AI voice agents?",answer:"AI voice agents are autonomous software systems that converse with users through spoken language in real time. They combine speech recognition, a reasoning engine (an LLM or a speech-to-intent model), and speech synthesis under an orchestration runtime that manages timing and interruptions. They answer phone lines, run inside apps and devices, and execute tasks through function calls."},{question:"How is a voice agent different from a chatbot or an IVR?",answer:(0,h.Y)(a.Fragment,null,"A traditional"," ",(0,h.Y)(d.A,{href:"/blog/smart-ivr-python-ai-call-center/"},"IVR routes callers")," ","through fixed menus with prerecorded prompts; it does not understand open-ended speech. A chatbot converses in text, where a delay of a few seconds is acceptable. A voice agent must understand free-form speech and respond under human conversational timing, with median turn gaps near 200 ms, while handling interruptions mid-sentence. That real-time constraint is what makes voice agents an engineering discipline of their own.")},{question:"Can AI voice agents run on-device?",answer:(0,h.Y)(a.Fragment,null,"Yes."," ",(0,h.Y)(d.A,{href:"/products/voice/wake-word/"},"Wake word detection"),","," ",(0,h.Y)(d.A,{href:"/products/voice/voice-activity-detection/"},"voice activity detection"),", ",(0,h.Y)(d.A,{href:"/products/voice/speech-to-text/"},"speech-to-text"),","," ",(0,h.Y)(d.A,{href:"/products/voice/speech-to-intent/"},"speech-to-intent"),", ",(0,h.Y)(d.A,{href:"/products/language/llm/"},"LLM inference"),", and"," ",(0,h.Y)(d.A,{href:"/products/voice/streaming-text-to-speech/"},"text-to-speech")," ","can all execute on-device, from mobile phones to embedded hardware. On-device execution removes network round-trips from the latency budget, keeps audio on the user's hardware, and works offline."," ",(0,h.Y)(d.A,{href:"/guide/voice-agents/on-device-voice-agents/"},"Chapter 6")," ","covers the full on-device stack and the hybrid patterns that reserve the cloud for heavier reasoning.")}];t.default=()=>(0,h.Y)(l.A,null,(0,h.Y)(i.Helmet,null,(0,h.Y)("meta",{charSet:"utf-8"}),(0,h.Y)("title",null,"AI Voice Agents: The Complete Engineering Guide"),(0,h.Y)("meta",{name:"description",content:"What AI voice agents are, how they work, and how to build them: a 13-part engineering guide covering architecture, latency, on-device deployment, and scale."}),(0,h.Y)("meta",{property:"og:title",content:"AI Voice Agents: The Complete Engineering Guide"}),(0,h.Y)("meta",{property:"og:description",content:"What AI voice agents are, how they work, and how to build them: a 13-part engineering guide covering architecture, latency, on-device deployment, and scale."}),(0,h.Y)("meta",{property:"og:url",content:"https://picovoice.ai/guide/voice-agents/"}),(0,h.Y)("meta",{property:"og:image",content:"https://picovoice.ai/og_images/og_picovoice_logo.png"}),(0,h.Y)("meta",{property:"og:type",content:"article"}),(0,h.Y)("meta",{name:"keywords",content:"ai voice agents, voice ai agents, voice agents, what is a voice agent, voice agent ar
1chitecture, build a voice agent, on-device voice ai"}),(0,h.Y)("meta",{name:"robots",content:"index, follow"}),(0,h.Y)("link",{rel:"canonical",href:"https://picovoice.ai/guide/voice-agents/"})),(0,h.Y)(r.A,{fluid:!0,className:"guide"},(0,h.Y)(r.A,{className:"guide-article"},(0,h.Y)("h1",null,"What Are AI Voice Agents?"),(0,h.Y)("p",null,"AI voice agents are software systems that hold spoken conversations with people in real time: they listen to speech, extract meaning, decide on a response or action, and speak back with synthesized audio, all while the user can interrupt. Unlike IVR menus or text chatbots, a voice agent operates under human conversational timing, where the median gap between speakers is roughly 200 ms across languages (",(0,h.Y)(d.A,{href:"https://www.pnas.org/doi/10.1073/pnas.0903616106"},"Stivers et al., 2009, PNAS"),")."),(0,h.Y)("p",null,"This guide is a complete engineering reference for building them: the architecture, the latency math, the build approaches, the deployment options from cloud to fully on-device, and the operational work that keeps an agent running in production."),(0,h.Y)("h2",null,"Why Voice Agents Now?"),(0,h.Y)("p",null,"Three shifts converged to make production voice agents buildable:"),(0,h.Y)("ul",null,(0,h.Y)("li",null,(0,h.Y)("strong",null,"The full stack streams.")," Speech-to-text engines emit partial transcripts while the user is mid-sentence, LLMs generate tokens incrementally, and text-to-speech engines synthesize audio from the first tokens. The stages of a conversation can overlap instead of queueing, which is what closes the gap to human turn-taking."," ",(0,h.Y)(d.A,{href:"/products/voice/streaming-text-to-speech/"},"Orca Streaming Text-to-Speech")," ","reaches 106 ms first-token-to-speech, 3.1x faster than ElevenLabs Streaming at 335 ms (",(0,h.Y)(d.A,{href:"/docs/benchmark/tts/"},"TTS benchmark"),")."),(0,h.Y)("li",null,(0,h.Y)("strong",null,"Reasoning became general.")," LLMs replaced hand-built dialog trees with models that handle open-ended requests, call functions, and ground answers in retrieved documents."),(0,h.Y)("li",null,(0,h.Y)("strong",null,"Inference moved onto the device.")," Apple ships Apple Intelligence with a roughly 3-billion-
1parameter foundation model running on-device, escalating to Private Cloud Compute only for heavier requests (",(0,h.Y)(d.A,{href:"https://machinelearning.apple.com/research/introducing-apple-foundation-models"},"Apple Machine Learning Research"),"). Google ships Gemini Nano for on-device inference on Android (",(0,h.Y)(d.A,{href:"https://developer.android.com/ai/gemini-nano"},"Android developer docs"),"). The platform vendors chose on-device-first with cloud fallback, and the same architecture is now available to any voice agent team.")),(0,h.Y)(c.A,{variant:"view"},(0,h.Y)("p",null,"The third shift is the one this guide treats as a first-class design axis. Where each component runs (cloud, on-device, or hybrid) sets a voice agent's latency floor, privacy posture, offline behavior, and cost curve at scale. Cloud pipelines bill by usage, so their costs grow unbounded with conversation volume; on-device inference is cost-effective at scale because it runs on hardware you already ship. Every architecture chapter below carries this decision through.")),(0,h.Y)("h2",null,"Who This Guide Is For?"),(0,h.Y)("ul",null,(0,h.Y)("li",null,(0,h.Y)("strong",null,"Production developers")," building a voice agent and deciding between managed APIs, orchestration frameworks, and component-level builds."),(0,h.Y)("li",null,(0,h.Y)("strong",null,"Architects")," defining latency budgets, telephony paths, reference architectures, and where inference runs."),(0,h.Y)("li",null,(0,h.Y)("strong",null,"Technical product leaders")," evaluating cost at scale, compliance exposure, and vendor lock-in before committing to a stack.")),(0,h.Y)("h2",null,"The Guide, Chapter by Chapter"),(()=>{const e=[];for(const t of u.mv){let n=e.find(e=>e.title===t.part);n||(n={title:t.part,chapters:[]},e.push(n)),n.chapters.push(t)}return e})().map(e=>(0,h.Y)("div",{key:e.title},(0,h.Y)("div",{className:"guide-part-label"},e.title),e.chapters.map(e=>(0,h.Y)(o.N_,{key:e.slug,to:(0,u.j3)(e),className:"guide-chapter-list-item"},(0,h.Y)("div",{className:"guide-chapter-list-item__number"},String(e.number).padStart(2,"0")),(0,h.Y)("div",null,(0,h.Y)("div",{className:"guide-chapter-list-item__title"},e.title),(0,h.Y)("div",{className:"guide-chapter-list-item__description"},e.description)))))),(0,h.Y)("p",null,"For definitions of the vocabulary used across these chapters, endpointing, barge-in, cascade, speech-to-speech, VAD, wake word, and the rest, see the"," ",(0,h.Y)(d.A,{href:"/docs/glossary/"},"Picovoice voice AI glossary"),"."),(0,h.Y)(c.A,{variant:"example"},(0,h.Y)("p",null,"The guide's worked examples use on-device engines you can run today:"," ",(0,h.Y)(d.A,{href:"/products/voice/wake-word/"},"Porcupine Wake Word"),","," ",(0,h.Y)(d.A,{href:"/products/voice/voice-activity-detection/"},"Cobra Voice Activity Detection"),","," ",(0,h.Y)(d.A,{href:"/products/voice/streaming-speech-to-text/"},"Cheetah Streaming Speech-to-Text"),","," ",(0,h.Y)(d.A,{href:"/products/voice/speech-to-intent/"},"Rhino Speech-to-Intent"),", ",(0,h.Y)(d.A,{href:"/products/language/llm/"},"picoLLM"),", and"," ",(0,h.Y)(d.A,{href:"/products/voice/streaming-text-to-speech/"},"Orca Streaming Text-to-Speech"),". Each ships with a published, open-source benchmark (",(0,h.Y)(d.A,{href:"/docs/benchmark/stt/"},"STT"),","," ",(0,h.Y)(d.A,{href:"/docs/benchmark/tts/"},"TTS"),","," ",(0,h.Y)(d.A,{href:"/docs/benchmark/nlu/"},"NLU"),","," ",(0,h.Y)(d.A,{href:"/docs/benchmark/vad/"},"VAD"),","," ",(0,h.Y)(d.A,{href:"/docs/benchmark/wake-word/"},"wake word"),"), so every claim in this guide about a Picovoice engine traces to a reproducible measurement. To see a complete agent assembled, the cookbook walks through an"," ",(0,h.Y)(d.A,{href:"/cookbook/llm-voice-ai-agent-assistant/"},"LLM-powered voice agent")," ","and a fully"," ",(0,h.Y)(d.A,{href:"/cookbook/embedded-ai-voice-assistant/"},"embedded voice assistant"),".")),(0,h.Y)("h2",null,"Where to Start"),(0,h.Y)("p",null,"Read"," ",(0,h.Y)(d.A,{href:"/guide/voice-agents/how-voice-agents-work/"},"Chapter 1")," ","if you are new to voice agent architecture. Jump to"," ",(0,h.Y)(d.A,{href:"/guide/voice-agents/build-approaches/"}
1,"Chapter 2")," ","if you are choosing a stack this quarter. Go straight to"," ",(0,h.Y)(d.A,{href:"/guide/voice-agents/getting-started/"},"Chapter 11")," ","if you want a working agent this week."),(0,h.Y)("h2",null,"Frequently Asked Questions"),(0,h.Y)(s.A,{questions:p}))))}}]);

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.