1<!DOCTYPE html><html lang="en"><head><meta charSet="utf-8"/><meta name="viewport" content="width=device-width, initial-scale=1"/><link rel="preload" href="/_next/static/media/e4af272ccee01ff0-s.p.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="stylesheet" href="/_next/static/css/7cca8e2c5137bd71.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/css/ede4122d18cd4431.css" data-precedence="next"/><link rel="preload" as="script" fetchPriority="low" href="/_next/static/chunks/webpack-c3efc9d46f3603d1.js"/>
1<script src="/_next/static/chunks/67cfe1a8-e0dea788f0436fdd.js" async=""></script>
1<script src="/_next/static/chunks/221-8c332328a2aadaa8.js" async=""></script>
1<script src="/_next/static/chunks/main-app-432dbcf9b265c3b5.js" async=""></script>
1<script src="/_next/static/chunks/353-a741de40784eeec2.js" async=""></script>
1<script src="/_next/static/chunks/517-3b8472ca0849e221.js" async=""></script>
1<script src="/_next/static/chunks/684-1cdde04ae5d4eb94.js" async=""></script>
1<script src="/_next/static/chunks/710-ce4278fb30f391f9.js" async=""></script>
1<script src="/_next/static/chunks/103-8734b6095f9d6e3e.js" async=""></script>
1<script src="/_next/static/chunks/944-584c068e0832c1d8.js" async=""></script>
1<script src="/_next/static/chunks/250-cdb4fb0ed9df8907.js" async=""></script>
1<script src="/_next/static/chunks/app/blog/page-383dc845b0c3d49a.js" async=""></script>
1<script src="/_next/static/chunks/438d328a-7affb0267e06113a.js" async=""></script>
1<script src="/_next/static/chunks/08ffd5a1-43352236a6ddb7f3.js" async=""></script>
1<script src="/_next/static/chunks/933-3e6b272cc96ac52d.js" async=""></script>
1<script src="/_next/static/chunks/app/layout-4b4d9230d4442cea.js" async=""></script>
1<script src="/_next/static/chunks/79-daee750508cf8ea9.js" async=""></script>
1<script src="/_next/static/chunks/225-c3203937dd1c3a1e.js" async=""></script>
1<script src="/_next/static/chunks/539-48d82b370880b46d.js" async=""></script>
1<script src="/_next/static/chunks/273-08e358a7d11f3a35.js" async=""></script>
1<script src="/_next/static/chunks/357-329abbf4f575845d.js" async=""></script>
1<script src="/_next/static/chunks/965-0e953887da00953d.js" async=""></script>
1<script src="/_next/static/chunks/app/page-8151b2b539ed540e.js" async=""></script>
1<link rel="preload" href="https://plausible.io/js/script.file-downloads.outbound-links.js" as="script"/><title>Blog | CognitiveLab</title><meta name="description" content="Latest articles, announcements and insights from our team of AI researchers and engineers"/><meta property="og:title" content="Blog | CognitiveLab"/><meta property="og:description" content="Latest articles, announcements and insights from our team of AI researchers and engineers"/><meta property="og:image" content="https://cognitivelab.in/assets/cognitivelab-og.png"/><meta property="og:image:width" content="1200"/><meta property="og:image:height" content="630"/><meta property="og:image:alt" content="CognitiveLab Blog"/><meta name="twitter:card" content="summary_large_image"/><meta name="twitter:title" content="Blog | CognitiveLab"/><meta name="twitter:description" content="Latest articles, announcements and insights from our team of AI researchers and engineers"/><meta name="twitter:image" content="https://cognitivelab.in/assets/cognitivelab-og.png"/><link rel="icon" href="/favicon.ico" type="image/x-icon" sizes="32x32"/><meta name="next-size-adjust"/>
1<script src="/_next/static/chunks/polyfills-78c92fac7aa8fdd8.js" noModule=""></script>
1</head><body class="__className_f367f3 relative min-h-screen"><!--$!--><template data-dgst="BAILOUT_TO_CLIENT_SIDE_RENDERING"></template><!--/$-->
1<script>!function(){try{var d=document.documentElement,c=d.classList;c.remove('light','dark');var e=localStorage.getItem('theme');if('system'===e||(!e&&false)){var t='(prefers-color-scheme: dark)',m=window.matchMedia(t);if(m.media!==t||m.matches){d.style.colorScheme = 'dark';c.add('dark')}else{d.style.colorScheme = 'light';c.add('light')}}else if(e){c.add(e|| '')}else{c.add('dark')}if(e==='light'||e==='dark'||!e)d.style.colorScheme=e||'dark'}catch(e){}}()</script>
1<header><div class="fixed left-0 right-0 top-0 z-50 px-4 pt-4"><nav class="relative mx-auto w-[95vw] max-w-[95vw] rounded-2xl border border-stroke-1 transition-all duration-300 ease-in-out bg-background/40" style="transform:translateY(-100px)"><div class="mx-auto px-4 py-2"><div class="flex h-16 items-center justify-between"><a class="flex items-center gap-x-2" href="/"><img alt="CognitiveLab" loading="lazy" width="50" height="50" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Ficons%2Fcognitivelab-logo.png&w=64&q=75 1x, /_next/image?url=%2Ficons%2Fcognitivelab-logo.png&w=128&q=75 2x" src="/_next/image?url=%2Ficons%2Fcognitivelab-logo.png&w=128&q=75"/><span class="text-xl font-semibold text-n-1 lg:text-2xl">CognitiveLab</span></a><div class="hidden lg:flex lg:flex-1 lg:items-center lg:justify-end"><nav aria-label="Main" data-orientation="horizontal" dir="ltr" class="relative"><div style="position:relative"><ul data-orientation="horizontal" class="flex items-center space-x-6" dir="ltr"><li><a href="/about" class="text-lg text-n-1 hover:text-color-1" data-radix-collection-item="">About</a></li><li><button id="radix-:R38qba:-trigger-radix-:R1b8qba:" data-state="closed" aria-expanded="false" aria-controls="radix-:R38qba:-content-radix-:R1b8qba:" class="group flex items-center text-lg text-n-1 hover:text-color-1" data-radix-collection-item="">Projects<!-- --> <svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-down ml-1 h-4 w-4 transition-transform group-data-[state=open]:-rotate-180"><path d="m6 9 6 6 6-6"></path></svg></button></li><li><a href="/publications" class="text-lg text-n-1 hover:text-color-1" data-radix-collection-item="">Research</a></li><li><a href="/blog" class="text-lg text-n-1 hover:text-color-1" data-radix-collection-item="">Blog</a></li></ul></div></nav><a class="ml-6" target="_blank" href="/contact"><button class="group relative inline-flex h-11 animate-rainbow cursor-pointer items-center justify-center rounded-lg border-0 border-t border-t-white bg-[length:200%] px-8 py-2 font-semibold text-white transition-colors [background-clip:padding-box,border-box,border-box] [background-origin:border-box] [border:calc(0.08*1rem)_solid_transparent] focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-ring disabled:pointer-events-none disabled:opacity-50 before:absolute before:bottom-[-20%] before:left-1/2 before:z-0 before:h-1/5 before:w-3/5 before:-translate-x-1/2 before:animate-rainbow before:bg-[linear-gradient(90deg,hsl(var(--color-1)),hsl(var(--color-5)),hsl(var(--color-3)),hsl(var(--color-4)),hsl(var(--color-2)))] before:bg-[length:200%] before:[filter:blur(calc(0.8*1rem))] bg-[linear-gradient(#121213,#121213),linear-gradient(#121213_50%,rgba(18,18,19,0.6)_80%,rgba(18,18,19,0)),linear-gradient(90deg,hsl(var(--color-1)),hsl(var(--color-5)),hsl(var(--color-3)),hsl(var(--color-4)),hsl(var(--color-2)))]">Contact Us</button></a></div><div class="lg:hidden"><button class="inline-flex items-center justify-center whitespace-nowrap rounded-md text-sm font-medium transition-colors focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-ring disabled:pointer-events-none disabled:opacity-50 border border-input bg-background shadow-sm hover:bg-accent hover:text-accent-foreground h-9 w-9" type="button" aria-haspopup="dialog" aria-expanded="false" aria-controls="radix-:R1oqba:" data-state="closed"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-menu h-6 w-6"><line x1="4" x2="20" y1="12" y2="12"></line><line x1="4" x2="20" y1="6" y2="6"></line><line x1="4" x2="20" y1="18" y2="18"></line></svg></button></div></div></div></nav></div></header><main><main class="overflow-hidden"><div id="hero" class="relative true -mt-[5.25rem] pt-[12rem]"><div style="height:50vh" class="jsx-3256262076492e18 relative"><div class="absolute inset-0 z-0 px-5 lg:px-7.5 xl:px-10" style="opacity:0"></div><div class="jsx-3256262076492e18 relative z-1 mx-auto flex h-full w-full items-center justify-center px-4 sm:px-6 lg:px-8"><div class="rounded-md border-2 border-gray-800 bg-transparent p-8 backdrop-blur-sm hover:border-gray-600" style="opacity:0;transform:translateY(20px)">
1<div class="jsx-3256262076492e18 z-10 mb-4 flex min-h-[4rem] cursor-pointer items-center justify-center"><div class="group relative mx-auto flex max-w-fit flex-row items-center justify-center rounded-2xl bg-black/40 px-4 py-1.5 text-sm font-medium shadow-[inset_0_-8px_10px_#8fdfff1f] backdrop-blur-sm transition-shadow duration-500 ease-out [--bg-size:300%] hover:shadow-[inset_0_-5px_10px_#8fdfff3f] dark:bg-black/40"><div class="absolute inset-0 block h-full w-full animate-gradient bg-gradient-to-r from-[#ffaa40]/50 via-[#9c40ff]/50 to-[#ffaa40]/50 bg-[length:var(--bg-size)_100%] p-[1px] [border-radius:inherit] ![mask-composite:subtract] [mask:linear-gradient(#fff_0_0)_content-box,linear-gradient(#fff_0_0)]"></div>âï¸<!-- --> <hr class="jsx-3256262076492e18 mx-2 h-4 w-[1px] shrink-0 bg-gray-300"/> <span class="jsx-3256262076492e18 metallic-text inline animate-gradient bg-gradient-to-r from-[#ffaa40] via-[#9c40ff] to-[#ffaa40] bg-[length:var(--bg-size)_100%] bg-clip-text text-transparent">Latest Articles</span><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-right ml-1 size-3 transition-transform duration-300 ease-in-out group-hover:translate-x-0.5"><path d="m9 18 6-6-6-6"></path></svg></div></div><h1 class="h1 mb-6 text-center" style="opacity:0;transform:translateY(20px)">Blog & <span class="animated-gradient-text metallic-text">Announcements</span></h1><p class="mx-auto mb-8 max-w-3xl text-center text-lg text-n-3 lg:text-xl" style="opacity:0;transform:translateY(20px)">Insights, tutorials, and updates from our team of AI researchers and engineers</p></div></div></div><div class="pointer-events-none absolute left-10 right-10 top-[1.25rem] hidden h-0.25 bg-n-6 xl:block"></div><svg class="pointer-events-none absolute left-[2.1875rem] top-[0.9375rem] z-2 hidden xl:block || """ width="11" height="11" fill="none"><path d="M7 1a1 1 0 0 0-1-1H5a1 1 0 0 0-1 1v2a1 1 0 0 1-1 1H1a1 1 0 0 0-1 1v1a1 1 0 0 0 1 1h2a1 1 0 0 1 1 1v2a1 1 0 0 0 1 1h1a1 1 0 0 0 1-1V8a1 1 0 0 1 1-1h2a1 1 0 0 0 1-1V5a1 1 0 0 0-1-1H8a1 1 0 0 1-1-1V1z" fill="#ada8c4"></path></svg><svg class="pointer-events-none absolute right-[2.1875rem] top-[0.9375rem] z-2 hidden xl:block || """ width="11" height="11" fill="none"><path d="M7 1a1 1 0 0 0-1-1H5a1 1 0 0 0-1 1v2a1 1 0 0 1-1 1H1a1 1 0 0 0-1 1v1a1 1 0 0 0 1 1h2a1 1 0 0 1 1 1v2a1 1 0 0 0 1 1h1a1 1 0 0 0 1-1V8a1 1 0 0 1 1-1h2a1 1 0 0 0 1-1V5a1 1 0 0 0-1-1H8a1 1 0 0 1-1-1V1z" fill="#ada8c4"></path></svg><div class="pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10"></div><div class="pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10"></div></div><div class="relative py-8 lg:py-10 xl:py-12 lg:py-24 xl:py-32 pb-16 pt-12"><div class="container"><div class="mb-8 flex items-end justify-between"><h2 class="text-4xl font-bold tracking-tight lg:text-5xl"><span class="animate-gradient-text">Featured</span> </h2><a class="group flex items-center text-n-3 transition-colors hover:text-n-1" href="/blog">View all<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-right ml-1 h-5 w-5 transition-transform group-hover:translate-x-1"><path d="m9 18 6-6-6-6"></path></svg></a></div><div class="mb-6"><div class="group relative flex h-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.01]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm lg:flex-row"><div class="relative h-64 p-3 sm:h-72 md:h-80 lg:h-auto lg:w-1/2 lg:p-4"><div class="relative h-full w-full overflow-hidden rounded-xl"><img alt="Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval" loading="lazy" decoding="async" data-nimg="fill" class="h-full object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6 z-10"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-transparent bg-destructive text-destructive-foreground hover:bg-destructive/80 font-medium">Announcement</div></div></div><div class="flex flex-1 flex-col p-6 sm:p-7 lg:p-8"><div class="mb-3 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2025-12-08T10:00:00.000Z">December 8, 2025</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-xl font-bold text-n-1 transition-colors sm:text-2xl lg:line-clamp-3 lg:text-3xl"><a href="/blog/introducing-netraembed">Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval</a></h3><p class="mb-4 line-clamp-2 flex-grow text-sm text-n-3 sm:line-clamp-3 sm:text-base lg:line-clamp-4">We're excited to announce NetraEmbed and ColNetraEmbed, achieving 152% improvement over existing systems in multilingual document retrieval. Supporting 22 languages with state-of-the-art performance.</p><div class="mt-auto flex flex-col gap-3 sm:flex-row sm:flex-wrap sm:items-end sm:justify-between"><div class="flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Multilingual AI</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Document Retrieval</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Vision-Language Models</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-sm font-medium transition-all" href="/blog/introducing-netraembed">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1.5 h-4 w-4 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div></div><div class="grid grid-cols-1 gap-6 sm:grid-cols-2 md:grid-cols-2 lg:grid-cols-3"><div class="w-full"><div class="group relative flex h-full w-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.02]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm"><div class="relative p-3"><div class="relative h-48 w-full overflow-hidden rounded-lg md:h-52"><img alt="CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana" loading="lazy" decoding="async" data-nimg="fill" class="object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-transparent bg-destructive text-destructive-foreground hover:bg-destructive/80 font-medium">Announcement</div></div></div><div class="flex flex-1 flex-col p-5 md:p-6"><div class="mb-2 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2025-04-30T10:00:00.000Z">April 30, 2025</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-lg font-bold text-n-1 transition-colors md:text-xl"><a href="/blog/introducing-nayana">CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana</a></h3><p class="mb-4 line-clamp-3 flex-grow text-sm text-n-3">CognitiveLab is proud to be selected as a recipient of Meta's prestigious Llama Impact Grant 2024, accelerating our revolutionary Nayana projectâa multilingual (22 languages, 10 Indic), multimodal AI ecosystem that democratizes AI across global languages.</p><div class="mt-auto flex flex-wrap items-end justify-between"><div class="mb-2 flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Nayana</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-xs font-medium transition-all" href="/blog/introducing-nayana">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1 h-3.5 w-3.5 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div></div><div class="w-full"><div class="group relative flex h-full w-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.02]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm"><div class="relative p-3"><div class="relative h-48 w-full overflow-hidden rounded-lg md:h-52"><img alt="Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers" loading="lazy" decoding="async" data-nimg="fill" class="object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6"><div class="inline-flex items-center rounded-full px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-2 bg-black text-white hover:bg-primary-400 font-medium">Blog</div></div></div><div class="flex flex-1 flex-col p-5 md:p-6"><div class="mb-2 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2024-12-11T12:00:00.000Z">December 11, 2024</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-lg font-bold text-n-1 transition-colors md:text-xl"><a href="/blog/introducing-ai-engineering-academy">Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers</a></h3><p class="mb-4 line-clamp-3 flex-grow text-sm text-n-3">AI Engineering Academy offers comprehensive training programs and resources to help aspiring engineers master the technical skills and practical knowledge needed to build, deploy, and maintain AI systems at scale.</p><div class="mt-auto flex flex-wrap items-end justify-between"><div class="mb-2 flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Ai</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Engineering</div></di
1v><a class="text-primary-500 hover:text-gradient-500 flex items-center text-xs font-medium transition-all" href="/blog/introducing-ai-engineering-academy">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1 h-3.5 w-3.5 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div></div><div class="w-full"><div class="group relative flex h-full w-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.02]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm"><div class="relative p-3"><div class="relative h-48 w-full overflow-hidden rounded-lg md:h-52"><img alt="Introducing Omniparse: Universal Data Parsing" loading="lazy" decoding="async" data-nimg="fill" class="object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6"><div class="inline-flex items-center rounded-full px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-2 bg-black text-white hover:bg-primary-400 font-medium">Blog</div></div></div><div class="flex flex-1 flex-col p-5 md:p-6"><div class="mb-2 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2024-06-27T12:00:00.000Z">June 27, 2024</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-lg font-bold text-n-1 transition-colors md:text-xl"><a href="/blog/introducing-omniparse">Introducing Omniparse: Universal Data Parsing</a></h3><p class="mb-4 line-clamp-3 flex-grow text-sm text-n-3">Omniparse is an advanced document parsing platform that uses AI to extract structured data from any document format, enabling businesses to automate document processing workflows with unprecedented accuracy and efficiency.</p><div class="mt-auto flex flex-wrap items-end justify-between"><div class="mb-2 flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Omniparse</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-xs font-medium transition-all" href="/blog/introducing-omniparse">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1 h-3.5 w-3.5 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div></div></div></div><div class="pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10"></div><div class="pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10"></div></div><div class="relative py-8 lg:py-10 xl:py-12 lg:py-24 xl:py-32 pb-20 pt-4"><div class="container"><div class="mb-8"><h2 class="text-4xl font-bold tracking-tight lg:text-5xl">All<!-- --> <span class="text-gradient-500">Content</span></h2></div><div dir="ltr" data-orientation="horizontal" class="w-full"><div class="mb-6 overflow-auto pb-2"><div role="tablist" aria-orientation="horizontal" class="h-9 items-center text-muted-foreground inline-flex w-auto justify-start gap-4 overflow-visible rounded-lg border border-n-6 bg-n-7/50 p-1 backdrop-blur-sm" tabindex="-1" data-orientation="horizontal" style="outline:none"><button type="button" role="tab" aria-selected="true" aria-controls="radix-:R9f7rpaba:-content-all" data-state="active" id="radix-:R9f7rpaba:-trigger-all" class="inline-flex items-center justify-center whitespace-nowrap text-sm font-medium ring-offset-background transition-all focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 disabled:pointer-events-none disabled:opacity-50 data-[state=active]:bg-background data-[state=active]:text-foreground data-[state=active]:shadow rounded-md px-5 py-2" tabindex="-1" data-orientation="horizontal" data-radix-collection-item="">All</button><button type="button" role="tab" aria-selected="false" aria-controls="radix-:R9f7rpaba:-content-blog" data-state="inactive" id="radix-:R9f7rpaba:-trigger-blog" class="inline-flex items-center justify-center whitespace-nowrap text-sm font-medium ring-offset-background transition-all focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 disabled:pointer-events-none disabled:opacity-50 data-[state=active]:bg-background data-[state=active]:text-foreground data-[state=active]:shadow rounded-md px-5 py-2" tabindex="-1" data-orientation="horizontal" data-radix-collection-item="">Blog Posts</button><button type="button" role="tab" aria-selected="false" aria-controls="radix-:R9f7rpaba:-content-announcements" data-state="inactive" id="radix-:R9f7rpaba:-trigger-announcements" class="inline-flex items-center justify-center whitespace-nowrap text-sm font-medium ring-offset-background transition-all focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 disabled:pointer-events-none disabled:opacity-50 data-[state=active]:bg-background data-[state=active]:text-foreground data-[state=active]:shadow rounded-md px-5 py-2" tabindex="-1" data-orientation="horizontal" data-radix-collection-item="">Announcements</button></div></div><div data-state="active" data-orientation="horizontal" role="tabpanel" aria-labelledby="radix-:R9f7rpaba:-trigger-all" id="radix-:R9f7rpaba:-content-all" tabindex="0" class="ring-offset-background focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 mt
1-0" style="animation-duration:0s"><div class="flex flex-col space-y-6"><div class="group relative flex h-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.01]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm lg:flex-row"><div class="relative h-64 p-3 sm:h-72 md:h-80 lg:h-auto lg:w-1/2 lg:p-4"><div class="relative h-full w-full overflow-hidden rounded-xl"><img alt="Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval" loading="lazy" decoding="async" data-nimg="fill" class="h-full object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-netraembed%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6 z-10"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-transparent bg-destructive text-destructive-foreground hover:bg-destructive/80 font-medium">Announcement</div></div></div><div class="flex flex-1 flex-col p-6 sm:p-7 lg:p-8"><div class="mb-3 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2025-12-08T10:00:00.000Z">December 8, 2025</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-xl font-bold text-n-1 transition-colors sm:text-2xl lg:line-clamp-3 lg:text-3xl"><a href="/blog/introducing-netraembed">Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval</a></h3><p class="mb-4 line-clamp-2 flex-grow text-sm text-n-3 sm:line-clamp-3 sm:text-base lg:line-clamp-4">We're excited to announce NetraEmbed and ColNetraEmbed, achieving 152% improvement over existing systems in multilingual document retrieval. Supporting 22 languages with state-of-the-art performance.</p><div class="mt-auto flex flex-col gap-3 sm:flex-row sm:flex-wrap sm:items-end sm:justify-between"><div class="flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Multilingual AI</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Document Retrieval</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Vision-Language Models</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-sm font-medium transition-all" href="/blog/introducing-netraembed">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1.5 h-4 w-4 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div><div class="group relative flex h-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.01]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm lg:flex-row"><div class="relative h-64 p-3 sm:h-72 md:h-80 lg:h-auto lg:w-1/2 lg:p-4"><div class="relative h-full w-full overflow-hidden rounded-xl"><img alt="CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana" loading="lazy" decoding="async" data-nimg="fill" class="h-full object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-nayana%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6 z-10"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-transparent bg-destructive text-destructive-foreground hover:bg-destructive/80 font-medium">Announcement</div></div></div><div class="flex flex-1 flex-col p-6 sm:p-7 lg:p-8"><div class="mb-3 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2025-04-30T10:00:00.000Z">April 30, 2025</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-xl font-bold text-n-1 transition-colors sm:text-2xl lg:line-clamp-3 lg:text-3xl"><a href="/blog/introducing-nayana">CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana</a></h3><p class="mb-4 line-clamp-2 flex-grow text-sm text-n-3 sm:line-clamp-3 sm:text-base lg:line-clamp-4">CognitiveLab is proud to be selected as a recipient of Meta's prestigious Llama Impact Grant 2024, accelerating our revolutionary Nayana projectâa multilingual (22 languages, 10 Indic), multimodal AI ecosystem that democratizes AI across global languages.</p><div class="mt-auto flex flex-col gap-3 sm:flex-row sm:flex-wrap sm:items-end sm:justify-between"><div class="flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Nayana</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-sm font-medium transition-all" href="/blog/introducing-nayana">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1.5 h-4 w-4 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div><div class="group relative flex h-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.01]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm lg:flex-row"><div class="relative h-64 p-3 sm:h-72 md:h-80 lg:h-auto lg:w-1/2 lg:p-4"><div class="relative h-full w-full overflow-hidden rounded-xl"><img alt="Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers" loading="lazy" decoding="async" data-nimg="fill" class="h-full object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-ai-engineering-academy%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6 z-10"><div class="inline-flex items-center rounded-full px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-2 bg-black text-white hover:bg-primary-400 font-medium">Blog</div></div></div><div class="flex flex-1 flex-col p-6 sm:p-7 lg:p-8"><div class="mb-3 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2024-12-11T12:00:00.000Z">December 11, 2024</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-xl font-bold text-n-1 transition-colors sm:text-2xl lg:line-clamp-3 lg:text-3xl"><a href="/blog/introducing-ai-engineering-academy">Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers</a></h3><p class="mb-4 line-clamp-2 flex-grow text-sm text-n-3 sm:line-clamp-3 sm:text-base lg:line-clamp-4">AI Engineering Academy offers comprehensive training programs and resources to help aspiring engineers master the technical skills and practical knowledge needed to build, deploy, and maintain AI systems at scale.</p><div class="mt-auto flex flex-col gap-3 sm:flex-row sm:flex-wrap sm:items-end sm:justify-between"><div class="flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Ai</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Engineering</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-sm font-medium transition-all" href="/blog/introducing-ai-engineering-academy">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1.5 h-4 w-4 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div><div class="group relative flex h-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.01]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm lg:flex-row"><div class="relative h-64 p-3 sm:h-72 md:h-80 lg:h-auto lg:w-1/2 lg:p-4"><div class="relative h-full w-full overflow-hidden rounded-xl"><img alt="Introducing Omniparse: Universal Data Parsing" loading="lazy" decoding="async" data-nimg="fill" class="h-full object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-omniparse%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6 z-10"><div class="inline-flex items-center rounded-full px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-2 bg-black text-white hover:bg-primary-400 font-medium">Blog</div></div></div><div class="flex flex-1 flex-col p-6 sm:p-7 lg:p-8"><div class="mb-3 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2024-06-27T12:00:00.000Z">June 27, 2024</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-xl font-bold text-n-1 transition-colors sm:text-2xl lg:line-clamp-3 lg:text-3xl"><a href="/blog/introducing-omniparse">Introducing Omniparse: Universal Data Parsing</a></h3><p class="mb-4 line-clamp-2 flex-grow text-sm text-n-3 sm:line-clamp-3 sm:text-base lg:line-clamp-4">Omniparse is an advanced document pa
1rsing platform that uses AI to extract structured data from any document format, enabling businesses to automate document processing workflows with unprecedented accuracy and efficiency.</p><div class="mt-auto flex flex-col gap-3 sm:flex-row sm:flex-wrap sm:items-end sm:justify-between"><div class="flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Omniparse</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-sm font-medium transition-all" href="/blog/introducing-omniparse">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1.5 h-4 w-4 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div><div class="group relative flex h-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.01]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm lg:flex-row"><div class="relative h-64 p-3 sm:h-72 md:h-80 lg:h-auto lg:w-1/2 lg:p-4"><div class="relative h-full w-full overflow-hidden rounded-xl"><img alt="Introducing Indic LLM Leaderboard: Benchmarking Indian Language Models" loading="lazy" decoding="async" data-nimg="fill" class="h-full object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-indic-llm-leaderboard%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6 z-10"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-transparent bg-destructive text-destructive-foreground hover:bg-destructive/80 font-medium">Announcement</div></div></div><div class="flex flex-1 flex-col p-6 sm:p-7 lg:p-8"><div class="mb-3 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2024-04-25T12:00:00.000Z">April 25, 2024</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-xl font-bold text-n-1 transition-colors sm:text-2xl lg:line-clamp-3 lg:text-3xl"><a href="/blog/introducing-indic-llm-leaderboard">Introducing Indic LLM Leaderboard: Benchmarking Indian Language Models</a></h3><p class="mb-4 line-clamp-2 flex-grow text-sm text-n-3 sm:line-clamp-3 sm:text-base lg:line-clamp-4">The Indic LLM Leaderboard is a comprehensive evaluation platform for language models across 22+ Indian languages, providing standardized metrics and culturally relevant evaluation tasks to accelerate progress in Indian language AI research.</p><div class="mt-auto flex flex-col gap-3 sm:flex-row sm:flex-wrap sm:items-end sm:justify-between"><div class="flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Indic</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Llm</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-sm font-medium transition-all" href="/blog/introducing-indic-llm-leaderboard">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1.5 h-4 w-4 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div><div class="group relative flex h-full flex-col overflow-hidden rounded-2xl border border-n-6 bg-conic-gradient p-0.25 transition-all duration-500 hover:scale-[1.01]"><div class="relative z-10 flex h-full flex-col rounded-[18px] bg-n-8 backdrop-blur-sm lg:flex-row"><div class="relative h-64 p-3 sm:h-72 md:h-80 lg:h-auto lg:w-1/2 lg:p-4"><div class="relative h-full w-full overflow-hidden rounded-xl">
1<img alt="Introducing Ambari: Biligual Kannada English Large Language Model" loading="lazy" decoding="async" data-nimg="fill" class="h-full object-cover transition-transform duration-500 group-hover:scale-105" style="position:absolute;height:100%;width:100%;left:0;top:0;right:0;bottom:0;color:transparent" sizes="100vw" srcSet="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=640&q=75 640w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=750&q=75 750w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=828&q=75 828w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=1080&q=75 1080w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=1200&q=75 1200w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=1920&q=75 1920w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=2048&q=75 2048w, /_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=3840&q=75 3840w" src="/_next/image?url=%2Fassets%2Fblog%2Fintroducing-ambari%2Fcover.png&w=3840&q=75"/></div><div class="absolute left-6 top-6 z-10"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 text-xs transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-transparent bg-destructive text-destructive-foreground hover:bg-destructive/80 font-medium">Announcement</div></div></div><div class="flex flex-1 flex-col p-6 sm:p-7 lg:p-8"><div class="mb-3 flex items-center text-xs text-n-3"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-calendar mr-1.5 h-3.5 w-3.5"><path d="M8 2v4"></path><path d="M16 2v4"></path><rect width="18" height="18" x="3" y="4" rx="2"></rect><path d="M3 10h18"></path></svg><time dateTime="2024-01-01T12:00:00.000Z">January 1, 2024</time></div><h3 class="group-hover:text-gradient-500 mb-3 line-clamp-2 flex-none text-xl font-bold text-n-1 transition-colors sm:text-2xl lg:line-clamp-3 lg:text-3xl"><a href="/blog/introducing-ambari">Introducing Ambari: Biligual Kannada English Large Language Model</a></h3><p class="mb-4 line-clamp-2 flex-grow text-sm text-n-3 sm:line-clamp-3 sm:text-base lg:line-clamp-4">Ambari is a comprehensive natural language processing platform specifically designed for Indian languages, enabling developers to build sophisticated language applications with unprecedented accuracy and cultural context across 22+ Indian languages.</p><div class="mt-auto flex flex-col gap-3 sm:flex-row sm:flex-wrap sm:items-end sm:justify-between"><div class="flex flex-wrap gap-2"><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Introducing</div><div class="inline-flex items-center rounded-full border px-2.5 py-0.5 transition-colors focus:outline-none focus:ring-2 focus:ring-ring focus:ring-offset-2 border-n-5 bg-n-7/50 text-n-1 hover:border-primary-500 hover:text-primary-500 text-xs font-medium">Ambari</div></div><a class="text-primary-500 hover:text-gradient-500 flex items-center text-sm font-medium transition-all" href="/blog/introducing-ambari">Read more<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-arrow-right ml-1.5 h-4 w-4 transition-transform group-hover:translate-x-1"><path d="M5 12h14"></path><path d="m12 5 7 7-7 7"></path></svg></a></div></div></div></div></div></div><div data-state="inactive" data-orientation="horizontal" role="tabpanel" aria-labelledby="radix-:R9f7rpaba:-trigger-blog" hidden="" id="radix-:R9f7rpaba:-content-blog" tabindex="0" class="ring-offset-background focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 mt
1-0"></div><div data-state="inactive" data-orientation="horizontal" role="tabpanel" aria-labelledby="radix-:R9f7rpaba:-trigger-announcements" hidden="" id="radix-:R9f7rpaba:-content-announcements" tabindex="0" class="ring-offset-background focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 mt-0"></div></div></div><div class="pointer-events-none absolute left-10 right-10 top-[1.25rem] hidden h-0.25 bg-n-6 xl:block"></div><svg class="pointer-events-none absolute left-[2.1875rem] top-[0.9375rem] z-2 hidden xl:block || """ width="11" height="11" fill="none"><path d="M7 1a1 1 0 0 0-1-1H5a1 1 0 0 0-1 1v2a1 1 0 0 1-1 1H1a1 1 0 0 0-1 1v1a1 1 0 0 0 1 1h2a1 1 0 0 1 1 1v2a1 1 0 0 0 1 1h1a1 1 0 0 0 1-1V8a1 1 0 0 1 1-1h2a1 1 0 0 0 1-1V5a1 1 0 0 0-1-1H8a1 1 0 0 1-1-1V1z" fill="#ada8c4"></path></svg><svg class="pointer-events-none absolute right-[2.1875rem] top-[0.9375rem] z-2 hidden xl:block || """ width="11" height="11" fill="none"><path d="M7 1a1 1 0 0 0-1-1H5a1 1 0 0 0-1 1v2a1 1 0 0 1-1 1H1a1 1 0 0 0-1 1v1a1 1 0 0 0 1 1h2a1 1 0 0 1 1 1v2a1 1 0 0 0 1 1h1a1 1 0 0 0 1-1V8a1 1 0 0 1 1-1h2a1 1 0 0 0 1-1V5a1 1 0 0 0-1-1H8a1 1 0 0 1-1-1V1z" fill="#ada8c4"></path></svg><div class="pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10"></div><div class="pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10"></div></div></main></main><div class="relative py-8 lg:py-10 xl:py-12 lg:py-24 xl:py-32 !px-0 !py-10"><div class="container pt-10"><div class="flex flex-col items-center gap-10 md:items-start md:justify-between lg:flex-row"><div class="flex items-center lg:items-start"><img alt="CognitiveLab Logo" loading="lazy" width="80" height="80" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Ficons%2Fcognitivelab-logo.png&w=96&q=75 1x, /_next/image?url=%2Ficons%2Fcognitivelab-logo.png&w=256&q=75 2x" src="/_next/image?url=%2Ficons%2Fcognitivelab-logo.png&w=256&q=75"/><div class="ml-4 flex flex-col"><h3 class="text-xl font-semibold md:text-2xl">CognitiveLab</h3><p class="mt-2 max-w-[250px] text-sm text-n-4">Transforming Enterprises with AI Solutions at Scale</p></div></div><div class="grid grid-cols-2 gap-8 md:grid-cols-3 lg:gap-20"><div class="flex flex-col gap-4"><h3 class="text-lg font-semibold text-n-1">Products</h3><a class="text-n-4 transition-colors hover:text-n-1" href="/omniparse">OmniParse</a><a class="text-n-4 transition-colors hover:text-n-1" href="/aiengineering">AI Engineering Academy</a><a class="text-n-4 transition-colors hover:text-n-1" href="/nayana">Nayana</a><a class="text-n-4 transition-colors hover:text-n-1" href="/storyblocks">StoryBlocks</a><a class="text-n-4 transition-colors hover:text-n-1" target="_blank" rel="noopener noreferrer" href="https://github.com/adithya-s-k/RAG-SaaS">RAG SaaS</a></div><div class="flex flex-col gap-4"><h3 class="text-lg font-semibold text-n-1">Resources</h3><a class="text-n-4 transition-colors hover:text-n-1" href="/blog">Blog</a><a class="text-n-4 transition-colors hover:text-n-1" target="_blank" rel="noopener noreferrer" href="https://docs.cognitivelab.in">Documentation</a><a class="text-n-4 transition-colors hover:text-n-1" href="/publications">Research</a></div><div class="flex flex-col gap-4"><h3 class="text-lg font-semibold text-n-1">Company</h3><a class="text-n-4 transition-colors hover:text-n-1" href="/about">About</a><a class="text-n-4 transition-colors hover:text-n-1" href="/contact">Contact</a><a class="text-n-4 transition-colors hover:text-n-1" href="/privacy-policy">Privacy Policy</a><a class="text-n-4 transition-colors hover:text-n-1" href="/terms-of-service">Terms of Service</a></div></div></div><div class="mt-10 flex flex-col items-center justify-between gap-6 md:flex-row"><p class="text-sm text-n-4">© <!-- -->2025<!-- --> CognitiveLab. All rights reserved.</p><div class="flex gap-4"><a href="https://www.linkedin.com/company/cognitivelabai/" target="_blank" rel="noopener noreferrer" class="text-n-4 transition-colors hover:text-n-1"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-linkedin"><path d="M16 8a6 6 0 0 1 6 6v7h-4v-7a2 2 0 0 0-2-2 2 2 0 0 0-2 2v7h-4v-7a6 6 0 0 1 6-6z"></path><rect width="4" height="12" x="2" y="9"></rect><circle cx="4" cy="4" r="2"></circle></svg></a><a href="https://github.com/adithya-s-k" target="_blank" rel="noopener noreferrer" class="text-n-4 transition-colors hover:text-n-1"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-github"><path d="M15 22v-4a4.8 4.8 0 0 0-1-3.5c3 0 6-2 6-5.5.08-1.25-.27-2.48-1-3.5.28-1.15.28-2.35 0-3.5 0 0-1 0-3 1.5-2.64-.5-5.36-.5-8 0C6 2 5 2 5 2c-.3 1.15-.3 2.35 0 3.5A5.403 5.403 0 0 0 4 9c0 3.5 3 5.5 6 5.5-.39.49-.68 1.05-.85 1.65-.17.6-.22 1.23-.15 1.85v4"></path><path d="M9 18c-4.51 2-5-2-7-2"></path></svg></a><a href="https://x.com/cognitivelab_ai" target="_blank" rel="noopener noreferrer" class="text-n-4 transition-colors hover:text-n-1"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-twitter"><path d="M22 4s-.7 2.1-2 3.4c1.6 10-9.4 17.3-18 11.6 2.2.1 4.4-.6 6-2C3 15.5.5 9.6 3 5c2.2 2.6 5.6 4.1 9 4-.9-4.2 4-6.6 7-3.8 1.1 0 3-1.2 3-1.2z"></path></svg></a></div></div></div><div class="pointer-events-none absolute left-10 right-10 top-[1.25rem] hidden h-0.25 bg-n-6 xl:block"></div><svg class="pointer-events-none absolute left-[2.1875rem] top-[0.9375rem] z-2 hidden xl:block || """ width="11" height="11" fill="none"><path d="M7 1a1 1 0 0 0-1-1H5a1 1 0 0 0-1 1v2a1 1 0 0 1-1 1H1a1 1 0 0 0-1 1v1a1 1 0 0 0 1 1h2a1 1 0 0 1 1 1v2a1 1 0 0 0 1 1h1a1 1 0 0 0 1-1V8a1 1 0 0 1 1-1h2a1 1 0 0 0 1-1V5a1 1 0 0 0-1-1H8a1 1 0 0 1-1-1V1z" fill="#ada8c4"></path></svg><svg class="pointer-events-none absolute right-[2.1875rem] top-[0.9375rem] z-2 hidden xl:block || """ width="11" height="11" fill="none"><path d="M7 1a1 1 0 0 0-1-1H5a1 1 0 0 0-1 1v2a1 1 0 0 1-1 1H1a1 1 0 0 0-1 1v1a1 1 0 0 0 1 1h2a1 1 0 0 1 1 1v2a1 1 0 0 0 1 1h1a1 1 0 0 0 1-1V8a1 1 0 0 1 1-1h2a1 1 0 0 0 1-1V5a1 1 0 0 0-1-1H8a1 1 0 0 1-1-1V1z" fill="#ada8c4"></path></svg><div class="pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10"></div><div class="pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10"></div></div>
1<script src="/_next/static/chunks/webpack-c3efc9d46f3603d1.js" async=""></script>
1<script>(self.__next_f=self.__next_f||[]).push([0]);self.__next_f.push([2,null])</script>
1<script>self.__next_f.push([1,"1:HL[\"/_next/static/media/e4af272ccee01ff0-s.p.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n2:HL[\"/_next/static/css/7cca8e2c5137bd71.css\",\"style\"]\n3:HL[\"/_next/static/css/ede4122d18cd4431.css\",\"style\"]\n"])</script>
1<script>self.__next_f.push([1,"4:I[43787,[],\"\"]\n6:I[14202,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"default\"]\n7:I[71102,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"\"]\n8:I[48691,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"default\"]\nd:I[68309,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"Tabs\"]\ne:I[68309,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"TabsList\"]\nf:I[68309,[\"353\",\"static/chunks/353"])</script>
1<script>self.__next_f.push([1,"-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"TabsTrigger\"]\n10:I[68309,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"TabsContent\"]\n2c:I[40521,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"944\",\"static/chunks/944-584c068e0832c1d8.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"404\",\"static/chunks/app/blog/page-383dc845b0c3d49a.js\"],\"BottomLine\"]\n2d:I[83099,[],\"\"]\n2e:I[42506,[],\"\"]\n2f:I[54047,[\"616\",\"static/chunks/438d328a-7affb0267e06113a.js\",\"791\",\"static/chunks/08ffd5a1-43352236a6ddb7f3.js\",\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"933\",\"static/chunks/933-3e6b272cc96ac52d.js\",\"185\",\"static/chunks/app/layout-4b4d9230d4442cea.js\"],\"PostHogProvider\"]\n30:I[69066,[\"616\",\"static/chunks/438d328a-7affb0267e06113a.js\",\"791\",\"static/chunks/08ffd5a1-43352236a6ddb7f3.js\",\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"933\",\"static/chunks/933-3e6b272cc96ac52d.js\",\"185\",\"static/chunks/app/layout-4b4d9230d4442cea.js\"],\"ThemeProvider\"]"])</script>
1<script>self.__next_f.push([1,"\n31:I[94412,[\"616\",\"static/chunks/438d328a-7affb0267e06113a.js\",\"791\",\"static/chunks/08ffd5a1-43352236a6ddb7f3.js\",\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"933\",\"static/chunks/933-3e6b272cc96ac52d.js\",\"185\",\"static/chunks/app/layout-4b4d9230d4442cea.js\"],\"default\"]\n32:I[93300,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chunks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"static/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"default\"]\n33:I[47357,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chunks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"static/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"default\"]\n34:I[99257,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chunks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"sta"])</script>
1<script>self.__next_f.push([1,"tic/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"default\"]\n35:I[23602,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chunks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"static/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"default\"]\n36:I[13366,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chunks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"static/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"default\"]\n37:I[40773,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chunks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"static/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"default\"]\n38:I[53989,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chu"])</script>
1<script>self.__next_f.push([1,"nks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"static/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"Image\"]\n3a:I[78539,[\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"710\",\"static/chunks/710-ce4278fb30f391f9.js\",\"79\",\"static/chunks/79-daee750508cf8ea9.js\",\"225\",\"static/chunks/225-c3203937dd1c3a1e.js\",\"250\",\"static/chunks/250-cdb4fb0ed9df8907.js\",\"539\",\"static/chunks/539-48d82b370880b46d.js\",\"273\",\"static/chunks/273-08e358a7d11f3a35.js\",\"357\",\"static/chunks/357-329abbf4f575845d.js\",\"965\",\"static/chunks/965-0e953887da00953d.js\",\"931\",\"static/chunks/app/page-8151b2b539ed540e.js\"],\"default\"]\n3b:I[11864,[\"616\",\"static/chunks/438d328a-7affb0267e06113a.js\",\"791\",\"static/chunks/08ffd5a1-43352236a6ddb7f3.js\",\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"933\",\"static/chunks/933-3e6b272cc96ac52d.js\",\"185\",\"static/chunks/app/layout-4b4d9230d4442cea.js\"],\"default\"]\n3c:I[43722,[\"616\",\"static/chunks/438d328a-7affb0267e06113a.js\",\"791\",\"static/chunks/08ffd5a1-43352236a6ddb7f3.js\",\"353\",\"static/chunks/353-a741de40784eeec2.js\",\"517\",\"static/chunks/517-3b8472ca0849e221.js\",\"684\",\"static/chunks/684-1cdde04ae5d4eb94.js\",\"103\",\"static/chunks/103-8734b6095f9d6e3e.js\",\"933\",\"static/chunks/933-3e6b272cc96ac52d.js\",\"185\",\"static/chunks/app/layout-4b4d9230d4442cea.js\"],\"\"]\n3e:I[64384,[],\"\"]\n9:T60b7,"])</script>
1<script>self.__next_f.push([1,"\n## The Multilingual Document Challenge\n\nA pharmaceutical manufacturer's quality team [spends 30% of their time](https://www.mastercontrol.com/uk/gxp-lifeline/electronic-document-management-for-multilingual-compliance/) just maintaining consistency between language versions of the same document across facilities in Poland, Portugal, and Germany. A construction project manager in Singapore needs critical safety specifications from a Korean supplier's documentation, but language barriers delay risk identification. A customer support agent in Brazil searches their knowledge base for a solution-only to find the answer exists in English documentation they can't effectively access.\n\nThese aren't edge cases. They represent the daily reality facing [global enterprises, researchers, and organizations](https://www.moveworks.com/us/en/resources/blog/ai-powered-multilingual-it-support) operating across linguistic boundaries. While AI has made tremendous strides in understanding English documents, [76% of online shoppers prefer information in their native language](https://www.helpscout.com/blog/multilingual-ai-support/), and [40% will never buy from websites in other languages](https://www.helpscout.com/blog/multilingual-ai-support/). For businesses with global operations, multilingual documentation isn't a nice-to-have-it's mission-critical.\n\nThe technical reality is even more challenging. When master documents require updates, [synchronizing changes across multiple language versions creates a documentation management nightmare](https://www.mastercontrol.com/uk/gxp-lifeline/electronic-document-management-for-multilingual-compliance/). Documentation delays from multilingual review cycles slow batch releases and impact market availability. For [construction projects overseas](https://ascelibrary.org/doi/10.1061/JCEMD4.COENG-14273), the inability to quickly search and retrieve information from foreign-language documents increases project risk and costs.\n\nThe numbers tell a stark story: existing state-of-the-art document retrieval systems score just 0.284 on cross-lingual retrieval tasks-performance so poor it's essentially unusable in production. In our testing, these systems consistently fail to find relevant documents when queries and documents are in different languages, making them unreliable for real-world multilingual workflows. When your business operates in multiple languages, this gap translates directly into lost productivity, missed opportunities, and frustrated teams.\n\n## Introducing NetraEmbed: Breaking the Language Barrier\n\nToday, we're excited to announce **[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed)** and **[ColNetraEmbed](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)**, two models that fundamentally change what's possible in multilingual and multimodal document retrieval. These aren't incremental improvements-they represent a 152% leap forward in performance, bringing cross-lingual document search from barely functional to genuinely useful.\n\nWhat makes this particularly powerful is the **multimodal** approach. Traditional document search systems rely on extracting text through OCR, which loses critical information like charts, diagrams, tables, and document layout. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) processes documents as images, preserving all visual elements and understanding how information is structured on the page. This means you can search for \"revenue growth chart\" and find the actual chart, or search for \"organizational hierarchy\" and locate the relevant diagram-across any of the 22 supported languages.\n\n## The Numbers That Matter\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) achieves a score of **0.716 NDCG@5** on cross-lingual retrieval, compared to the previous best of 0.284. In practical terms, this means the system can now reliably find the right document even when your query and documents are in different languages. For monolingual searches, the performance is even better at **0.738 NDCG@5**, an 80% improvement over existing solutions.\n\nHere's what makes this particularly compelling for businesses: we didn't sacrifice English performance to achieve multilingual capability. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) scores **0.554** on standard English benchmarks, remaining competitive with models designed exclusively for English. You get global capability without compromising on your primary language.\n\nThe system supports 22 languages spanning diverse writing systems: English, Spanish, French, German, Italian, Hindi, Marathi, Sanskrit, Kannada, Telugu, Tamil, Malayalam, Chinese, Japanese, Korean, Arabic, Bengali, Gujarati, Odia, Punjabi, Russian, and Thai. Whether you're working with Latin script, Devanagari, Chinese characters, Arabic script, or any combination thereof, NetraEmbed handles them with consistent reliability.\n\n## Two Models, Different Strengths\n\nWe're releasing two variants to serve different deployment needs:\n\n**[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed)** is our flagship single-vector model designed for large-scale deployments. It stores each document as a compact embedding-just 10 KB compared to 2.5 MB for traditional multi-vector approaches. This 250x efficie
1ncy gain means you can index millions of documents without breaking the bank on storage costs. For enterprises managing vast document repositories, this efficiency isn't just convenient-it's transformative.\n\nThe model offers flexible sizing through Matryoshka representation learning. Choose 768 dimensions for maximum speed and minimal storage (retaining 95% of full accuracy), 1536 dimensions for balanced performance, or the full 2560 dimensions for absolute maximum accuracy. The beauty of this approach? You can switch between these sizes without retraining or reloading the model.\n\n**[ColNetraEmbed](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)** takes a different approach with multi-vector representations, offering token-level matching for applications requiring fine-grained retrieval. It scores 0.637 on cross-lingual tasks and 0.670 on monolingual searches. While slightly less accurate than [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed), it provides superior interpretability-you can see exactly which parts of a document matched your query, making it ideal for compliance and legal applications where explainability matters.\n\n## Try It Yourself\n\nExperience NetraEmbed in action with our interactive demo. Upload your documents in any of the 22 supported languages and search across them-even with queries in a different language. The system understands both text and visual elements like charts and tables.\n\n**[Launch Interactive Demo â](https://huggingface.co/spaces/AdithyaSK/NetraEmbed)**\n\n\u003c!-- ### Watch NetraEmbed in Action\n\nSee how NetraEmbed handles multilingual document retrieval with a practical demonstration:\n\n\u003cdiv style=\"position: relative; width: 100%; padding-bottom: 56.25%; margin: 2rem 0;\"\u003e\n\u003ciframe style=\"position: absolute; top: 0; left: 0; width: 100%; height: 100%;\" src=\"https://www.youtube.com/embed/er80aybQ9is\" title=\"Nayana - Pioneering Multilingual Document AI Research\" frameborder=\"0\" allow=\"accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share\" referrerpolicy=\"strict-origin-when-cross-origin\" allowfullscreen\u003e\u003c/iframe\u003e\n\u003c/div\u003e --\u003e\n\n## Why Multimodal Matters\n\nThe multimodal capability isn't just a technical feature-it's a fundamental advantage for real-world document search:\n\n**Visual Information Preservation**: Financial reports aren't just text-they're charts showing trends, tables comparing metrics, and diagrams illustrating relationships. OCR-based systems extract the text but lose the visual context. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) sees the whole picture, literally. When you search for \"quarterly revenue comparison,\" it can find the bar chart showing Q1 vs Q2 vs Q3, even if that information isn't explicitly written in text.\n\n**Layout Understanding**: The way information is arranged on a page often conveys meaning. A two-column contract has different sections serving different purposes. A research paper's structure-abstract, methodology, results-carries semantic value. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) understands these structural patterns, improving retrieval accuracy for documents where layout matters.\n\n**Cost-Effective at Scale**: Running accurate OCR pipelines at scale can be expensive, especially for multilingual documents requiring language-specific processing. With modern vision-language models excelling at Document Visual Question Answering (DocVQA), there's a more efficient approach: embed document images directly with [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) for retrieval, then pass only the relevant retrieved documents to VLMs for question answering. This two-stage pipeline dramatically reduces costs-you only pay for expensive VLM inference on the handful of documents that matter, rather than processing your entire corpus through OCR and then VLMs. At enterprise scale with millions of documents, this architecture can reduce processing costs by orders of magnitude while maintaining high accuracy.\n\n**No OCR Errors**: Every additional processing step introduces errors. Poor scan quality, unusual fonts, mixed languages, or handwritten annotations can break OCR systems. By processing documents as images, [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) sidesteps these failure modes entirely. It works equally well with pristine PDFs and low-quality scans.\n\n**Complex Documents**: Try searching for information in a document mixing Hindi text with English technical terms, containing embedded charts, and using domain-specific symbols. Traditional systems struggle with even one of these challenges. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) handles all of them simultaneously because it processes the complete visual and textual context together.\n\n## Real-World Applications\n\nFor multinational corporations, NetraEmbed means your knowledge management system finally works across all your offices. That product specification in German? Instantly f
1indable by your team in India. Those compliance documents in Japanese? Accessible to your European auditors without translation delays.\n\nResearch institutions can now build truly global academic search systems. A researcher querying in English can discover relevant papers published in Chinese or Arabic, dramatically expanding the accessible knowledge base and accelerating scientific collaboration across linguistic boundaries.\n\nCultural heritage organizations can make their multilingual archives genuinely searchable. Historical documents in various regional languages become accessible to scholars worldwide, democratizing access to cultural knowledge that was previously locked behind language barriers.\n\n## Introducing NayanaIR Benchmark: A New Standard for Multilingual Document Retrieval\n\nA critical challenge in advancing multilingual document retrieval has been the absence of comprehensive evaluation frameworks. Existing benchmarks primarily focus on English or limited language coverage, making it impossible to rigorously evaluate systems across diverse writing systems and linguistic families. To address this gap, we're introducing **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)** - comprehensive multilingual multimodal document retrieval benchmark.\n\n**Why NayanaIR Matters:**\n\nThe benchmark provides **23 datasets** covering both cross-lingual retrieval (queries in one language, documents in another) and monolingual retrieval across all 22 supported languages. With nearly **28,000 document images** and over **5,400 queries** in BEIR-compatible format, it enables standardized evaluation across diverse script families including Latin, Devanagari, Dravidian, CJK, Arabic, and more.\n\nWhat makes NayanaIR particularly valuable is its **comprehe
1nsive coverage and balanced design**. Each monolingual dataset contains approximately 1,000 documents and 200 queries, while the cross-lingual dataset spans all 22 languages with 5,870 parallel documents. This balanced structure ensures fair comparison across languages and prevents bias toward high-resource languages. The benchmark uses industry-standard metrics (NDCG@5/10, Recall@5/10, MAP@10, MRR@10) and graded relevance scoring, making results directly comparable with existing English benchmarks.\n\nFor the research community, NayanaIR provides the foundation to develop and evaluate truly multilingual document retrieval systems. It removes the barrier of having to create language-specific evaluation datasets and enables researchers to measure progress across the full spectrum of linguistic diversity. Whether you're building commercial systems or advancing academic research, NayanaIR offers the rigorous evaluation framework needed to push the field forward.\n\n## The Technology Behind the Breakthrough\n\nOur M3DR (Multilingual Multimodal Document Retrieval) framework powers both [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) and [ColNetraEmbed](https://huggingface.co/Cognitive-Lab/ColNetraEmbed). We developed a sophisticated synthetic data generation pipeline that creates parallel multilingual document datasets while preserving layout and visual elements-critical for documents where structure conveys meaning.\n\nThe training process uses advanced query synthesis with large vision-language models, generating diverse question types across all 22 languages. This ensures the models don't just memorize patterns but truly understand the relationship between queries and document content across linguistic boun
1daries.\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) is part of our broader **[Nayana initiative](https://cognitivelab.in/nayana)**-a comprehensive effort to build multilingual, multimodal document intelligence that goes beyond retrieval to deep understanding. While [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) finds documents, our upcoming Nayana models will answer questions about them, extract insights, and enable true document comprehension across languages.\n\n## Getting Started\n\nReady to integrate [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) into your systems? Here's how the complete ingestion and retrieval workflow works:\n\n\n*Figure: End-to-end workflow showing document ingestion, embedding generation, and multilingual retrieval with NetraEmbed*\n\n### How It Works\n\n**Document Ingestion**: Your multilingual documents (PDFs, images, scans) are processed as images, preserving all visual elements including charts, tables, and layout. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) generates compact embeddings (10 KB per document) that capture both textual and visual information. These embeddings are stored in your vector database of choice (Pinecone, Weaviate, Qdrant, etc.).\n\n**Query Processing**: When a user submits a search query in any of the 22 supported languages, the query text is similarly embedded using [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed). The model understands the semantic intent across languages, enabling true cross-lingual retrieval.\n\n**Retrieval**: The system performs similarity search in your vector database to find the most relevant documents, regardless of the language mismatch between query and documents. Results are ranked by relevance, and you can retrieve documents in Japanese using an English query, or vice versa-all with state-of-the-art accuracy.\n\n**Integration with VLMs**: For question-answering workflows, pass the retrieved document images directly to vision-language models for DocVQA. This eliminates OCR overhead and cost, as you only run expensive VLM inference on the handful of relevant documents identified by [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed).\n\n### Resources\n\nBoth models are available on **[HuggingFace](https://huggingface.co/Cognitive-Lab)** with complete documentation and implementation guides. The **[NetraEmbed model](https://huggingface.co/Cognitive-Lab/NetraEmbed)** and **[ColNetraEmbed model](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)** include example code for common use cases, from simple document search to complex RAG pipelines.\n\nThe **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)** is also freely available for researchers and teams who want to evaluate performance on their specific language combinations or compare against their own systems.\n\nOur research paper, **\"M3DR: Towards Universal Multilingual Multimodal Document Retrieval\"**, details the complete methodology and is available on **[arXiv](https://arxiv.org/abs/2512.03514)**. The paper includes extensive ablation studies, architectural decisions, and performance analyses across all language pairs.\n\n## Performance at a Glance\n\n### Cross-Lingual Retrieval Performance\n\nHere's how [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) compares to existing solutions on cross-lingual retrieval (when query and document are in different languages):\n\n\n*Figure: Cross-lingual retrieval performance comparison showing NetraEmbed's 152% improvement over existing baselines*\n\n| Model | NDCG@5 | Recall@10 | MAP@10 | MRR@10 | Improvement over ColPali |\n|-------|--------|-----------|--------|--------|--------------------------|\n| **NetraEmbed (Ours)** | **0.716** | **0.871** | **0.703** | **0.775** | **+152%** |\n| **ColNetraEmbed (Ours)** | **0.637** | **0.700** | **0.610** | **0.610** | **+124%** |\n| Jina-Embeddings-v4 | 0.435 | 0.435 | 0.390 | 0.548 | +53% |\n| ColNomic-Embed-3B | 0.315 | 0.320 | 0.267 | 0.444 | +11% |\n| ColPali-v1.3 | 0.284 | 0.347 | 0.249 | 0.403 | Baseline |\n| ColQwen2.5-v0.2 | 0.143 | 0.160 | 0.127 | 0.220 | -50% |\n| GME-Qwen2-VL-2B | 0.235 | 0.308 | 0.209 | 0.314 | -17% |\n\n### Monolingual Retrieval Performance\n\nFor searches within a single language, [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) maintains exceptional consistency:\n\n\n*Figure: Monolingual retrieval performance showing consistent state-of-the-art results across all supported languages*\n\n| Model | NDCG@5 | Recall@10 | MAP@10 | MRR@10 |\n|-------|--------|-----------|--------|--------|\n| **NetraEmbed (Ours)** | **0.738** | **0.844** | **0.709** | **0.751** |\n| **ColNetraEmbed (Ours)** | **0.670** | **0.764** | **0.645** | **0.686** |\n| ColNomic-Embed-3B | 0.534 | 0.603 | 0.515 | 0.546 |\n| GME-Qwen2-VL-2B | 0.444 | 0.525 | 0.426 | 0.452 |\n| ColQwen2.5-v0.2 | 0.453 | 0.513 | 0.437 | 0.464 |\n| ColPali-v1.3 | 0.410 | 0.484 | 0.393 | 0.422 |\n\n### English Performance (ViDoRe v2 Benchmark)\n\nNetraEmbed maintains competitive English performance while excelling at multilingual tasks:\n\n| Model | NDCG@5 | Recall@10 | MAP@10 | MRR@10 |\n|-------|--------|-----------|--------|--------|\n| ColQwen2.5-v0.2 | 0.592 | 0.664 | 0.484 | 0.711 |\n| Jina-Embeddings-v4 | 0.576 | 0.686 | - | - |\n| GME-Qwen2-VL-2B | 0.574 | 0.630 | 0.466 | 0.690 |\n| ColNomic-Embed-3B | 0.556 | 0.633 | 0.451 | 0.672 |\n| **NetraEmbed (Ours)** | **0.554** | **0.637** | **0.437** | **0.647** |\n| **ColNetraEmbed (Ours)** | **0.551** | **0.664** | **0.445** | **0.645** |\n| ColQwen2-v1.0 | 0.545 | 0.640 | 0.438 | 0.653 |\n| ColPali-v1.3 | 0.538 | 0.627 | 0.436 | 0.644 |\n\n### Matryoshka Embedding Flexibility\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) offers flexible dimension sizes for different deployment scenarios:\n\n| Dimensions | Storage/Doc | NDCG@5 | Relative Performance | Best For |\n|------------|-------------|--------|---------------------|----------|\n| 768 | ~3 KB | 0.680 | 95.0% | Billion-scale deployments, edge devices |\n| 1536 | ~6 KB | 0.706 | 98.6% | Balanced production systems |\n| 2560 (full) | ~10 KB | 0.716 | 100.0% | Maximum accuracy requirements |\n\n**Key Insight**: The 768-dimensional variant retains 95% of full performance while reducing storage by 70%, making large-scale deployment economically viable.\n\n\n## What's Next: From Retrieval to Understanding\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) solves the critical first step-finding the right documents across languages. But this is just the beginning of our vision for multilingual document intelligence.\n\nAs part of our **[Nayana initiative](https://cognitivelab.in/nayana)**, we're developing specialized vision-language models that don't just retrieve documents-they understand and answer questions about them. Imagine uploading a financial report in Japanese and asking detailed questions in English: \"What was the revenue growth in Q3?\" or \"Which product line had the highest margin?\" These models will process the document image directly and provide accurate answers, preserving all visual context from charts and tables.\n\nThe Nayana family of models represents the next frontier in multilingual, multimodal document intelligence. Where [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) finds the right document, Nayana models will extract insights from it-across any of the 22 supported languages, understanding both text and visual elements like charts, diagrams, and complex layouts. This creates a complete pipeline: retrieve with [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed), then understand with Nayana.\n\nWe're also expanding language coverage beyond the current 22, with particular focus on truly low-resource languages where the need is greatest. Additionally, we're developing specialized fine-tuned variants for specific industries-legal document search, medical records retrieval, and academic research applications. Each domain has unique requirements, and we're building solutions that address them while maintaining the multilingual capabilities that make our models unique.\n\nFor enterprises interested in early access to Nayana models, custom deployments, or specialized language support, we offer consultation and collaboration opportunities. Reach out to our team at **[[email protected]](mailto:[email protected])** to discu
1ss your specific requirements.\n\n## Join the Community\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) is built on open research principles. We're releasing our models, the **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)**, and training methodology to enable the research community to build on this foundation. If you're working on multilingual AI, document understanding, or related challenges, we'd love to collaborate.\n\nExplore the **[models on HuggingFace](https://huggingface.co/Cognitive-Lab)**, read the **[research paper](https://arxiv.org/abs/2512.03514)**, and reach out if you have questions or feedback at **[[email protected]](mailto:[email protected])**. Together, we can make document intelligence truly universal.\n\n## Acknowledgments\n\nThis work benefited from compute credits for training, inference, and evaluation provided by **[Modal](https://modal.com)**, acknowledged as a compute sponsor. Dataset curation and synthesis were supported by the **[Meta Llama Impact Grant](https://about.fb.com/news/2025/04/llama-impact-grant-recipients/)** through our **[Nayana initiative](https://cognitivelab.in/nayana)**. We appreciate Meta for continued support of our research efforts at **[CognitiveLab](https://cognitivelab.in)**.\n\n---\n\n**Quick Links:**\n\n- **[NetraEmbed Model on HuggingFace](https://huggingface.co/Cognitive-Lab/NetraEmbed)** - Download and integrate the single-vector model\n- **[ColNetraEmbed Model on HuggingFace](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)** - Multi-vector variant for fine-grained retrieval\n- **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)** - Evaluation datasets across 22 languages\n- **[Nayana Initiative](https://cognitivelab.in/nayana)** - Next-generation document understanding models\n- **[Research Paper on arXiv](https://arxiv.org/abs/2512.03514)** - M3DR: Towards Universal Multilingual Multimodal Document Retrieval\n- **[CognitiveLab](https://cognitivelab.in)** - Explore our other AI research and products\n- **[Contact Us](mailto:[email protected])** - Questions, collaborations, or enterprise inquiries\n"])</script>
1<script>self.__next_f.push([1,"a:T1aad,"])</script>
1<script>self.__next_f.push([1,"\n# CognitiveLab Wins Meta's Prestigious Llama Impact Grant 2024\n\nWe are thrilled to announce that **CognitiveLab has been selected as a recipient of Meta's prestigious Llama Impact Grant 2024!** This significant recognition comes with substantial non-dilutive funding that will accelerate our mission to democratize AI across global languages.\n\n## About the Llama Impact Grant\n\nThe Llama Impact Grant is Meta's initiative to support innovative projects leveraging AI to drive positive societal impact. Selected from thousands of global applicants, CognitiveLab stood out for our transformative vision of building truly inclusive AI that serves diverse linguistic communities.\n\nThis grant provides:\n\n- Substantial financial support to accelerate our R\u0026D\n- Increased visibility in the global AI ecosystem\n\n## Powering Project Nayana: Our Vision for Inclusive AI\n\nThe Llama Impact Grant will primarily fuel our flagship initiative: **Project Nayana** (meaning \"eyes\" in Hindi)âa revolutionary, unified foundation model ecosystem designed to see, understand, and generate content across languages and modalities.\n\n### The AI Language Gap We're Solving\n\nToday's cutting-edge AI remains largely English-centric, creating a technological divide that excludes billions from fully participating in the AI revolution. This gap is particularly pronounced across India and the Global South, where linguistic diversity is rich but AI support is limited.\n\nThe consequences are significant:\n\n- Limited access to advanced AI tools for non-English speakers\n- Barriers to knowledge discovery in regional languages\n- Uneven distribution of AI benefits across global communities\n- Cultural and linguistic nuances lost in translation\n\n### Introducing Nayana: Breaking Down AI's Language Barriers\n\nProject Nayana represents our comprehensive solution to these challengesâa unified AI ecosystem with deep multilingual (22 languages, including 10 Indic), multimodal (vision + text), and multitask intelligence capabilities.\n\n#### What Makes Nayana Revolutionary:\n\n1. **Unprecedented Language Coverage:** While most AI models focus on a handful of major languages, Nayana supports 22 diverse languages, with special emphasis on 10 Indic languages that serve over 1 billion people.\n\n2. **Our Technical Breakthroughs:**\n\n - **Novel Synthetic Data Engine:** We've developed a groundbreaking pipeline that generates millions of high-fidelity, layout-preserving, annotated document images across all 22 target languages\n - **Unified Architecture:** Instead of fragmented pipelines, Nayana uses a single, powerful architecture for multiple tasks\n - **Cultural Context Preservation:** Our models understand cultural nuances and context-specific language use\n\n3. **Current Research Focus Areas:**\n - **Layout-Preserving Translation:** Maintaining document structure across languages\n - **Cross-Script Transfer Learning:** Leveraging patterns across different writing systems\n - **Efficient Multimodal Fusion:** Integrating text, vision, and potentially audio in resource-efficient ways\n - **Low-Resource Language Optimization:** Specialized techniques for languages with limited data\n\n#### The Nayana Ecosystem: Building Blocks of Inclusive AI\n\nNayana isn't just one model; it's a comprehensive ecosystem with multiple components:\n\n1. **Model Family (Powered by Llama Insights):**\n\n - **Nayana OCR:** Our award-winning OCR technology delivers unparalleled accuracy for 22 languages\n - **Nayana-2B:** Compact multilingual, multimodal model optimized for edge deployment\n - **Nayana-7B:** Our flagship model with advanced reasoning capabilities across all supported languages\n - **Nayana Retriever:** Specialized models for efficient cross-modal, multilingual information retrieval\n\n2. **Supporting Infrastructure:**\n - **NayanaBench:** Comprehensive evaluation framework for rigorous multilingual, multimodal testing\n - **ViViD Framework:** Our unified architecture handling multiple document intelligence tasks\n - **Open-Source Tooling:** Developer-friendly libraries and APIs for c
1ommunity adoption\n\n## Our Roadmap: What's Next for Nayana\n\nWith the Llama Impact Grant accelerating our progress, here's what we're working on:\n\n### Immediate Focus (Next 6 Months):\n\n- Completing our pre-training pipeline for all 22 languages\n- Releasing Nayana OCR open-source with comprehensive documentation\n- Publishing our synthetic data generation methodology\n- Establishing partnerships with key educational and governmental organizations\n\n### Medium-Term Goals (6-12 Months):\n\n- Launching Nayana-2B and Nayana-7B general-purpose models\n- Developing specialized fine-tuned versions for key sectors (education, healthcare, governance)\n- Creating an accessible API for developers to integrate Nayana capabilities\n- Expanding our benchmark suite to cover more languages and tasks\n\n### Long-Term Vision:\n\n- Extending language coverage to 50+ global languages\n- Integrating speech recognition and generation capabilities\n- Building specialized domain models for critical sectors\n- Establishing a sustainable open-source community around the project\n\n## Real-World Impact: How Nayana Will Transform Lives\n\nThe Llama Impact Grant significantly boosts our ability to deliver transformative solutions across sectors:\n\n- **Preserving Cultural Heritage:** Digitizing and making accessible millions of historical documents and manuscripts in regional languages\n- **Revolutionizing Education:** Creating AI tutors and translation tools that understand cultural context and regional educational needs\n- **Transforming Governance:** Making public services truly accessible in citizens' native languages\n- **Improving Healthcare:** Breaking down communication barriers in medical settings and enabling multilingual health records\n- **Driving Economic Opportunity:** Empowering local businesses with AI tools that understand regional markets and languages\n\n## Join Our Journey\n\nThe recognition from Meta's Llama Impact Grant marks a pivotal moment for CognitiveLab and for inclusive AI globally. We invite researchers, developers, organizations, and language enthusiasts to join us in building an AI future where language is never a barrier.\n\n**Get involved with Nayana:**\n\n- **[Explore the full Nayana ecosystem on our dedicated page](/nayana)**\n- **[Read Meta's Llama Impact Grant announcement](https://about.fb.com/news/2025/04/llama-impact-grant-recipients/?utm_source=AIatMeta\u0026utm_medium=organic_social\u0026utm_content=image\u0026utm_campaign=llamacon)**\n- **[Check out our models and research on Hugging Face](https://huggingface.co/Nayana-cognitivelab)**\n\nFor enterprise inquiries, research collaborations, and partnership opportunities, please contact us at **[email protected]**.\n\nLet's build an AI that truly speaks everyone's language.\n"])</script>
1<script>self.__next_f.push([1,"b:Tea6,"])</script>
1<script>self.__next_f.push([1,"\n# Introducing AI Engineering Academy\n\n\u003c!--\n\u003cdiv align=\"center\"\u003e\n \u003ch1 \u003e\u003ca href=\"https://aiengineering.academy/\" target=\"_blank\"\u003eAIEngineering.academy\u003c/a\u003e\u003c/h1\u003e\n \u003cp\u003eNavigating the World of AI, One Step at a Time\u003c/p\u003e\n \u003cp\u003eInitiative from \u003ca href=\"https://cognitivelab.in/\" target=\"_blank\"\u003eCognitiveLab\u003c/a\u003e\u003c/p\u003e\n\u003c/div\u003e\n\n\u003cimg src=\"./assets/banner.png\" alt=\"Ai Engineering. Academy\"\u003e\n\n[](https://github.com/adithya-s-k/AI-Engineering.academy/stargazers)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/network/members)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/issues)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/pulls)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/blob/main/LICENSE) --\u003e\n\nAI-related careers are becoming increasingly sought-after. However, the abundance of learning resources scattered across the internet can lead to confusion about where to start.\u0026#x20;\n\n**AI Engineering Academy** aims to provide a structured learning path to help you learn Applied GenAI effectively.\n\n## Roadmap\n\n### [**1. Prompt Engineering**](PromptEngineering/)\n\nThis roadmap will cover the basics of prompt engineering and its role in various AI applications.\n\n### [**2. Retrieval Augmented Generation (RAG)**](RAG/)\n\nIf you've been in the AI space, you might have heard of RAG (Retrieval Augmented Generation). In this roadmap, we will cover:\n\n- Understanding what RAG is\n- Implementing RAG from scratch without using any frameworks\n- Choosing the best RAG system for your needs\n- Taking a RAG system to production\n\n### [**3. Fine-tuning**](Finetuning/)\n\n**(coming soon)**\n\nWe will debunk some myths about fine-tuning and explore where it can be effectively used. Fine-tuning can be a powerful tool, but it's often misunderstood. Expect to learn:\n\n- The true potential of fine-tuning\n- Proper techniques for fine-tuning models\n\n### [**4. Deployment**](Deployment/)\n\n**(coming soon)**\n\nWhile everything might work perfectly locally, putting models into the hands of users requires careful consideration. In this roadmap, we will cover:\n\n- Deploying ML and DL models into production\n- Working with different cloud providers\n- Exploring various deployment modes\n\n### [**5. AI Agents**](Agents/)\n\n**(coming soon)**\n\n### [**6. Projects**](Projects/)\n\n**(coming soon)**\n\nConsists of multiple hands-on end-to-end AI projects!!\n\n## Repo Admin ð¨âð¼\n\n\u003cdiv align=\"center\"\u003e\n \u003ctable\u003e\n \u003ctr\u003e\n \u003ctd align=\"center\"\u003e\u003ca href=\"https://github.com/adithya-s-k\"\u003e\u003cimg src=\"https://avatars.githubusercontent.com/u/27956426?s=400\u0026u=582ecb2d706a63fc67eb1b54579c7ab19cf391fd\u0026v=4\" width=\"100px;\" alt=\"\"/\u003e\u003cbr /\u003e\u003csub\u003e\u003cb\u003eAdithya S Kolavi\u003c/b\u003e\u003c/sub\u003e\u003c/a\u003e\u003cbr /\u003e\u003ca href=\"https://github.com/adithya-s-k/AI-Engineering.academy/commits?author=adithya-s-k\" title=\"Code\"\u003eð»\u003c/a\u003e\u003c/td\u003e\n \u003c/tr\u003e\n \u003c/table\u003e\n\u003c/div\u003e\n\n## Contributors ð\n\n\u003ca href=\"https://github.com/adithya-s-k/AI-Engineering.academy/graphs/contributors\"\u003e\n \u003cimg src=\"https://contrib.rocks/image?repo=adithya-s-k/AI-Engineering.academy\" /\u003e\n\nThanks to these wonderful people for their contributions!\n\u003c/a\u003e\n\n\u003cp align=\"center\"\u003e\n \u003ca href=\"https://adithyask.com\"\u003e\n \u003cimg src=\"https://api.star-history.com/svg?repos=adithya-s-k/AI-Engineering.academy\u0026type=Date\" alt=\"Star History Chart\"\u003e\n \u003c/a\u003e\n\u003c/p\u003e\n"])</script>
1<script>self.__next_f.push([1,"c:T306d,"])</script>
1<script>self.__next_f.push([1,"\n# Introducing Omniparse: Universal Document Data Extraction\n\n\n[](https://github.com/adithya-s-k/omniparse/stargazers)\n[](https://github.com/adithya-s-k/omniparse/network/members)\n[](https://github.com/adithya-s-k/omniparse/issues)\n[](https://github.com/adithya-s-k/omniparse/pulls)\n[](https://github.com/adithya-s-k/omniparse/blob/main/LICENSE)\n\n\u003e [!IMPORTANT]\n\u003e\n\u003e OmniParse is a platform that ingests and parses any unstructured data into structured, actionable data optimized for GenAI (LLM) applications. Whether you are working with documents, tables, images, videos, audio files, or web pages, OmniParse prepares your data to be clean, structured, and ready for AI applications such as RAG, fine-tuning, and more\n\n## Try it out\n\n[](https://colab.research.google.com/github/adithya-s-k/omniparse/blob/main/examples/OmniParse_GoogleColab.ipynb)\n\n## Intro\n\nhttps://github.com/adithya-s-k/omniparse/assets/27956426/457d8b5b-9573-44da-8bcf-616000651a13\n\n## Features\n\nâ
Completely local, no external APIs \\\nâ
Fits in a T4 GPU \\\nâ
Supports ~20 file types \\\nâ
Convert documents, multimedia, and web pages to high-quality structured markdown \\\nâ
Table extraction, image extraction/captioning, audio/video transcription, web page crawling \\\nâ
Easily deployable using Docker and Skypilot \\\nâ
Colab friendly \\\nâ
Interative UI powered by Gradio\n\n### Why OmniParse ?\n\nIt's challenging to process data as it comes in different shapes and sizes. OmniParse aims to be an ingestion/parsing platform where you can ingest any type of data, such as documents, images, audio, video, and web content, and get the most structured and actionable output that is GenAI (LLM) friendly.\n\n## Installation\n\n\u003e [!IMPORTANT]\n\u003e The server only works on Linux-based systems. This is due to certain dependencies and system-specific configurations that are not compatible with Windows or macOS.\n\n```bash\ngit clone https://github.com/adithya-s-k/omniparse\ncd omniparse\n```\n\nCreate a Virtual Environment:\n\n```bash\nconda create -n omniparse-venv python=3.10\nconda activate omniparse-venv\n```\n\nInstall Dependencies:\n\n```bash\npoetry install\n# or\npip install -e .\n# or\npip install -r pyproject.toml\n```\n\n### ð³ï¸ Docker\n\nTo use OmniParse with Docker, execute the following commands:\n\n1. Pull the OmniParse API Docker image from Docker Hub:\n2. Run the Docker container, exposing port 8000:\n ðð¼[Docker Image](https://hub.docker.com/r/savatar101/omniparse)\n\n```bash\ndocker pull savatar101/omniparse:0.1\n# if you are running on a gpu\ndocker run --gpus all -p 8000:8000 savatar101/omniparse:0.1\n# else\ndocker run -p 8000:8000 savatar101/omniparse:0.1\n```\n\nAlternatively, if you prefer to build the Docker image locally:\nThen, run the Docker container as follows:\n\n```bash\ndocker build -t omniparse .\n# if you are running on a gpu\ndocker run --gpus all -p 8000:8000 omniparse\n# else\ndocker run -p 8000:8000 omniparse\n\n```\n\n## Usage\n\nRun the Server:\n\n```bash\npython server.py --host 0.0.0.0 --p
1ort 8000 --documents --media --web\n```\n\n- `--documents`: Load in all the models that help you parse and ingest documents (Surya OCR series of models and Florence-2).\n- `--media`: Load in Whisper model to transcribe audio and video files.\n- `--web`: Set up selenium crawler.\n\nDownload Models:\nIf you want to download the models before starting the server\n\n```bash\npython download.py --documents --media --web\n```\n\n- `--documents`: Load in all the models that help you parse and ingest documents (Surya OCR series of models and Florence-2).\n- `--media`: Load in Whisper model to transcribe audio and video files.\n- `--web`: Set up selenium crawler.\n\n## Supported Data Types\n\n| Type | Supported Extensions |\n| --------- | --------------------------------------- |\n| Documents | .doc, .docx, .pdf, .ppt, .pptx |\n| Images | .png, .jpg, .jpeg, .tiff, .bmp, .heic |\n| Video | .mp4, .mkv, .avi, .mov |\n| Audio | .mp3, .wav, .aac |\n| Web | dynamic webpages, http://\u003canything\u003e.com |\n\n\u003cdetails\u003e\n\u003csummary\u003e\u003ch2\u003eAPI Endpoints\u003c/h2\u003e\u003c/summary\u003e\n\n\u003e Client library compatible with Langchain, llamaindex, and haystack integrations coming soon.\n\n- [OmniParse](#omniparse)\n - [Try it out](#try-it-out)\n - [Intro](#intro)\n - [Features](#features)\n - [Why OmniParse ?](#why-omniparse-)\n - [Installation](#installation)\n - [ð³ï¸ Docker](#ï¸-docker)\n - [Usage](#usage)\n - [Supported Data Types](#supported-data-types)\n - [Document Parsing](#document-parsing)\n - [Parse Any Document](#parse-any-document)\n - [Parse PDF](#parse-pdf)\n - [Parse PowerPoint](#parse-powerpoint)\n - [Parse Word Document](#parse-word-document)\n - [Media Parsing](#media-parsing)\n - [Parse Image](#parse-image)\n - [Process Image](#process-image)\n - [Parse Video](#parse-video)\n - [Parse Audio](#parse-audio)\n - [Website Parsing](#website-parsing)\n - [Parse Website](#parse-website)\n - [Coming Soon/ RoadMap](#coming-soon-roadmap)\n - [Limitations](#limitations)\n - [License](#license)\n - [Commercial Usage](#commercial-usage)\n - [Acknowledgements](#acknowledgements)\n - [Contact](#contact)\n\n### Document Parsing\n\n#### Parse Any Document\n\nEndpoint: `/parse_document`\nMethod: POST\n\nParses PDF, PowerPoint, or Word documents.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/document\" http://localhost:8000/parse_document\n```\n\n#### Parse PDF\n\nEndpoint: `/parse_document/pdf`\nMethod: POST\n\nParses PDF documents.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/document.pdf\" http://localhost:8000/parse_document/pdf\n```\n\n#### Parse PowerPoint\n\nEndpoint: `/parse_document/ppt`\nMethod: POST\n\nParses PowerPoint presentations.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/presentation.ppt\" http://localhost:8000/parse_document/ppt\n```\n\n#### Parse Word Document\n\nEndpoint: `/parse_document/docs`\nMethod: POST\n\nParses Word documents.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/document.docx\" http://localhost:8000/parse_document/docs\n```\n\n### Media Parsing\n\n\u003c!-- #### Parse Any Media\n\nEndpoint: `/parse_media`\nMethod: POST\n\nParses images, videos, or audio files.\n\nCurl command:\n```\ncurl -X POST -F \"file=@/path/to/media_file\" http://localhost:8000/parse_media\n``` --\u003e\n\n#### Parse Image\n\nEndpoint: `/parse_image/image`\nMethod: POST\n\nParses image files (PNG, JPEG, JPG, TIFF, WEBP).\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/image.jpg\" http://localhost:8000/parse_media/image\n```\n\n#### Process Image\n\nEndpoint: `/parse_image/process_image`\nMethod: POST\n\nProcesses an image with a specific task.\n\nPossible task inputs:\n`OCR | OCR with Region | Caption | Detailed Caption | More Detailed Caption | Object Detection | Dense Region Caption | Region Proposal`\n\nCurl command:\n\n```\ncurl -X POST -F \"image=@/path/to/image.jpg\" -F \"task=Caption\" -F \"prompt=Optional prompt\" http://localhost:8000/parse_media/process_image\n```\n\nArguments:\n\n- `image`: The image file\n- `task`: The processing task (e.g., Caption, Object Detection)\n- `prompt`: Optional prompt for certain tasks\n\n#### Parse Video\n\nEndpoint: `/parse_media/video`\nMethod: POST\n\nParses video files (MP4, AVI, MOV, MKV).\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/video.mp4\" http://localhost:8000/parse_media/video\n```\n\n#### Parse Audio\n\nEndpoint: `/parse_media/audio`\nMethod: POST\n\nParses audio files (MP3, WAV, FLAC).\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/audio.mp3\" http://localhost:8000/parse_media/audio\n```\n\n### Website Parsing\n\n#### Parse Website\n\nEndpoint: `/parse_website/parse`\nMethod: POST\n\nParses a website given its URL.\n\nCurl command:\n\n```\ncurl -X POST -H \"Content-Type: application/json\" -d '{\"url\": \"https://example.com\"}' http://localhost:8000/parse_website\n```\n\nArguments:\n\n- `url`: The URL of the website to parse\n\n\u003c/details\u003e\n\n## Coming Soon/ RoadMap\n\nð¦ LlamaIndex | Langchain | Haystack integrations coming soon\nð Batch processing data\nâ Dynamic chunking and structured data extraction based on specified Schema \nð ï¸ One magic API: just feed in your file prompt what you want, and we will take care of the rest \nð§ Dynamic model selection and support for external APIs \nð Batch processing for handling multiple files at once \nð¦ New open-source model to replace Surya OCR and Marker\n\n**Final goal**: replace all the different models currently being used with a single MultiModel Model to parse any type of data and get the data you need.\n\n## Limitations\n\nThere is a need for a GPU with 8~10 GB minimum VRAM as we are using deep learning models.\n\\\n\nDocument Parsing Limitations\n\\\n\n- [Marker](https://github.com/VikParuchuri/marker) which is the underlying PDF parser will not convert 100% of equations to LaTe
1X because it has to detect and then convert them.\n- It is good at parsing english but might struggle for languages such as Chinese\n- Tables are not always formatted 100% correctly; text can be in the wrong column.\n- Whitespace and indentations are not always respected.\n- Not all lines/spans will be joined properly.\n- This works best on digital PDFs that won't require a lot of OCR. It's optimized for speed, and limited OCR is used to fix errors.\n- To fit all the models in the GPU, we are using the smallest variants, which might not offer the best-in-class performance.\n\n## License\n\nOmniParse is licensed under the GPL-3.0 license. See `LICENSE` for more information.\nThe project uses Marker under the hood, which has a commercial license that needs to be followed. Here are the details:\n\n### Commercial Usage\n\nMarker and Surya OCR Models are designed to be as widely accessible as possible while still funding development and training costs. Research and personal usage are always allowed, but there are some restrictions on commercial usage.\nThe weights for the models are licensed under cc-by-nc-sa-4.0. However, this restriction is waived for any organization with less than $5M USD in gross revenue in the most recent 12-month period AND less than $5M in lifetime VC/angel funding raised. To remove the GPL license requirements (dual-license) and/or use the weights commercially over the revenue limit, check out the options provided.\nPlease refer to [Marker](https://github.com/VikParuchuri/marker) for more Information about the License of the Model weights\n\n## Acknowledgements\n\nThis project builds upon the remarkable [Marker](https://github.com/VikParuchuri/marker) project created by [Vik Paruchuri](https://twitter.com/VikParuchuri). We express our gratitude for the inspiration and foundation provided by this project. Special thanks to [Surya-OCR](https://github.com/VikParuchuri/surya) and [Texify](https://github.com/VikParuchuri/texify) for the OCR models extensively used in this project, and to [Crawl4AI](https://github.com/unclecode/crawl4ai) for their contributions.\n\nModels being used:\n\n- Surya OCR, Detect, Layout, Order, and Texify\n- Florence-2 base\n- Whisper Small\n\nThank you to the authors for their contributions to these models.\n\n---\n\n## Contact\n\n\u003cp align=\"center\"\u003e\n \u003ca href=\"https://adithyask.com\"\u003e\n \u003cimg src=\"https://api.star-history.com/svg?repos=adithya-s-k/omniparse\u0026type=Date\" alt=\"Star History Chart\"\u003e\n \u003c/a\u003e\n\u003c/p\u003e\nFor any inquiries, please contact us at [email protected]\n\n\u003c!--\nInstall the client:\n\n```bash\npip install omniparse_client\n```\n\nExample usage:\n\n```python\nfrom omniparse_client import OmniParse\n\n# Initialize the parser\nparser = OmniParse(\n base_url=\"http://localhost:8000\",\n api_key=\"op-...\", # get the API key from dev.omniparse.com\n verbose=True,\n language=\"en\"\n)\n\n# Parse a document\ndocument = parser.load_data('path/to/document.pdf')\n\n# Convert to markdown\nparser.save_to_markdown(document)\n```\n --\u003e\n"])</script>
1<script>self.__next_f.push([1,"12:{\"name\":\"CognitiveLab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"}\n13:{\"url\":\"/assets/blog/introducing-netraembed/cover.png\"}\n14:[\"Multilingual AI\",\"Document Retrieval\",\"Vision-Language Models\",\"RAG\"]\n15:T60b7,"])</script>
1<script>self.__next_f.push([1,"\n## The Multilingual Document Challenge\n\nA pharmaceutical manufacturer's quality team [spends 30% of their time](https://www.mastercontrol.com/uk/gxp-lifeline/electronic-document-management-for-multilingual-compliance/) just maintaining consistency between language versions of the same document across facilities in Poland, Portugal, and Germany. A construction project manager in Singapore needs critical safety specifications from a Korean supplier's documentation, but language barriers delay risk identification. A customer support agent in Brazil searches their knowledge base for a solution-only to find the answer exists in English documentation they can't effectively access.\n\nThese aren't edge cases. They represent the daily reality facing [global enterprises, researchers, and organizations](https://www.moveworks.com/us/en/resources/blog/ai-powered-multilingual-it-support) operating across linguistic boundaries. While AI has made tremendous strides in understanding English documents, [76% of online shoppers prefer information in their native language](https://www.helpscout.com/blog/multilingual-ai-support/), and [40% will never buy from websites in other languages](https://www.helpscout.com/blog/multilingual-ai-support/). For businesses with global operations, multilingual documentation isn't a nice-to-have-it's mission-critical.\n\nThe technical reality is even more challenging. When master documents require updates, [synchronizing changes across multiple language versions creates a documentation management nightmare](https://www.mastercontrol.com/uk/gxp-lifeline/electronic-document-management-for-multilingual-compliance/). Documentation delays from multilingual review cycles slow batch releases and impact market availability. For [construction projects overseas](https://ascelibrary.org/doi/10.1061/JCEMD4.COENG-14273), the inability to quickly search and retrieve information from foreign-language documents increases project risk and costs.\n\nThe numbers tell a stark story: existing state-of-the-art document retrieval systems score just 0.284 on cross-lingual retrieval tasks-performance so poor it's essentially unusable in production. In our testing, these systems consistently fail to find relevant documents when queries and documents are in different languages, making them unreliable for real-world multilingual workflows. When your business operates in multiple languages, this gap translates directly into lost productivity, missed opportunities, and frustrated teams.\n\n## Introducing NetraEmbed: Breaking the Language Barrier\n\nToday, we're excited to announce **[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed)** and **[ColNetraEmbed](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)**, two models that fundamentally change what's possible in multilingual and multimodal document retrieval. These aren't incremental improvements-they represent a 152% leap forward in performance, bringing cross-lingual document search from barely functional to genuinely useful.\n\nWhat makes this particularly powerful is the **multimodal** approach. Traditional document search systems rely on extracting text through OCR, which loses critical information like charts, diagrams, tables, and document layout. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) processes documents as images, preserving all visual elements and understanding how information is structured on the page. This means you can search for \"revenue growth chart\" and find the actual chart, or search for \"organizational hierarchy\" and locate the relevant diagram-across any of the 22 supported languages.\n\n## The Numbers That Matter\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) achieves a score of **0.716 NDCG@5** on cross-lingual retrieval, compared to the previous best of 0.284. In practical terms, this means the system can now reliably find the right document even when your query and documents are in different languages. For monolingual searches, the performance is even better at **0.738 NDCG@5**, an 80% improvement over existing solutions.\n\nHere's what makes this particularly compelling for businesses: we didn't sacrifice English performance to achieve multilingual capability. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) scores **0.554** on standard English benchmarks, remaining competitive with models designed exclusively for English. You get global capability without compromising on your primary language.\n\nThe system supports 22 languages spanning diverse writing systems: English, Spanish, French, German, Italian, Hindi, Marathi, Sanskrit, Kannada, Telugu, Tamil, Malayalam, Chinese, Japanese, Korean, Arabic, Bengali, Gujarati, Odia, Punjabi, Russian, and Thai. Whether you're working with Latin script, Devanagari, Chinese characters, Arabic script, or any combination thereof, NetraEmbed handles them with consistent reliability.\n\n## Two Models, Different Strengths\n\nWe're releasing two variants to serve different deployment needs:\n\n**[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed)** is our flagship single-vector model designed for large-scale deployments. It stores each document as a compact embedding-just 10 KB compared to 2.5 MB for traditional multi-vector approaches. This 250x efficie
1ncy gain means you can index millions of documents without breaking the bank on storage costs. For enterprises managing vast document repositories, this efficiency isn't just convenient-it's transformative.\n\nThe model offers flexible sizing through Matryoshka representation learning. Choose 768 dimensions for maximum speed and minimal storage (retaining 95% of full accuracy), 1536 dimensions for balanced performance, or the full 2560 dimensions for absolute maximum accuracy. The beauty of this approach? You can switch between these sizes without retraining or reloading the model.\n\n**[ColNetraEmbed](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)** takes a different approach with multi-vector representations, offering token-level matching for applications requiring fine-grained retrieval. It scores 0.637 on cross-lingual tasks and 0.670 on monolingual searches. While slightly less accurate than [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed), it provides superior interpretability-you can see exactly which parts of a document matched your query, making it ideal for compliance and legal applications where explainability matters.\n\n## Try It Yourself\n\nExperience NetraEmbed in action with our interactive demo. Upload your documents in any of the 22 supported languages and search across them-even with queries in a different language. The system understands both text and visual elements like charts and tables.\n\n**[Launch Interactive Demo â](https://huggingface.co/spaces/AdithyaSK/NetraEmbed)**\n\n\u003c!-- ### Watch NetraEmbed in Action\n\nSee how NetraEmbed handles multilingual document retrieval with a practical demonstration:\n\n\u003cdiv style=\"position: relative; width: 100%; padding-bottom: 56.25%; margin: 2rem 0;\"\u003e\n\u003ciframe style=\"position: absolute; top: 0; left: 0; width: 100%; height: 100%;\" src=\"https://www.youtube.com/embed/er80aybQ9is\" title=\"Nayana - Pioneering Multilingual Document AI Research\" frameborder=\"0\" allow=\"accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share\" referrerpolicy=\"strict-origin-when-cross-origin\" allowfullscreen\u003e\u003c/iframe\u003e\n\u003c/div\u003e --\u003e\n\n## Why Multimodal Matters\n\nThe multimodal capability isn't just a technical feature-it's a fundamental advantage for real-world document search:\n\n**Visual Information Preservation**: Financial reports aren't just text-they're charts showing trends, tables comparing metrics, and diagrams illustrating relationships. OCR-based systems extract the text but lose the visual context. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) sees the whole picture, literally. When you search for \"quarterly revenue comparison,\" it can find the bar chart showing Q1 vs Q2 vs Q3, even if that information isn't explicitly written in text.\n\n**Layout Understanding**: The way information is arranged on a page often conveys meaning. A two-column contract has different sections serving different purposes. A research paper's structure-abstract, methodology, results-carries semantic value. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) understands these structural patterns, improving retrieval accuracy for documents where layout matters.\n\n**Cost-Effective at Scale**: Running accurate OCR pipelines at scale can be expensive, especially for multilingual documents requiring language-specific processing. With modern vision-language models excelling at Document Visual Question Answering (DocVQA), there's a more efficient approach: embed document images directly with [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) for retrieval, then pass only the relevant retrieved documents to VLMs for question answering. This two-stage pipeline dramatically reduces costs-you only pay for expensive VLM inference on the handful of documents that matter, rather than processing your entire corpus through OCR and then VLMs. At enterprise scale with millions of documents, this architecture can reduce processing costs by orders of magnitude while maintaining high accuracy.\n\n**No OCR Errors**: Every additional processing step introduces errors. Poor scan quality, unusual fonts, mixed languages, or handwritten annotations can break OCR systems. By processing documents as images, [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) sidesteps these failure modes entirely. It works equally well with pristine PDFs and low-quality scans.\n\n**Complex Documents**: Try searching for information in a document mixing Hindi text with English technical terms, containing embedded charts, and using domain-specific symbols. Traditional systems struggle with even one of these challenges. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) handles all of them simultaneously because it processes the complete visual and textual context together.\n\n## Real-World Applications\n\nFor multinational corporations, NetraEmbed means your knowledge management system finally works across all your offices. That product specification in German? Instantly f
1indable by your team in India. Those compliance documents in Japanese? Accessible to your European auditors without translation delays.\n\nResearch institutions can now build truly global academic search systems. A researcher querying in English can discover relevant papers published in Chinese or Arabic, dramatically expanding the accessible knowledge base and accelerating scientific collaboration across linguistic boundaries.\n\nCultural heritage organizations can make their multilingual archives genuinely searchable. Historical documents in various regional languages become accessible to scholars worldwide, democratizing access to cultural knowledge that was previously locked behind language barriers.\n\n## Introducing NayanaIR Benchmark: A New Standard for Multilingual Document Retrieval\n\nA critical challenge in advancing multilingual document retrieval has been the absence of comprehensive evaluation frameworks. Existing benchmarks primarily focus on English or limited language coverage, making it impossible to rigorously evaluate systems across diverse writing systems and linguistic families. To address this gap, we're introducing **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)** - comprehensive multilingual multimodal document retrieval benchmark.\n\n**Why NayanaIR Matters:**\n\nThe benchmark provides **23 datasets** covering both cross-lingual retrieval (queries in one language, documents in another) and monolingual retrieval across all 22 supported languages. With nearly **28,000 document images** and over **5,400 queries** in BEIR-compatible format, it enables standardized evaluation across diverse script families including Latin, Devanagari, Dravidian, CJK, Arabic, and more.\n\nWhat makes NayanaIR particularly valuable is its **comprehe
1nsive coverage and balanced design**. Each monolingual dataset contains approximately 1,000 documents and 200 queries, while the cross-lingual dataset spans all 22 languages with 5,870 parallel documents. This balanced structure ensures fair comparison across languages and prevents bias toward high-resource languages. The benchmark uses industry-standard metrics (NDCG@5/10, Recall@5/10, MAP@10, MRR@10) and graded relevance scoring, making results directly comparable with existing English benchmarks.\n\nFor the research community, NayanaIR provides the foundation to develop and evaluate truly multilingual document retrieval systems. It removes the barrier of having to create language-specific evaluation datasets and enables researchers to measure progress across the full spectrum of linguistic diversity. Whether you're building commercial systems or advancing academic research, NayanaIR offers the rigorous evaluation framework needed to push the field forward.\n\n## The Technology Behind the Breakthrough\n\nOur M3DR (Multilingual Multimodal Document Retrieval) framework powers both [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) and [ColNetraEmbed](https://huggingface.co/Cognitive-Lab/ColNetraEmbed). We developed a sophisticated synthetic data generation pipeline that creates parallel multilingual document datasets while preserving layout and visual elements-critical for documents where structure conveys meaning.\n\nThe training process uses advanced query synthesis with large vision-language models, generating diverse question types across all 22 languages. This ensures the models don't just memorize patterns but truly understand the relationship between queries and document content across linguistic boun
1daries.\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) is part of our broader **[Nayana initiative](https://cognitivelab.in/nayana)**-a comprehensive effort to build multilingual, multimodal document intelligence that goes beyond retrieval to deep understanding. While [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) finds documents, our upcoming Nayana models will answer questions about them, extract insights, and enable true document comprehension across languages.\n\n## Getting Started\n\nReady to integrate [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) into your systems? Here's how the complete ingestion and retrieval workflow works:\n\n\n*Figure: End-to-end workflow showing document ingestion, embedding generation, and multilingual retrieval with NetraEmbed*\n\n### How It Works\n\n**Document Ingestion**: Your multilingual documents (PDFs, images, scans) are processed as images, preserving all visual elements including charts, tables, and layout. [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) generates compact embeddings (10 KB per document) that capture both textual and visual information. These embeddings are stored in your vector database of choice (Pinecone, Weaviate, Qdrant, etc.).\n\n**Query Processing**: When a user submits a search query in any of the 22 supported languages, the query text is similarly embedded using [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed). The model understands the semantic intent across languages, enabling true cross-lingual retrieval.\n\n**Retrieval**: The system performs similarity search in your vector database to find the most relevant documents, regardless of the language mismatch between query and documents. Results are ranked by relevance, and you can retrieve documents in Japanese using an English query, or vice versa-all with state-of-the-art accuracy.\n\n**Integration with VLMs**: For question-answering workflows, pass the retrieved document images directly to vision-language models for DocVQA. This eliminates OCR overhead and cost, as you only run expensive VLM inference on the handful of relevant documents identified by [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed).\n\n### Resources\n\nBoth models are available on **[HuggingFace](https://huggingface.co/Cognitive-Lab)** with complete documentation and implementation guides. The **[NetraEmbed model](https://huggingface.co/Cognitive-Lab/NetraEmbed)** and **[ColNetraEmbed model](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)** include example code for common use cases, from simple document search to complex RAG pipelines.\n\nThe **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)** is also freely available for researchers and teams who want to evaluate performance on their specific language combinations or compare against their own systems.\n\nOur research paper, **\"M3DR: Towards Universal Multilingual Multimodal Document Retrieval\"**, details the complete methodology and is available on **[arXiv](https://arxiv.org/abs/2512.03514)**. The paper includes extensive ablation studies, architectural decisions, and performance analyses across all language pairs.\n\n## Performance at a Glance\n\n### Cross-Lingual Retrieval Performance\n\nHere's how [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) compares to existing solutions on cross-lingual retrieval (when query and document are in different languages):\n\n\n*Figure: Cross-lingual retrieval performance comparison showing NetraEmbed's 152% improvement over existing baselines*\n\n| Model | NDCG@5 | Recall@10 | MAP@10 | MRR@10 | Improvement over ColPali |\n|-------|--------|-----------|--------|--------|--------------------------|\n| **NetraEmbed (Ours)** | **0.716** | **0.871** | **0.703** | **0.775** | **+152%** |\n| **ColNetraEmbed (Ours)** | **0.637** | **0.700** | **0.610** | **0.610** | **+124%** |\n| Jina-Embeddings-v4 | 0.435 | 0.435 | 0.390 | 0.548 | +53% |\n| ColNomic-Embed-3B | 0.315 | 0.320 | 0.267 | 0.444 | +11% |\n| ColPali-v1.3 | 0.284 | 0.347 | 0.249 | 0.403 | Baseline |\n| ColQwen2.5-v0.2 | 0.143 | 0.160 | 0.127 | 0.220 | -50% |\n| GME-Qwen2-VL-2B | 0.235 | 0.308 | 0.209 | 0.314 | -17% |\n\n### Monolingual Retrieval Performance\n\nFor searches within a single language, [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) maintains exceptional consistency:\n\n\n*Figure: Monolingual retrieval performance showing consistent state-of-the-art results across all supported languages*\n\n| Model | NDCG@5 | Recall@10 | MAP@10 | MRR@10 |\n|-------|--------|-----------|--------|--------|\n| **NetraEmbed (Ours)** | **0.738** | **0.844** | **0.709** | **0.751** |\n| **ColNetraEmbed (Ours)** | **0.670** | **0.764** | **0.645** | **0.686** |\n| ColNomic-Embed-3B | 0.534 | 0.603 | 0.515 | 0.546 |\n| GME-Qwen2-VL-2B | 0.444 | 0.525 | 0.426 | 0.452 |\n| ColQwen2.5-v0.2 | 0.453 | 0.513 | 0.437 | 0.464 |\n| ColPali-v1.3 | 0.410 | 0.484 | 0.393 | 0.422 |\n\n### English Performance (ViDoRe v2 Benchmark)\n\nNetraEmbed maintains competitive English performance while excelling at multilingual tasks:\n\n| Model | NDCG@5 | Recall@10 | MAP@10 | MRR@10 |\n|-------|--------|-----------|--------|--------|\n| ColQwen2.5-v0.2 | 0.592 | 0.664 | 0.484 | 0.711 |\n| Jina-Embeddings-v4 | 0.576 | 0.686 | - | - |\n| GME-Qwen2-VL-2B | 0.574 | 0.630 | 0.466 | 0.690 |\n| ColNomic-Embed-3B | 0.556 | 0.633 | 0.451 | 0.672 |\n| **NetraEmbed (Ours)** | **0.554** | **0.637** | **0.437** | **0.647** |\n| **ColNetraEmbed (Ours)** | **0.551** | **0.664** | **0.445** | **0.645** |\n| ColQwen2-v1.0 | 0.545 | 0.640 | 0.438 | 0.653 |\n| ColPali-v1.3 | 0.538 | 0.627 | 0.436 | 0.644 |\n\n### Matryoshka Embedding Flexibility\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) offers flexible dimension sizes for different deployment scenarios:\n\n| Dimensions | Storage/Doc | NDCG@5 | Relative Performance | Best For |\n|------------|-------------|--------|---------------------|----------|\n| 768 | ~3 KB | 0.680 | 95.0% | Billion-scale deployments, edge devices |\n| 1536 | ~6 KB | 0.706 | 98.6% | Balanced production systems |\n| 2560 (full) | ~10 KB | 0.716 | 100.0% | Maximum accuracy requirements |\n\n**Key Insight**: The 768-dimensional variant retains 95% of full performance while reducing storage by 70%, making large-scale deployment economically viable.\n\n\n## What's Next: From Retrieval to Understanding\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) solves the critical first step-finding the right documents across languages. But this is just the beginning of our vision for multilingual document intelligence.\n\nAs part of our **[Nayana initiative](https://cognitivelab.in/nayana)**, we're developing specialized vision-language models that don't just retrieve documents-they understand and answer questions about them. Imagine uploading a financial report in Japanese and asking detailed questions in English: \"What was the revenue growth in Q3?\" or \"Which product line had the highest margin?\" These models will process the document image directly and provide accurate answers, preserving all visual context from charts and tables.\n\nThe Nayana family of models represents the next frontier in multilingual, multimodal document intelligence. Where [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) finds the right document, Nayana models will extract insights from it-across any of the 22 supported languages, understanding both text and visual elements like charts, diagrams, and complex layouts. This creates a complete pipeline: retrieve with [NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed), then understand with Nayana.\n\nWe're also expanding language coverage beyond the current 22, with particular focus on truly low-resource languages where the need is greatest. Additionally, we're developing specialized fine-tuned variants for specific industries-legal document search, medical records retrieval, and academic research applications. Each domain has unique requirements, and we're building solutions that address them while maintaining the multilingual capabilities that make our models unique.\n\nFor enterprises interested in early access to Nayana models, custom deployments, or specialized language support, we offer consultation and collaboration opportunities. Reach out to our team at **[[email protected]](mailto:[email protected])** to discu
1ss your specific requirements.\n\n## Join the Community\n\n[NetraEmbed](https://huggingface.co/Cognitive-Lab/NetraEmbed) is built on open research principles. We're releasing our models, the **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)**, and training methodology to enable the research community to build on this foundation. If you're working on multilingual AI, document understanding, or related challenges, we'd love to collaborate.\n\nExplore the **[models on HuggingFace](https://huggingface.co/Cognitive-Lab)**, read the **[research paper](https://arxiv.org/abs/2512.03514)**, and reach out if you have questions or feedback at **[[email protected]](mailto:[email protected])**. Together, we can make document intelligence truly universal.\n\n## Acknowledgments\n\nThis work benefited from compute credits for training, inference, and evaluation provided by **[Modal](https://modal.com)**, acknowledged as a compute sponsor. Dataset curation and synthesis were supported by the **[Meta Llama Impact Grant](https://about.fb.com/news/2025/04/llama-impact-grant-recipients/)** through our **[Nayana initiative](https://cognitivelab.in/nayana)**. We appreciate Meta for continued support of our research efforts at **[CognitiveLab](https://cognitivelab.in)**.\n\n---\n\n**Quick Links:**\n\n- **[NetraEmbed Model on HuggingFace](https://huggingface.co/Cognitive-Lab/NetraEmbed)** - Download and integrate the single-vector model\n- **[ColNetraEmbed Model on HuggingFace](https://huggingface.co/Cognitive-Lab/ColNetraEmbed)** - Multi-vector variant for fine-grained retrieval\n- **[NayanaIR Benchmark](https://huggingface.co/collections/Cognitive-Lab/nayanair-bench)** - Evaluation datasets across 22 languages\n- **[Nayana Initiative](https://cognitivelab.in/nayana)** - Next-generation document understanding models\n- **[Research Paper on arXiv](https://arxiv.org/abs/2512.03514)** - M3DR: Towards Universal Multilingual Multimodal Document Retrieval\n- **[CognitiveLab](https://cognitivelab.in)** - Explore our other AI research and products\n- **[Contact Us](mailto:[email protected])** - Questions, collaborations, or enterprise inquiries\n"])</script>
1<script>self.__next_f.push([1,"11:{\"title\":\"Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval\",\"excerpt\":\"We're excited to announce NetraEmbed and ColNetraEmbed, achieving 152% improvement over existing systems in multilingual document retrieval. Supporting 22 languages with state-of-the-art performance.\",\"coverImage\":\"/assets/blog/introducing-netraembed/cover.png\",\"date\":\"2025-12-08T10:00:00.000Z\",\"contentType\":\"announcement\",\"author\":\"$12\",\"ogImage\":\"$13\",\"tags\":\"$14\",\"slug\":\"introducing-netraembed\",\"content\":\"$15\"}\n17:{\"name\":\"CognitiveLab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"}\n18:{\"url\":\"/assets/blog/introducing-nayana/cover.png\"}\n19:T1aad,"])</script>
1<script>self.__next_f.push([1,"\n# CognitiveLab Wins Meta's Prestigious Llama Impact Grant 2024\n\nWe are thrilled to announce that **CognitiveLab has been selected as a recipient of Meta's prestigious Llama Impact Grant 2024!** This significant recognition comes with substantial non-dilutive funding that will accelerate our mission to democratize AI across global languages.\n\n## About the Llama Impact Grant\n\nThe Llama Impact Grant is Meta's initiative to support innovative projects leveraging AI to drive positive societal impact. Selected from thousands of global applicants, CognitiveLab stood out for our transformative vision of building truly inclusive AI that serves diverse linguistic communities.\n\nThis grant provides:\n\n- Substantial financial support to accelerate our R\u0026D\n- Increased visibility in the global AI ecosystem\n\n## Powering Project Nayana: Our Vision for Inclusive AI\n\nThe Llama Impact Grant will primarily fuel our flagship initiative: **Project Nayana** (meaning \"eyes\" in Hindi)âa revolutionary, unified foundation model ecosystem designed to see, understand, and generate content across languages and modalities.\n\n### The AI Language Gap We're Solving\n\nToday's cutting-edge AI remains largely English-centric, creating a technological divide that excludes billions from fully participating in the AI revolution. This gap is particularly pronounced across India and the Global South, where linguistic diversity is rich but AI support is limited.\n\nThe consequences are significant:\n\n- Limited access to advanced AI tools for non-English speakers\n- Barriers to knowledge discovery in regional languages\n- Uneven distribution of AI benefits across global communities\n- Cultural and linguistic nuances lost in translation\n\n### Introducing Nayana: Breaking Down AI's Language Barriers\n\nProject Nayana represents our comprehensive solution to these challengesâa unified AI ecosystem with deep multilingual (22 languages, including 10 Indic), multimodal (vision + text), and multitask intelligence capabilities.\n\n#### What Makes Nayana Revolutionary:\n\n1. **Unprecedented Language Coverage:** While most AI models focus on a handful of major languages, Nayana supports 22 diverse languages, with special emphasis on 10 Indic languages that serve over 1 billion people.\n\n2. **Our Technical Breakthroughs:**\n\n - **Novel Synthetic Data Engine:** We've developed a groundbreaking pipeline that generates millions of high-fidelity, layout-preserving, annotated document images across all 22 target languages\n - **Unified Architecture:** Instead of fragmented pipelines, Nayana uses a single, powerful architecture for multiple tasks\n - **Cultural Context Preservation:** Our models understand cultural nuances and context-specific language use\n\n3. **Current Research Focus Areas:**\n - **Layout-Preserving Translation:** Maintaining document structure across languages\n - **Cross-Script Transfer Learning:** Leveraging patterns across different writing systems\n - **Efficient Multimodal Fusion:** Integrating text, vision, and potentially audio in resource-efficient ways\n - **Low-Resource Language Optimization:** Specialized techniques for languages with limited data\n\n#### The Nayana Ecosystem: Building Blocks of Inclusive AI\n\nNayana isn't just one model; it's a comprehensive ecosystem with multiple components:\n\n1. **Model Family (Powered by Llama Insights):**\n\n - **Nayana OCR:** Our award-winning OCR technology delivers unparalleled accuracy for 22 languages\n - **Nayana-2B:** Compact multilingual, multimodal model optimized for edge deployment\n - **Nayana-7B:** Our flagship model with advanced reasoning capabilities across all supported languages\n - **Nayana Retriever:** Specialized models for efficient cross-modal, multilingual information retrieval\n\n2. **Supporting Infrastructure:**\n - **NayanaBench:** Comprehensive evaluation framework for rigorous multilingual, multimodal testing\n - **ViViD Framework:** Our unified architecture handling multiple document intelligence tasks\n - **Open-Source Tooling:** Developer-friendly libraries and APIs for c
1ommunity adoption\n\n## Our Roadmap: What's Next for Nayana\n\nWith the Llama Impact Grant accelerating our progress, here's what we're working on:\n\n### Immediate Focus (Next 6 Months):\n\n- Completing our pre-training pipeline for all 22 languages\n- Releasing Nayana OCR open-source with comprehensive documentation\n- Publishing our synthetic data generation methodology\n- Establishing partnerships with key educational and governmental organizations\n\n### Medium-Term Goals (6-12 Months):\n\n- Launching Nayana-2B and Nayana-7B general-purpose models\n- Developing specialized fine-tuned versions for key sectors (education, healthcare, governance)\n- Creating an accessible API for developers to integrate Nayana capabilities\n- Expanding our benchmark suite to cover more languages and tasks\n\n### Long-Term Vision:\n\n- Extending language coverage to 50+ global languages\n- Integrating speech recognition and generation capabilities\n- Building specialized domain models for critical sectors\n- Establishing a sustainable open-source community around the project\n\n## Real-World Impact: How Nayana Will Transform Lives\n\nThe Llama Impact Grant significantly boosts our ability to deliver transformative solutions across sectors:\n\n- **Preserving Cultural Heritage:** Digitizing and making accessible millions of historical documents and manuscripts in regional languages\n- **Revolutionizing Education:** Creating AI tutors and translation tools that understand cultural context and regional educational needs\n- **Transforming Governance:** Making public services truly accessible in citizens' native languages\n- **Improving Healthcare:** Breaking down communication barriers in medical settings and enabling multilingual health records\n- **Driving Economic Opportunity:** Empowering local businesses with AI tools that understand regional markets and languages\n\n## Join Our Journey\n\nThe recognition from Meta's Llama Impact Grant marks a pivotal moment for CognitiveLab and for inclusive AI globally. We invite researchers, developers, organizations, and language enthusiasts to join us in building an AI future where language is never a barrier.\n\n**Get involved with Nayana:**\n\n- **[Explore the full Nayana ecosystem on our dedicated page](/nayana)**\n- **[Read Meta's Llama Impact Grant announcement](https://about.fb.com/news/2025/04/llama-impact-grant-recipients/?utm_source=AIatMeta\u0026utm_medium=organic_social\u0026utm_content=image\u0026utm_campaign=llamacon)**\n- **[Check out our models and research on Hugging Face](https://huggingface.co/Nayana-cognitivelab)**\n\nFor enterprise inquiries, research collaborations, and partnership opportunities, please contact us at **[email protected]**.\n\nLet's build an AI that truly speaks everyone's language.\n"])</script>
1<script>self.__next_f.push([1,"16:{\"title\":\"CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana\",\"excerpt\":\"CognitiveLab is proud to be selected as a recipient of Meta's prestigious Llama Impact Grant 2024, accelerating our revolutionary Nayana projectâa multilingual (22 languages, 10 Indic), multimodal AI ecosystem that democratizes AI across global languages.\",\"coverImage\":\"/assets/blog/introducing-nayana/cover.png\",\"date\":\"2025-04-30T10:00:00.000Z\",\"contentType\":\"announcement\",\"author\":\"$17\",\"ogImage\":\"$18\",\"slug\":\"introducing-nayana\",\"content\":\"$19\"}\n1b:{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"}\n1c:{\"url\":\"/assets/blog/introducing-ai-engineering-academy/cover.png\"}\n1d:Tea6,"])</script>
1<script>self.__next_f.push([1,"\n# Introducing AI Engineering Academy\n\n\u003c!--\n\u003cdiv align=\"center\"\u003e\n \u003ch1 \u003e\u003ca href=\"https://aiengineering.academy/\" target=\"_blank\"\u003eAIEngineering.academy\u003c/a\u003e\u003c/h1\u003e\n \u003cp\u003eNavigating the World of AI, One Step at a Time\u003c/p\u003e\n \u003cp\u003eInitiative from \u003ca href=\"https://cognitivelab.in/\" target=\"_blank\"\u003eCognitiveLab\u003c/a\u003e\u003c/p\u003e\n\u003c/div\u003e\n\n\u003cimg src=\"./assets/banner.png\" alt=\"Ai Engineering. Academy\"\u003e\n\n[](https://github.com/adithya-s-k/AI-Engineering.academy/stargazers)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/network/members)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/issues)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/pulls)\n[](https://github.com/adithya-s-k/AI-Engineering.academy/blob/main/LICENSE) --\u003e\n\nAI-related careers are becoming increasingly sought-after. However, the abundance of learning resources scattered across the internet can lead to confusion about where to start.\u0026#x20;\n\n**AI Engineering Academy** aims to provide a structured learning path to help you learn Applied GenAI effectively.\n\n## Roadmap\n\n### [**1. Prompt Engineering**](PromptEngineering/)\n\nThis roadmap will cover the basics of prompt engineering and its role in various AI applications.\n\n### [**2. Retrieval Augmented Generation (RAG)**](RAG/)\n\nIf you've been in the AI space, you might have heard of RAG (Retrieval Augmented Generation). In this roadmap, we will cover:\n\n- Understanding what RAG is\n- Implementing RAG from scratch without using any frameworks\n- Choosing the best RAG system for your needs\n- Taking a RAG system to production\n\n### [**3. Fine-tuning**](Finetuning/)\n\n**(coming soon)**\n\nWe will debunk some myths about fine-tuning and explore where it can be effectively used. Fine-tuning can be a powerful tool, but it's often misunderstood. Expect to learn:\n\n- The true potential of fine-tuning\n- Proper techniques for fine-tuning models\n\n### [**4. Deployment**](Deployment/)\n\n**(coming soon)**\n\nWhile everything might work perfectly locally, putting models into the hands of users requires careful consideration. In this roadmap, we will cover:\n\n- Deploying ML and DL models into production\n- Working with different cloud providers\n- Exploring various deployment modes\n\n### [**5. AI Agents**](Agents/)\n\n**(coming soon)**\n\n### [**6. Projects**](Projects/)\n\n**(coming soon)**\n\nConsists of multiple hands-on end-to-end AI projects!!\n\n## Repo Admin ð¨âð¼\n\n\u003cdiv align=\"center\"\u003e\n \u003ctable\u003e\n \u003ctr\u003e\n \u003ctd align=\"center\"\u003e\u003ca href=\"https://github.com/adithya-s-k\"\u003e\u003cimg src=\"https://avatars.githubusercontent.com/u/27956426?s=400\u0026u=582ecb2d706a63fc67eb1b54579c7ab19cf391fd\u0026v=4\" width=\"100px;\" alt=\"\"/\u003e\u003cbr /\u003e\u003csub\u003e\u003cb\u003eAdithya S Kolavi\u003c/b\u003e\u003c/sub\u003e\u003c/a\u003e\u003cbr /\u003e\u003ca href=\"https://github.com/adithya-s-k/AI-Engineering.academy/commits?author=adithya-s-k\" title=\"Code\"\u003eð»\u003c/a\u003e\u003c/td\u003e\n \u003c/tr\u003e\n \u003c/table\u003e\n\u003c/div\u003e\n\n## Contributors ð\n\n\u003ca href=\"https://github.com/adithya-s-k/AI-Engineering.academy/graphs/contributors\"\u003e\n \u003cimg src=\"https://contrib.rocks/image?repo=adithya-s-k/AI-Engineering.academy\" /\u003e\n\nThanks to these wonderful people for their contributions!\n\u003c/a\u003e\n\n\u003cp align=\"center\"\u003e\n \u003ca href=\"https://adithyask.com\"\u003e\n \u003cimg src=\"https://api.star-history.com/svg?repos=adithya-s-k/AI-Engineering.academy\u0026type=Date\" alt=\"Star History Chart\"\u003e\n \u003c/a\u003e\n\u003c/p\u003e\n"])</script>
1<script>self.__next_f.push([1,"1a:{\"title\":\"Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers\",\"excerpt\":\"AI Engineering Academy offers comprehensive training programs and resources to help aspiring engineers master the technical skills and practical knowledge needed to build, deploy, and maintain AI systems at scale.\",\"coverImage\":\"/assets/blog/introducing-ai-engineering-academy/cover.png\",\"date\":\"2024-12-11T12:00:00.000Z\",\"contentType\":\"blog\",\"author\":\"$1b\",\"ogImage\":\"$1c\",\"slug\":\"introducing-ai-engineering-academy\",\"content\":\"$1d\"}\n1f:{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"}\n20:{\"url\":\"/assets/blog/introducing-omniparse/cover.png\"}\n21:T306d,"])</script>
1<script>self.__next_f.push([1,"\n# Introducing Omniparse: Universal Document Data Extraction\n\n\n[](https://github.com/adithya-s-k/omniparse/stargazers)\n[](https://github.com/adithya-s-k/omniparse/network/members)\n[](https://github.com/adithya-s-k/omniparse/issues)\n[](https://github.com/adithya-s-k/omniparse/pulls)\n[](https://github.com/adithya-s-k/omniparse/blob/main/LICENSE)\n\n\u003e [!IMPORTANT]\n\u003e\n\u003e OmniParse is a platform that ingests and parses any unstructured data into structured, actionable data optimized for GenAI (LLM) applications. Whether you are working with documents, tables, images, videos, audio files, or web pages, OmniParse prepares your data to be clean, structured, and ready for AI applications such as RAG, fine-tuning, and more\n\n## Try it out\n\n[](https://colab.research.google.com/github/adithya-s-k/omniparse/blob/main/examples/OmniParse_GoogleColab.ipynb)\n\n## Intro\n\nhttps://github.com/adithya-s-k/omniparse/assets/27956426/457d8b5b-9573-44da-8bcf-616000651a13\n\n## Features\n\nâ
Completely local, no external APIs \\\nâ
Fits in a T4 GPU \\\nâ
Supports ~20 file types \\\nâ
Convert documents, multimedia, and web pages to high-quality structured markdown \\\nâ
Table extraction, image extraction/captioning, audio/video transcription, web page crawling \\\nâ
Easily deployable using Docker and Skypilot \\\nâ
Colab friendly \\\nâ
Interative UI powered by Gradio\n\n### Why OmniParse ?\n\nIt's challenging to process data as it comes in different shapes and sizes. OmniParse aims to be an ingestion/parsing platform where you can ingest any type of data, such as documents, images, audio, video, and web content, and get the most structured and actionable output that is GenAI (LLM) friendly.\n\n## Installation\n\n\u003e [!IMPORTANT]\n\u003e The server only works on Linux-based systems. This is due to certain dependencies and system-specific configurations that are not compatible with Windows or macOS.\n\n```bash\ngit clone https://github.com/adithya-s-k/omniparse\ncd omniparse\n```\n\nCreate a Virtual Environment:\n\n```bash\nconda create -n omniparse-venv python=3.10\nconda activate omniparse-venv\n```\n\nInstall Dependencies:\n\n```bash\npoetry install\n# or\npip install -e .\n# or\npip install -r pyproject.toml\n```\n\n### ð³ï¸ Docker\n\nTo use OmniParse with Docker, execute the following commands:\n\n1. Pull the OmniParse API Docker image from Docker Hub:\n2. Run the Docker container, exposing port 8000:\n ðð¼[Docker Image](https://hub.docker.com/r/savatar101/omniparse)\n\n```bash\ndocker pull savatar101/omniparse:0.1\n# if you are running on a gpu\ndocker run --gpus all -p 8000:8000 savatar101/omniparse:0.1\n# else\ndocker run -p 8000:8000 savatar101/omniparse:0.1\n```\n\nAlternatively, if you prefer to build the Docker image locally:\nThen, run the Docker container as follows:\n\n```bash\ndocker build -t omniparse .\n# if you are running on a gpu\ndocker run --gpus all -p 8000:8000 omniparse\n# else\ndocker run -p 8000:8000 omniparse\n\n```\n\n## Usage\n\nRun the Server:\n\n```bash\npython server.py --host 0.0.0.0 --p
1ort 8000 --documents --media --web\n```\n\n- `--documents`: Load in all the models that help you parse and ingest documents (Surya OCR series of models and Florence-2).\n- `--media`: Load in Whisper model to transcribe audio and video files.\n- `--web`: Set up selenium crawler.\n\nDownload Models:\nIf you want to download the models before starting the server\n\n```bash\npython download.py --documents --media --web\n```\n\n- `--documents`: Load in all the models that help you parse and ingest documents (Surya OCR series of models and Florence-2).\n- `--media`: Load in Whisper model to transcribe audio and video files.\n- `--web`: Set up selenium crawler.\n\n## Supported Data Types\n\n| Type | Supported Extensions |\n| --------- | --------------------------------------- |\n| Documents | .doc, .docx, .pdf, .ppt, .pptx |\n| Images | .png, .jpg, .jpeg, .tiff, .bmp, .heic |\n| Video | .mp4, .mkv, .avi, .mov |\n| Audio | .mp3, .wav, .aac |\n| Web | dynamic webpages, http://\u003canything\u003e.com |\n\n\u003cdetails\u003e\n\u003csummary\u003e\u003ch2\u003eAPI Endpoints\u003c/h2\u003e\u003c/summary\u003e\n\n\u003e Client library compatible with Langchain, llamaindex, and haystack integrations coming soon.\n\n- [OmniParse](#omniparse)\n - [Try it out](#try-it-out)\n - [Intro](#intro)\n - [Features](#features)\n - [Why OmniParse ?](#why-omniparse-)\n - [Installation](#installation)\n - [ð³ï¸ Docker](#ï¸-docker)\n - [Usage](#usage)\n - [Supported Data Types](#supported-data-types)\n - [Document Parsing](#document-parsing)\n - [Parse Any Document](#parse-any-document)\n - [Parse PDF](#parse-pdf)\n - [Parse PowerPoint](#parse-powerpoint)\n - [Parse Word Document](#parse-word-document)\n - [Media Parsing](#media-parsing)\n - [Parse Image](#parse-image)\n - [Process Image](#process-image)\n - [Parse Video](#parse-video)\n - [Parse Audio](#parse-audio)\n - [Website Parsing](#website-parsing)\n - [Parse Website](#parse-website)\n - [Coming Soon/ RoadMap](#coming-soon-roadmap)\n - [Limitations](#limitations)\n - [License](#license)\n - [Commercial Usage](#commercial-usage)\n - [Acknowledgements](#acknowledgements)\n - [Contact](#contact)\n\n### Document Parsing\n\n#### Parse Any Document\n\nEndpoint: `/parse_document`\nMethod: POST\n\nParses PDF, PowerPoint, or Word documents.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/document\" http://localhost:8000/parse_document\n```\n\n#### Parse PDF\n\nEndpoint: `/parse_document/pdf`\nMethod: POST\n\nParses PDF documents.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/document.pdf\" http://localhost:8000/parse_document/pdf\n```\n\n#### Parse PowerPoint\n\nEndpoint: `/parse_document/ppt`\nMethod: POST\n\nParses PowerPoint presentations.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/presentation.ppt\" http://localhost:8000/parse_document/ppt\n```\n\n#### Parse Word Document\n\nEndpoint: `/parse_document/docs`\nMethod: POST\n\nParses Word documents.\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/document.docx\" http://localhost:8000/parse_document/docs\n```\n\n### Media Parsing\n\n\u003c!-- #### Parse Any Media\n\nEndpoint: `/parse_media`\nMethod: POST\n\nParses images, videos, or audio files.\n\nCurl command:\n```\ncurl -X POST -F \"file=@/path/to/media_file\" http://localhost:8000/parse_media\n``` --\u003e\n\n#### Parse Image\n\nEndpoint: `/parse_image/image`\nMethod: POST\n\nParses image files (PNG, JPEG, JPG, TIFF, WEBP).\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/image.jpg\" http://localhost:8000/parse_media/image\n```\n\n#### Process Image\n\nEndpoint: `/parse_image/process_image`\nMethod: POST\n\nProcesses an image with a specific task.\n\nPossible task inputs:\n`OCR | OCR with Region | Caption | Detailed Caption | More Detailed Caption | Object Detection | Dense Region Caption | Region Proposal`\n\nCurl command:\n\n```\ncurl -X POST -F \"image=@/path/to/image.jpg\" -F \"task=Caption\" -F \"prompt=Optional prompt\" http://localhost:8000/parse_media/process_image\n```\n\nArguments:\n\n- `image`: The image file\n- `task`: The processing task (e.g., Caption, Object Detection)\n- `prompt`: Optional prompt for certain tasks\n\n#### Parse Video\n\nEndpoint: `/parse_media/video`\nMethod: POST\n\nParses video files (MP4, AVI, MOV, MKV).\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/video.mp4\" http://localhost:8000/parse_media/video\n```\n\n#### Parse Audio\n\nEndpoint: `/parse_media/audio`\nMethod: POST\n\nParses audio files (MP3, WAV, FLAC).\n\nCurl command:\n\n```\ncurl -X POST -F \"file=@/path/to/audio.mp3\" http://localhost:8000/parse_media/audio\n```\n\n### Website Parsing\n\n#### Parse Website\n\nEndpoint: `/parse_website/parse`\nMethod: POST\n\nParses a website given its URL.\n\nCurl command:\n\n```\ncurl -X POST -H \"Content-Type: application/json\" -d '{\"url\": \"https://example.com\"}' http://localhost:8000/parse_website\n```\n\nArguments:\n\n- `url`: The URL of the website to parse\n\n\u003c/details\u003e\n\n## Coming Soon/ RoadMap\n\nð¦ LlamaIndex | Langchain | Haystack integrations coming soon\nð Batch processing data\nâ Dynamic chunking and structured data extraction based on specified Schema \nð ï¸ One magic API: just feed in your file prompt what you want, and we will take care of the rest \nð§ Dynamic model selection and support for external APIs \nð Batch processing for handling multiple files at once \nð¦ New open-source model to replace Surya OCR and Marker\n\n**Final goal**: replace all the different models currently being used with a single MultiModel Model to parse any type of data and get the data you need.\n\n## Limitations\n\nThere is a need for a GPU with 8~10 GB minimum VRAM as we are using deep learning models.\n\\\n\nDocument Parsing Limitations\n\\\n\n- [Marker](https://github.com/VikParuchuri/marker) which is the underlying PDF parser will not convert 100% of equations to LaTe
1X because it has to detect and then convert them.\n- It is good at parsing english but might struggle for languages such as Chinese\n- Tables are not always formatted 100% correctly; text can be in the wrong column.\n- Whitespace and indentations are not always respected.\n- Not all lines/spans will be joined properly.\n- This works best on digital PDFs that won't require a lot of OCR. It's optimized for speed, and limited OCR is used to fix errors.\n- To fit all the models in the GPU, we are using the smallest variants, which might not offer the best-in-class performance.\n\n## License\n\nOmniParse is licensed under the GPL-3.0 license. See `LICENSE` for more information.\nThe project uses Marker under the hood, which has a commercial license that needs to be followed. Here are the details:\n\n### Commercial Usage\n\nMarker and Surya OCR Models are designed to be as widely accessible as possible while still funding development and training costs. Research and personal usage are always allowed, but there are some restrictions on commercial usage.\nThe weights for the models are licensed under cc-by-nc-sa-4.0. However, this restriction is waived for any organization with less than $5M USD in gross revenue in the most recent 12-month period AND less than $5M in lifetime VC/angel funding raised. To remove the GPL license requirements (dual-license) and/or use the weights commercially over the revenue limit, check out the options provided.\nPlease refer to [Marker](https://github.com/VikParuchuri/marker) for more Information about the License of the Model weights\n\n## Acknowledgements\n\nThis project builds upon the remarkable [Marker](https://github.com/VikParuchuri/marker) project created by [Vik Paruchuri](https://twitter.com/VikParuchuri). We express our gratitude for the inspiration and foundation provided by this project. Special thanks to [Surya-OCR](https://github.com/VikParuchuri/surya) and [Texify](https://github.com/VikParuchuri/texify) for the OCR models extensively used in this project, and to [Crawl4AI](https://github.com/unclecode/crawl4ai) for their contributions.\n\nModels being used:\n\n- Surya OCR, Detect, Layout, Order, and Texify\n- Florence-2 base\n- Whisper Small\n\nThank you to the authors for their contributions to these models.\n\n---\n\n## Contact\n\n\u003cp align=\"center\"\u003e\n \u003ca href=\"https://adithyask.com\"\u003e\n \u003cimg src=\"https://api.star-history.com/svg?repos=adithya-s-k/omniparse\u0026type=Date\" alt=\"Star History Chart\"\u003e\n \u003c/a\u003e\n\u003c/p\u003e\nFor any inquiries, please contact us at [email protected]\n\n\u003c!--\nInstall the client:\n\n```bash\npip install omniparse_client\n```\n\nExample usage:\n\n```python\nfrom omniparse_client import OmniParse\n\n# Initialize the parser\nparser = OmniParse(\n base_url=\"http://localhost:8000\",\n api_key=\"op-...\", # get the API key from dev.omniparse.com\n verbose=True,\n language=\"en\"\n)\n\n# Parse a document\ndocument = parser.load_data('path/to/document.pdf')\n\n# Convert to markdown\nparser.save_to_markdown(document)\n```\n --\u003e\n"])</script>
1<script>self.__next_f.push([1,"1e:{\"title\":\"Introducing Omniparse: Universal Data Parsing\",\"excerpt\":\"Omniparse is an advanced document parsing platform that uses AI to extract structured data from any document format, enabling businesses to automate document processing workflows with unprecedented accuracy and efficiency.\",\"coverImage\":\"/assets/blog/introducing-omniparse/cover.png\",\"date\":\"2024-06-27T12:00:00.000Z\",\"contentType\":\"blog\",\"author\":\"$1f\",\"ogImage\":\"$20\",\"slug\":\"introducing-omniparse\",\"content\":\"$21\"}\n22:T3084,"])</script>
1<script>self.__next_f.push([1,"\n# Introducing the Indic LLM Leaderboard\n\n## Introduction\n\nIn the ever-evolving landscape of artificial intelligence (AI) and natural language processing (NLP), the focus on Indic languages, spoken by over a billion people in South Asia, is gaining momentum. Despite their vast user base and cultural significance, Indic languages face numerous challenges, ranging from data scarcity to the lack of sophisticated language tools. In this blog post, we'll delve into Indic Eval, a nimble evaluation suite crafted for appraising Indic LLMs across various tasks. Furthermore, we'll look into a leaderboard equipped with specialized benchmarks to assess and compare the performance of Indic Eval systems. This platform offers a holistic approach to evaluating model efficacy in the Indic language modeling sphere.\n\nâ\n\n## **Why an Indic LLM Leaderboard is Required ?**\n\nRecent advancements in Indic Large Language Models (LLMs) underscore progress, yet the absence of a unified evaluation framework complicates tracking and comparison. This challenge exacerbates existing issues like data scarcity and inadequate language tools. A unified evaluation framework with benchmarks is crucial to overcome these challenges and drive meaningful advancement in the field.\n\n### **About CognitiveLab:**\n\nFounded by [Adithya S K](https://linktr.ee/adithyaskolavi), CognitiveLab specializes in providing AI solutions at scale and undertaking research-based tasks. This initiative aims to create a unified platform where Indic LLMs can be compared using specially crafted datasets. Originally developed for internal use, CognitiveLab is now open-sourcing this framework to further aid the Indic LLM ecosystem.\n\nAfter the release of [Amabri, a 7b parameter English-Kannada bilingual LLM](https://www.cognitivelab.in/blog/introducing-indic-llm-leaderboard), CognitiveLab sought to compare it with other open-source LLMs to identify areas for improvement. Thus, the Indic LLM suite was born, consisting of two projects:\n\n## Solution\n\n- [Indic-Eval](https://github.com/adithya-s-k/indic_eval): A lightweight evaluation suite tailored specifically for assessing Indic LLMs across a diverse range of tasks, aiding in performance assessment and comparison within the Indian language context.\n- [Indic LLM Leaderboard](https://huggingface.co/spaces/Cognitive-Lab/indic_llm_leaderboard): Utilizes the [indic_eval](https://github.com/adithya-s-k/indic_eval) evaluation framework, incorporating state-of-the-art translated benchmarks like ARC, Hellaswag, MMLU, among others. Supporting seven Indic languages, it offers a comprehensive platform for assessing model performance and comparing results within the Indic language modeling landscape.\n\nâ\n\n### What comes with the alpha release\n\nThe alpha release of the **Indic LLM Leaderboard** and **Indic Eval**. These tools are pivotal steps towards standardizing evaluations within the field.\n\nThe Indic LLM Leaderboard is an evolving platform, aiming to streamline evaluations for Language Model (LLM) models tailored to Indic languages. While this **alpha release is far from perfect**, it signifies a crucial initial step towards establishing evaluation standards within the community.\n\n### **Features:**\n\nAs of this release, the following base models have been added into the leaderboard to use are reference:\n\n- `meta meta-llama/Llama-2-7b-hf`\n- `google/gemma-7b`\n\nTasks incorporated into the platform:\n\n- `ARC-Easy:{language}`\n- `ARC-Challenge:{language}`\n- `Hellaswag:{language}`\n\nFor evaluation purposes, each task includes 5-shot prompting. Further experimentation will determine the most optimal balance between evaluation time and accuracy.\n\nwe are currently testing a lot more benchmarking datasets and planning to integrate with indic_eval\n\n### **Datasets:**\n\nDatasets utilized for evaluation are accessible via the following link: [Indic LLM Leaderboard Eval Suite](https://huggingface.co/collections/Cognitive-Lab/indic-llm-leaderboard-eval-suite-660ac4818695a785edee4e6f)\n\n### **Rationale for Alpha Release:**\n\nThe decision to label this release as alpha stems from the realization that extensive testing and experimentation are necessary. Key considerations include:\n\n- Selection of appropriate metrics for evaluation\n- Determination of the optimal few-shot learning parameters\n-
1Establishment of the ideal number of evaluation samples within the dataset\n\n### **Roadmap for Next Release:**\n\nAnticipate the following enhancements in the upcoming release:\n\n- Enhanced testing and accountability mechanisms\n- A refined version of the leaderboard\n- Defined benchmarks and standardized datasets\n- Bilingual evaluation support\n- Expansion of supported models\n- Implementation of more secure interaction mechanisms\n- Addition of support for additional languages\n\n## How to show up on the Leaderboard\n\nHere are the steps you will have to follows to put your model on the Indic LLM leaderboard\n\nâ\n\nClone the repo:\n\n```jsx\ngit clone \u003chttps://github.com/adithya-s-k/indic_eval\u003e\ncd indic_eval\n```\n\nCreate a virtual environment using virtualenv or conda depending on your preferences. We require Python 3.10 or above:`â`\n\n```jsx\nconda create -n indic-eval-venv python=3.10 \u0026\u0026 conda activate indic-eval-venv`âInstall the dependencies. For the default installation, you just need:â`\n```\n\n```jsx\npip install .\n```\n\nIf you want to evaluate models with frameworks like `accelerate` or `peft`, you will need to specify the optional dependencies group that fits your use case (`accelerate`, `tgi`, `optimum`, `quantization`, `adapters`, `nanotron`):\n\n```jsx\npip install .[optional1,optional2]\n```\n\nThe setup tested most is:\n\n```jsx\npip install .[accelerate,quantization,adapters]\n```\n\nâ\nIf you want to push your results to the Hugging Face Hub, don't forget to add your access token to the environment variable `HUGGING_FACE_HUB_TOKEN`. You can do this by running:\n\n```jsx\nhuggingface-cli login\n```\n\n## Command to Run Indic Eval and Push to Indic LLM Leaderboard\n\n```bash\naccelerate launch run_indic_evals_accelerate.py \\\n --model_args=\"pretrained=\u003cpath to model on the hub\u003e\" \\\n --language kannada \\\n --tasks indic_llm_leaderboard \\\n --output_dir output_dir \\\n --push_to_leaderboard \[email protected]\u003e \\\n```\n\nIt's as simple as that.ð\n\nFor `--push_to_leaderboard`, provide an email id through which we can contact you in case of verification. This email won't be shared anywhere. It's only required for future verification of the model's scores and for authenticity.\n\nAfter you have installed all the required packages, run the following command:\n\nFor multi-GPU configuration, please refer to the docs of [Indic_Eval](https://github.com/adithya-s-k/indic_eval).\n\n## Some common questions\n\n### **What is the minimum `requirement` for GPUs to run the evaluation?**\n\n- The evaluation can easily run on a single A100 GPU, but the framework also supports multi-GPU based evaluation to speed up the process.\n\n### **What languages are supported by the evaluation framework?**\n\n- The following languages are supported by default: `english`, `kannada`, `hindi`, `tamil`, `telugu`, `gujarati`, `marathi`, `malayalam`\n\n### **How can I put my model on the leaderboard?**\n\n- Please follow the steps shown in the Submit tab or refer to the indic_eval for more details.\n\n### **How does the leaderboard work?**\n\n- After running indic_eval on the model of your choice, the results are pushed to a server and stored in a database. The Frontend Leaderboard accesses the server and retrieves the latest models in the database along with their respective benchmarks and metadata. The entire system is deployed in India and is as secure as possible.\n\n### **How is it different from the Open LLM leaderboard?**\n\n- This project was mainly inspired by the Open LLM leaderboard. However, due to limited computation resources, we standardized the evaluation library with standard benchmarks. You can run the evaluation on your GPUs and the leaderboard will serve as a unified platform to compare models. We used indictrans2 and other translation APIs to translate the benchmarking dataset into seven Indian languages to ensure reliability and consistency in the output.\n\n### **Why does it take so much time to load the results?**\n\n- We are running the server on a serverless instance which has a cold start problem, so it might sometimes take a while.\n\n### **What benchmarks are offered?**\n\n- The current Indic Benchmarks offered by the indic_eval library can be found in this collection: https://huggingface.co/collections/Cognitive-Lab/indic-llm-leaderboard-eval-suite-660ac4818695a785edee4e6f. They include ARC Easy, ARC Challenge, Hellaswag, Boolq, and MMLU.\n\n### **How much time does it take to run the evaluation using indic_eval?**\n\n- Depen
1ding on which GPU you are running, the time for evaluation varies.\n- From our testing, it takes 5 to 10 hours to run the whole evaluation on a single GPU depends on the GPU\n- It's much faster when using multiple GPUs.\n\n### **How does the verification step happen?**\n\nWhile running the evaluation, you are given an option to push results to the leaderboard with\n\n- `-push_to_leaderboard \[email protected]\u003e`. You will need to provide an email address through which we can contact you. If we find any anomaly in the evaluation score, we will contact you through this email for verification of results.\n\n## **Use Cases:**\n\n- **Research and Development**: Researchers and developers can utilize the Indic LLM Framework to adapt pre-trained LLMs to specific domains and languages, facilitating the development of customized language models tailored to diverse applications such as sentiment analysis, machine translation, and text generation.\n- **Model Evaluation and Comparison**: The Indic Eval Suite enables researchers to evaluate the performance of Indic LLMs across various tasks and benchmarks, allowing for comprehensive performance assessment and comparison. This aids in identifying strengths and weaknesses of different models and guiding future research directions.\n- **Benchmarking and Progress Tracking**: The Indic LLM Leaderboard provides a centralized platform for benchmarking the performance of Indic LLMs against standardized benchmarks and tracking progress over time. This fosters transparency, collaboration, and healthy competition within the research community, driving continuous improvement in Indic language modeling\n\n## **Call for Collaborative Effort:**\n\nTo foster collaboration and discussion surrounding evaluations, a [WhatsApp group](https://chat.whatsapp.com/CUb6eS50lX2JHX2D4j13d1) is being established and we can also connect on Hugging faces discord [indic_llm channel](https://discord.com/channels/879548962464493619/1189605147068858408)\n\n**â**\n\n## **Contribute**\n\nAll the projects are completely open source with different licenses, so anyone can contribute.\n\nThe current leaderboard is in alpha release, and many more changes are forthcoming:\n\n- More robust benchmarks tailored for Indic languages.\n- Easier integration with [indic_eval](https://github.com/adithya-s-k/indic_eval).\n\nâ\n\n## Conclusion\n\nThe alpha release of the Indic LLM Leaderboard and Indic Eval marks an important first step towards establishing standardized evaluation frameworks for Indic language models. However, it is clear that significant work remains to make these tools truly robust and accountable.\n\nThe decision to label this as an alpha release underscores the need for extensive testing, refinement of evaluation metrics, and determination of optimal setup parameters. The path ahead requires close collaboration and active participation from the open-source community.\n\nWe call upon researchers, developers, and language enthusiasts to join us in this endeavor. By pooling our collective expertise and resources, we can build a comprehensive, transparent, and reliable system to assess the performance of Indic LLMs. This will not only drive progress in the field but also ensure that advancements truly benefit the diverse linguistic landscape of the Indian subcontinent.\n\nThe road ahead is long, but the potential impact is immense. With your support and involvement, we are confident that the Indic LLM Leaderboard and Indic Eval will evolve into reliable and indispensable tools for the Indic language modeling community. Together, let us embark on this journey to unlock the full potential of Indic language technologies.\n"])</script>
1<script>self.__next_f.push([1,"23:T31bb,"])</script>
1<script>self.__next_f.push([1,"\n\u003c!-- ### Introducing Ambari --\u003e\n\n\u003c!--  --\u003e\n\nIn this blog, I am thrilled to share insights into the meticulous approach we undertook to train Amabri Base ([Cognitive-Lab/Ambari-7B-Instruct-v0.1](https://huggingface.co/Cognitive-Lab/Ambari-7B-Instruct-v0.1)) and Amabri Instruct ([Cognitive-Lab/Ambari-7B-Instruct-v0.1](https://huggingface.co/Cognitive-Lab/Ambari-7B-Instruct-v0.1)). Offering a high-level glimpse into our process, this narrative serves as a precursor to the forthcoming revelation of all technical detailsâ the culmination of extensive testing and evaluation. Stay tuned as we unravel the intricacies that led to the creation of Amabri, an innovative open-source bilingual Kannada-English Large Language Model.\n\n# Why We Built Amabri\n\n**Purpose Behind Amabri**\n\nIn the dynamic landscape of Large Language Models (LLMs), the creation of Amabri stemmed from a multifaceted purpose:\n\n- **Language Adaptation of LLMs:** Our primary objective was to pioneer language adaptability within LLMs, bridging the linguistic gap between Kannada and English.\n- **Training/Finetuning on a Modest 1B-Token Dataset:** Recognizing the constraints posed by smaller datasets, we aimed to push the boundaries of efficiency by training and finetuning Amabri on a relatively compact dataset of 1 billion tokens.\n- **Identifying the Most Efficient Process:** The quest for efficiency led us to meticulously explore and determine the most effective processes at each stage of Amabri's development.\n- **Observing World Knowledge Acquisition:** Amabri was conceived as a lens through which we could observe the accrual of world knowledge throughout the training process, shedding light on its adaptability and expansive learning capabilities.\n- **Optimizing Training Methods for Each Stage:** A crucial aspect of our endeavor was to discern and optimize the training methods suited for each developmental stage of Amabri.\n\nAs LLMs increasingly permeate mainstream usage, open-source models, while enriched in world knowledge, predominantly emerge from English-centric training. Amabri serves as a pioneering initiative to broaden this scope and adapt LLMs to diverse languages.\n\n## Introduction\n\nIn the evolving landscape of LLMs, the demand for vast amounts of training data, ranging from 1 trillion to 10 trillion tokens, has become a norm. However, this poses a challenge for languages with limited documented resources. In our pursuit, we focused on the adaptation of a pre-trained LLM, such as Llama/Mistral, to comprehend the nuances of a new languageâKannada in the case of Amabri. Despite Kannada not being classified as a very low-resource language, it served as an ideal candidate to test our hypotheses and methodologies. Rigorously defining the stages of training and finetuning, we set a cap of 1 billion training tokens for the entire process.\n\nSubsequently, we meticulously crafted datasets, distributed them accordingly, and delineated the stages of our process:\n\n- **Pre-training:** 500 Million tokens\n- **Bilingual Next Token Prediction/Translation:** ~300 Million tokens\n- **Instruct Finetuning/DPO Finetuning:** ~200 Million tokens\n\nThis deliberate approach laid the foundation for Amabri's development, pushing the boundaries of language adaptability within the realm of LLMs.\n\n## Tokenization\n\nTokenization, a critical component in the efficiency of language models, posed a unique challenge for Kannada text within the context of open-source LLMs. Many existing models inefficiently resort to character-level tokenization, especially during inference, impacting overall performance. To address this, we developed a specialized tokenization model for Kannada text using SentencePiece. This model was seamlessly integrated with the base Llama tokenizer, resulting in a comprehensive vocabulary of 49,600 , expanded by 17,600 .\n\nOur approach involved training the tokenizer model on three different dataset sizes, revealing optimal results with a dataset comprising 100,000 tokens. As we evolve Amabri, the upcoming iteration will feature a refined tokenization strategy, employing a reduced vocabulary size of 48,000. This adjustment, validated by insights shared by Andrej Karpathy in his Twitter post ([Andrej Karpathy on Twitter](https://twitter.com/karpathy/status/1621578354024677377)), is geared towards enhancing overall efficiency.\n\nCurious to explore the efficiency gains firsthand? You can test out the tokenizer in action [here](https://github.com/adithya-s-k/LLM-Alchemy-Chamber/blob/main/LLMs/ambari/tokeniser.ipynb).\n\n## Continual Pre-Training\n\n**Pre-Training**\n\nWith an efficie
1nt tokenizer in place, our next crucial step was the pre-training phase, aimed at familiarizing the model with the newly enriched vocabulary. To optimize this process, we curated a comprehensive dataset from diverse sources. Notably, we explored two distinct approaches during this phaseâpre-training with Lora and fully training the model. This strategic decision stemmed from our desire to discern the optimal path for Amabri's development.\n\nA detailed comparison between these methodologies will be unveiled shortly, but we've gleaned some initial observations:\n\n- Contrary to our hypothesis, full-weight fine-tuning did not result in a significant performance decrease compared to Lora.\n- The fully fine-tuned model exhibited a remarkable increase in predicting Kannada tokens when the preceding token was Kannada, indicating enhanced language understanding.\n- Additionally, we noted a heightened robustness in the generation capabilities of the fully fine-tuned model.\n\nWhile we acknowledge that our ongoing testing may refine these observations, this snapshot provides valuable insights into our progress. The pre-training phase employed a cluster of 2xA100 GPUs, taking approximately 25 hours for full-weight pre-training on a substantial corpus comprising 500 million tokens.\n\nIt's worth mentioning that the weights of the fully fine-tuned model are now available on [Hugging Face](https://huggingface.co/Cognitive-Lab/Ambari-7B-base-v0.1)ð¤ - https://huggingface.co/Cognitive-Lab/Ambari-7B-base-v0.1, contributing to the open-source knowledge sharing within the community.\n\n## Bilingual Next Token Prediction and Translation\n\n**Bilingual Next Token Prediction**\n\nThis phase, inspired by the open Hathi series by [sarvam.ai](http://sarvam.ai/), was an unplanned yet pivotal addition to our training strategy. Creating a dataset of 200,000 tokens, we utilized Lora for fine-tuning, aiming to equip the model with enhanced language understanding. As we progressed, our focus shifted towards instilling 'world knowledge' in Kannada. Given the scarcity of Kannada content, especially compared to English, we turned to translation. Leveraging IndicTrans2, we translated English content, primarily sourced from Wikipedia, into Kannada. However, instead of conventional monolingual next token prediction, we introduced a groundbreaking approach â bilingual next token prediction. Alternating sentences between Kannada and English, this method compelled the model to cross-lingually attend to information during next-token prediction. This nuanced approach not only fostered increased alignment between Kannada and English but also naturally balanced exposure to Hindi and English tokens during training. This stage added an extra layer of sophistication to Amabri's training journey.\n\n**Translation Finetuning**\n\nThe intention behind this phase was to establish a coherent relationship between English and corresponding Kannada tokens. Employing low-rank adaptation for fine-tuning, we encountered some challenges, notably with the decision to use a very low-rank value, which proved less effective. With a dataset size of 100,000 tokens, this stage presented limitations, and we acknowledge the need for improvements. As we refine this aspect of the training process, our commitment to enhancing the bilingual capabilities of Amabri remains unwavering.\n\n## Bilingual Instruct Fine-tuning\n\n**Bilingual Instruct Fine-tuning**\n\nIn this pivotal stage, we employed supervised fine-tuning with low-rank adaptation to mold the model's responsiveness. Embracing a chat template structure consisting of user prompts/instructions and corresponding responses, we ventured into the realm of Bilingual Instruct Fine-tuning. This approach involved training the model to adeptly respond in either English or Kannada based on the language specified in the user prompt or instruction.\n\nChat Template\n\n```bash\n\u003c|user|\u003e\n{user prompt / instruction}\n\u003c|endoftext|\u003e\n\u003c|assistant|\u003e\n{response}\n\u003c|endoftext|\u003e\n```\n\nFor instance, given a user prompt like\n\n\"Give me 10 Study tips
1in Kannada,\"\n\n\u003e Response\n\nthe model seamlessly generates a response in Kannada, maintaining linguistic coherence. To enrich the training process, we amalgamated various instruction datasets, including [Alpaca Instruct](https://huggingface.co/datasets/tatsu-lab/alpaca), [Dolly Instruct](https://huggingface.co/datasets/c-s-ale/dolly-15k-instruction-alpaca-format), and more. Leveraging translation APIs such as Google, Azure, and a custom deployment of the IndicTrans2 model from [ai4bharat](https://ai4bharat.iitm.ac.in/), we crafted a comprehensive bilingual instruct dataset.\n\nThe dataset, now publicly available on Hugging Face [here](https://huggingface.co/datasets/Cognitive-Lab/Kannada-Instruct-dataset), encompasses diverse linguistic scenarios. During training, we implemented supervised fine-tuning with four distinct representations:\n\n1. Kannada Instruction â Kannada Output\n2. English Instruction â Kannada Output\n3. Kannada Instruction â English Output\n4. English Instruction â Kannada Output\n\nThis meticulous approach not only familiarized the model with responding in different languages but also laid the groundwork for mastering various cross-lingual tasks.\n\nThe weights of this finely-tuned model are accessible on Hugging Face, and for a hands-on experience, you can explore the 4-bit quantized version on [chat.cognitivelab.in](https://chat.cognitivelab.in/).\n\n## DPO Fine-tuning\n\nIn the culminating phase of our model refinement, we delved into the world of Direct Preference Optimization (DPO). This strategic choice, inspired by the success observed in various open-source models, aimed not only to align our model but also to drive improvements in benchmarks. Embarking on this experimental journey, we leveraged the [Anthropic/hh-rlhf](https://huggingface.co/datasets/Anthropic/hh-rlhf) dataset. Translating it to Kannada, we subjected the model to DPO fine-tuning, currently undergoing a comprehensive evaluation to gauge its performance impact.\n\n## Learnings and Conclusion: Navigating Challenges and Unveiling Potential\n\n**Scope of Improvement**\n\n- **Lack of World Knowledge:** The model, trained with a capped dataset of 1 billion Kannada tokens, exhibits occasional hallucinations in response to highly specific queries, signaling a scope for improvement in imparting broader world knowledge.\n- **Translation Challenges:** Notably, translation nuances surface when handling nouns like names and places. Addressing this challenge involves allocating more training data specifically for the translation phase, a key focus for future enhancements.\n- **Full Weight Fine-tuning Dilemma:** An observation surfaced regarding the model's slight overfitting to predict Kannada tokens due to full weight fine-tuning. Future iterations will strategically decide between full weight and Lora for continual pre-training based on extensive tests and evaluations.\n\n## What's Next: Evolution and Enrichment\n\n- **Addition of Romanized Kannada:** A pivotal expansion awaits with the incorporation of Romanized Kannada, enriching the model's linguistic versatility. An illustrative example: âKannada **ನನà³à²¨ ಹà³à²¸à²°à³ à²
ಮಾಬà³à²°à²¿**â (Nanna hesaru amÄbri).\n- **Continuous Learning and Model Refinement:** Building on the learnings from this model version, we commit to refining our data pipelines and fine-tuning stages. The subsequent iteration will scale the training dataset to approximately 10 to 15 billion Kannada tokens, optimizing data distribution across diverse stages for enhanced efficiency.\n\n## Acknowledgments: Standing on the Shoulders of Inspiring Projects\n\nOur journey has been shaped by inspiration drawn from impactful projects. Special mention goes to [Sarvam.ai](http://sarvam.ai/) and the illuminating project [Tamil Llama](https://github.com/abhinand5/tamil-llama) by [Abhinand Balachandran](https://www.linkedin.com/in/abhinand-05/). Their contributions have been instrumental in steering the course of our own endeavors.\n"])</script>
1<script>self.__next_f.push([1,"25:{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"}\n26:{\"url\":\"/assets/blog/introducing-indic-llm-leaderboard/cover.png\"}\n27:T3084,"])</script>
1<script>self.__next_f.push([1,"\n# Introducing the Indic LLM Leaderboard\n\n## Introduction\n\nIn the ever-evolving landscape of artificial intelligence (AI) and natural language processing (NLP), the focus on Indic languages, spoken by over a billion people in South Asia, is gaining momentum. Despite their vast user base and cultural significance, Indic languages face numerous challenges, ranging from data scarcity to the lack of sophisticated language tools. In this blog post, we'll delve into Indic Eval, a nimble evaluation suite crafted for appraising Indic LLMs across various tasks. Furthermore, we'll look into a leaderboard equipped with specialized benchmarks to assess and compare the performance of Indic Eval systems. This platform offers a holistic approach to evaluating model efficacy in the Indic language modeling sphere.\n\nâ\n\n## **Why an Indic LLM Leaderboard is Required ?**\n\nRecent advancements in Indic Large Language Models (LLMs) underscore progress, yet the absence of a unified evaluation framework complicates tracking and comparison. This challenge exacerbates existing issues like data scarcity and inadequate language tools. A unified evaluation framework with benchmarks is crucial to overcome these challenges and drive meaningful advancement in the field.\n\n### **About CognitiveLab:**\n\nFounded by [Adithya S K](https://linktr.ee/adithyaskolavi), CognitiveLab specializes in providing AI solutions at scale and undertaking research-based tasks. This initiative aims to create a unified platform where Indic LLMs can be compared using specially crafted datasets. Originally developed for internal use, CognitiveLab is now open-sourcing this framework to further aid the Indic LLM ecosystem.\n\nAfter the release of [Amabri, a 7b parameter English-Kannada bilingual LLM](https://www.cognitivelab.in/blog/introducing-indic-llm-leaderboard), CognitiveLab sought to compare it with other open-source LLMs to identify areas for improvement. Thus, the Indic LLM suite was born, consisting of two projects:\n\n## Solution\n\n- [Indic-Eval](https://github.com/adithya-s-k/indic_eval): A lightweight evaluation suite tailored specifically for assessing Indic LLMs across a diverse range of tasks, aiding in performance assessment and comparison within the Indian language context.\n- [Indic LLM Leaderboard](https://huggingface.co/spaces/Cognitive-Lab/indic_llm_leaderboard): Utilizes the [indic_eval](https://github.com/adithya-s-k/indic_eval) evaluation framework, incorporating state-of-the-art translated benchmarks like ARC, Hellaswag, MMLU, among others. Supporting seven Indic languages, it offers a comprehensive platform for assessing model performance and comparing results within the Indic language modeling landscape.\n\nâ\n\n### What comes with the alpha release\n\nThe alpha release of the **Indic LLM Leaderboard** and **Indic Eval**. These tools are pivotal steps towards standardizing evaluations within the field.\n\nThe Indic LLM Leaderboard is an evolving platform, aiming to streamline evaluations for Language Model (LLM) models tailored to Indic languages. While this **alpha release is far from perfect**, it signifies a crucial initial step towards establishing evaluation standards within the community.\n\n### **Features:**\n\nAs of this release, the following base models have been added into the leaderboard to use are reference:\n\n- `meta meta-llama/Llama-2-7b-hf`\n- `google/gemma-7b`\n\nTasks incorporated into the platform:\n\n- `ARC-Easy:{language}`\n- `ARC-Challenge:{language}`\n- `Hellaswag:{language}`\n\nFor evaluation purposes, each task includes 5-shot prompting. Further experimentation will determine the most optimal balance between evaluation time and accuracy.\n\nwe are currently testing a lot more benchmarking datasets and planning to integrate with indic_eval\n\n### **Datasets:**\n\nDatasets utilized for evaluation are accessible via the following link: [Indic LLM Leaderboard Eval Suite](https://huggingface.co/collections/Cognitive-Lab/indic-llm-leaderboard-eval-suite-660ac4818695a785edee4e6f)\n\n### **Rationale for Alpha Release:**\n\nThe decision to label this release as alpha stems from the realization that extensive testing and experimentation are necessary. Key considerations include:\n\n- Selection of appropriate metrics for evaluation\n- Determination of the optimal few-shot learning parameters\n-
1Establishment of the ideal number of evaluation samples within the dataset\n\n### **Roadmap for Next Release:**\n\nAnticipate the following enhancements in the upcoming release:\n\n- Enhanced testing and accountability mechanisms\n- A refined version of the leaderboard\n- Defined benchmarks and standardized datasets\n- Bilingual evaluation support\n- Expansion of supported models\n- Implementation of more secure interaction mechanisms\n- Addition of support for additional languages\n\n## How to show up on the Leaderboard\n\nHere are the steps you will have to follows to put your model on the Indic LLM leaderboard\n\nâ\n\nClone the repo:\n\n```jsx\ngit clone \u003chttps://github.com/adithya-s-k/indic_eval\u003e\ncd indic_eval\n```\n\nCreate a virtual environment using virtualenv or conda depending on your preferences. We require Python 3.10 or above:`â`\n\n```jsx\nconda create -n indic-eval-venv python=3.10 \u0026\u0026 conda activate indic-eval-venv`âInstall the dependencies. For the default installation, you just need:â`\n```\n\n```jsx\npip install .\n```\n\nIf you want to evaluate models with frameworks like `accelerate` or `peft`, you will need to specify the optional dependencies group that fits your use case (`accelerate`, `tgi`, `optimum`, `quantization`, `adapters`, `nanotron`):\n\n```jsx\npip install .[optional1,optional2]\n```\n\nThe setup tested most is:\n\n```jsx\npip install .[accelerate,quantization,adapters]\n```\n\nâ\nIf you want to push your results to the Hugging Face Hub, don't forget to add your access token to the environment variable `HUGGING_FACE_HUB_TOKEN`. You can do this by running:\n\n```jsx\nhuggingface-cli login\n```\n\n## Command to Run Indic Eval and Push to Indic LLM Leaderboard\n\n```bash\naccelerate launch run_indic_evals_accelerate.py \\\n --model_args=\"pretrained=\u003cpath to model on the hub\u003e\" \\\n --language kannada \\\n --tasks indic_llm_leaderboard \\\n --output_dir output_dir \\\n --push_to_leaderboard \[email protected]\u003e \\\n```\n\nIt's as simple as that.ð\n\nFor `--push_to_leaderboard`, provide an email id through which we can contact you in case of verification. This email won't be shared anywhere. It's only required for future verification of the model's scores and for authenticity.\n\nAfter you have installed all the required packages, run the following command:\n\nFor multi-GPU configuration, please refer to the docs of [Indic_Eval](https://github.com/adithya-s-k/indic_eval).\n\n## Some common questions\n\n### **What is the minimum `requirement` for GPUs to run the evaluation?**\n\n- The evaluation can easily run on a single A100 GPU, but the framework also supports multi-GPU based evaluation to speed up the process.\n\n### **What languages are supported by the evaluation framework?**\n\n- The following languages are supported by default: `english`, `kannada`, `hindi`, `tamil`, `telugu`, `gujarati`, `marathi`, `malayalam`\n\n### **How can I put my model on the leaderboard?**\n\n- Please follow the steps shown in the Submit tab or refer to the indic_eval for more details.\n\n### **How does the leaderboard work?**\n\n- After running indic_eval on the model of your choice, the results are pushed to a server and stored in a database. The Frontend Leaderboard accesses the server and retrieves the latest models in the database along with their respective benchmarks and metadata. The entire system is deployed in India and is as secure as possible.\n\n### **How is it different from the Open LLM leaderboard?**\n\n- This project was mainly inspired by the Open LLM leaderboard. However, due to limited computation resources, we standardized the evaluation library with standard benchmarks. You can run the evaluation on your GPUs and the leaderboard will serve as a unified platform to compare models. We used indictrans2 and other translation APIs to translate the benchmarking dataset into seven Indian languages to ensure reliability and consistency in the output.\n\n### **Why does it take so much time to load the results?**\n\n- We are running the server on a serverless instance which has a cold start problem, so it might sometimes take a while.\n\n### **What benchmarks are offered?**\n\n- The current Indic Benchmarks offered by the indic_eval library can be found in this collection: https://huggingface.co/collections/Cognitive-Lab/indic-llm-leaderboard-eval-suite-660ac4818695a785edee4e6f. They include ARC Easy, ARC Challenge, Hellaswag, Boolq, and MMLU.\n\n### **How much time does it take to run the evaluation using indic_eval?**\n\n- Depen
1ding on which GPU you are running, the time for evaluation varies.\n- From our testing, it takes 5 to 10 hours to run the whole evaluation on a single GPU depends on the GPU\n- It's much faster when using multiple GPUs.\n\n### **How does the verification step happen?**\n\nWhile running the evaluation, you are given an option to push results to the leaderboard with\n\n- `-push_to_leaderboard \[email protected]\u003e`. You will need to provide an email address through which we can contact you. If we find any anomaly in the evaluation score, we will contact you through this email for verification of results.\n\n## **Use Cases:**\n\n- **Research and Development**: Researchers and developers can utilize the Indic LLM Framework to adapt pre-trained LLMs to specific domains and languages, facilitating the development of customized language models tailored to diverse applications such as sentiment analysis, machine translation, and text generation.\n- **Model Evaluation and Comparison**: The Indic Eval Suite enables researchers to evaluate the performance of Indic LLMs across various tasks and benchmarks, allowing for comprehensive performance assessment and comparison. This aids in identifying strengths and weaknesses of different models and guiding future research directions.\n- **Benchmarking and Progress Tracking**: The Indic LLM Leaderboard provides a centralized platform for benchmarking the performance of Indic LLMs against standardized benchmarks and tracking progress over time. This fosters transparency, collaboration, and healthy competition within the research community, driving continuous improvement in Indic language modeling\n\n## **Call for Collaborative Effort:**\n\nTo foster collaboration and discussion surrounding evaluations, a [WhatsApp group](https://chat.whatsapp.com/CUb6eS50lX2JHX2D4j13d1) is being established and we can also connect on Hugging faces discord [indic_llm channel](https://discord.com/channels/879548962464493619/1189605147068858408)\n\n**â**\n\n## **Contribute**\n\nAll the projects are completely open source with different licenses, so anyone can contribute.\n\nThe current leaderboard is in alpha release, and many more changes are forthcoming:\n\n- More robust benchmarks tailored for Indic languages.\n- Easier integration with [indic_eval](https://github.com/adithya-s-k/indic_eval).\n\nâ\n\n## Conclusion\n\nThe alpha release of the Indic LLM Leaderboard and Indic Eval marks an important first step towards establishing standardized evaluation frameworks for Indic language models. However, it is clear that significant work remains to make these tools truly robust and accountable.\n\nThe decision to label this as an alpha release underscores the need for extensive testing, refinement of evaluation metrics, and determination of optimal setup parameters. The path ahead requires close collaboration and active participation from the open-source community.\n\nWe call upon researchers, developers, and language enthusiasts to join us in this endeavor. By pooling our collective expertise and resources, we can build a comprehensive, transparent, and reliable system to assess the performance of Indic LLMs. This will not only drive progress in the field but also ensure that advancements truly benefit the diverse linguistic landscape of the Indian subcontinent.\n\nThe road ahead is long, but the potential impact is immense. With your support and involvement, we are confident that the Indic LLM Leaderboard and Indic Eval will evolve into reliable and indispensable tools for the Indic language modeling community. Together, let us embark on this journey to unlock the full potential of Indic language technologies.\n"])</script>
1<script>self.__next_f.push([1,"24:{\"title\":\"Introducing Indic LLM Leaderboard: Benchmarking Indian Language Models\",\"excerpt\":\"The Indic LLM Leaderboard is a comprehensive evaluation platform for language models across 22+ Indian languages, providing standardized metrics and culturally relevant evaluation tasks to accelerate progress in Indian language AI research.\",\"coverImage\":\"/assets/blog/introducing-indic-llm-leaderboard/cover.png\",\"date\":\"2024-04-25T12:00:00.000Z\",\"contentType\":\"announcement\",\"author\":\"$25\",\"ogImage\":\"$26\",\"slug\":\"introducing-indic-llm-leaderboard\",\"content\":\"$27\"}\n29:{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"}\n2a:{\"url\":\"/assets/blog/introducing-ambari/cover.png\"}\n2b:T31bb,"])</script>
1<script>self.__next_f.push([1,"\n\u003c!-- ### Introducing Ambari --\u003e\n\n\u003c!--  --\u003e\n\nIn this blog, I am thrilled to share insights into the meticulous approach we undertook to train Amabri Base ([Cognitive-Lab/Ambari-7B-Instruct-v0.1](https://huggingface.co/Cognitive-Lab/Ambari-7B-Instruct-v0.1)) and Amabri Instruct ([Cognitive-Lab/Ambari-7B-Instruct-v0.1](https://huggingface.co/Cognitive-Lab/Ambari-7B-Instruct-v0.1)). Offering a high-level glimpse into our process, this narrative serves as a precursor to the forthcoming revelation of all technical detailsâ the culmination of extensive testing and evaluation. Stay tuned as we unravel the intricacies that led to the creation of Amabri, an innovative open-source bilingual Kannada-English Large Language Model.\n\n# Why We Built Amabri\n\n**Purpose Behind Amabri**\n\nIn the dynamic landscape of Large Language Models (LLMs), the creation of Amabri stemmed from a multifaceted purpose:\n\n- **Language Adaptation of LLMs:** Our primary objective was to pioneer language adaptability within LLMs, bridging the linguistic gap between Kannada and English.\n- **Training/Finetuning on a Modest 1B-Token Dataset:** Recognizing the constraints posed by smaller datasets, we aimed to push the boundaries of efficiency by training and finetuning Amabri on a relatively compact dataset of 1 billion tokens.\n- **Identifying the Most Efficient Process:** The quest for efficiency led us to meticulously explore and determine the most effective processes at each stage of Amabri's development.\n- **Observing World Knowledge Acquisition:** Amabri was conceived as a lens through which we could observe the accrual of world knowledge throughout the training process, shedding light on its adaptability and expansive learning capabilities.\n- **Optimizing Training Methods for Each Stage:** A crucial aspect of our endeavor was to discern and optimize the training methods suited for each developmental stage of Amabri.\n\nAs LLMs increasingly permeate mainstream usage, open-source models, while enriched in world knowledge, predominantly emerge from English-centric training. Amabri serves as a pioneering initiative to broaden this scope and adapt LLMs to diverse languages.\n\n## Introduction\n\nIn the evolving landscape of LLMs, the demand for vast amounts of training data, ranging from 1 trillion to 10 trillion tokens, has become a norm. However, this poses a challenge for languages with limited documented resources. In our pursuit, we focused on the adaptation of a pre-trained LLM, such as Llama/Mistral, to comprehend the nuances of a new languageâKannada in the case of Amabri. Despite Kannada not being classified as a very low-resource language, it served as an ideal candidate to test our hypotheses and methodologies. Rigorously defining the stages of training and finetuning, we set a cap of 1 billion training tokens for the entire process.\n\nSubsequently, we meticulously crafted datasets, distributed them accordingly, and delineated the stages of our process:\n\n- **Pre-training:** 500 Million tokens\n- **Bilingual Next Token Prediction/Translation:** ~300 Million tokens\n- **Instruct Finetuning/DPO Finetuning:** ~200 Million tokens\n\nThis deliberate approach laid the foundation for Amabri's development, pushing the boundaries of language adaptability within the realm of LLMs.\n\n## Tokenization\n\nTokenization, a critical component in the efficiency of language models, posed a unique challenge for Kannada text within the context of open-source LLMs. Many existing models inefficiently resort to character-level tokenization, especially during inference, impacting overall performance. To address this, we developed a specialized tokenization model for Kannada text using SentencePiece. This model was seamlessly integrated with the base Llama tokenizer, resulting in a comprehensive vocabulary of 49,600 , expanded by 17,600 .\n\nOur approach involved training the tokenizer model on three different dataset sizes, revealing optimal results with a dataset comprising 100,000 tokens. As we evolve Amabri, the upcoming iteration will feature a refined tokenization strategy, employing a reduced vocabulary size of 48,000. This adjustment, validated by insights shared by Andrej Karpathy in his Twitter post ([Andrej Karpathy on Twitter](https://twitter.com/karpathy/status/1621578354024677377)), is geared towards enhancing overall efficiency.\n\nCurious to explore the efficiency gains firsthand? You can test out the tokenizer in action [here](https://github.com/adithya-s-k/LLM-Alchemy-Chamber/blob/main/LLMs/ambari/tokeniser.ipynb).\n\n## Continual Pre-Training\n\n**Pre-Training**\n\nWith an efficie
1nt tokenizer in place, our next crucial step was the pre-training phase, aimed at familiarizing the model with the newly enriched vocabulary. To optimize this process, we curated a comprehensive dataset from diverse sources. Notably, we explored two distinct approaches during this phaseâpre-training with Lora and fully training the model. This strategic decision stemmed from our desire to discern the optimal path for Amabri's development.\n\nA detailed comparison between these methodologies will be unveiled shortly, but we've gleaned some initial observations:\n\n- Contrary to our hypothesis, full-weight fine-tuning did not result in a significant performance decrease compared to Lora.\n- The fully fine-tuned model exhibited a remarkable increase in predicting Kannada tokens when the preceding token was Kannada, indicating enhanced language understanding.\n- Additionally, we noted a heightened robustness in the generation capabilities of the fully fine-tuned model.\n\nWhile we acknowledge that our ongoing testing may refine these observations, this snapshot provides valuable insights into our progress. The pre-training phase employed a cluster of 2xA100 GPUs, taking approximately 25 hours for full-weight pre-training on a substantial corpus comprising 500 million tokens.\n\nIt's worth mentioning that the weights of the fully fine-tuned model are now available on [Hugging Face](https://huggingface.co/Cognitive-Lab/Ambari-7B-base-v0.1)ð¤ - https://huggingface.co/Cognitive-Lab/Ambari-7B-base-v0.1, contributing to the open-source knowledge sharing within the community.\n\n## Bilingual Next Token Prediction and Translation\n\n**Bilingual Next Token Prediction**\n\nThis phase, inspired by the open Hathi series by [sarvam.ai](http://sarvam.ai/), was an unplanned yet pivotal addition to our training strategy. Creating a dataset of 200,000 tokens, we utilized Lora for fine-tuning, aiming to equip the model with enhanced language understanding. As we progressed, our focus shifted towards instilling 'world knowledge' in Kannada. Given the scarcity of Kannada content, especially compared to English, we turned to translation. Leveraging IndicTrans2, we translated English content, primarily sourced from Wikipedia, into Kannada. However, instead of conventional monolingual next token prediction, we introduced a groundbreaking approach â bilingual next token prediction. Alternating sentences between Kannada and English, this method compelled the model to cross-lingually attend to information during next-token prediction. This nuanced approach not only fostered increased alignment between Kannada and English but also naturally balanced exposure to Hindi and English tokens during training. This stage added an extra layer of sophistication to Amabri's training journey.\n\n**Translation Finetuning**\n\nThe intention behind this phase was to establish a coherent relationship between English and corresponding Kannada tokens. Employing low-rank adaptation for fine-tuning, we encountered some challenges, notably with the decision to use a very low-rank value, which proved less effective. With a dataset size of 100,000 tokens, this stage presented limitations, and we acknowledge the need for improvements. As we refine this aspect of the training process, our commitment to enhancing the bilingual capabilities of Amabri remains unwavering.\n\n## Bilingual Instruct Fine-tuning\n\n**Bilingual Instruct Fine-tuning**\n\nIn this pivotal stage, we employed supervised fine-tuning with low-rank adaptation to mold the model's responsiveness. Embracing a chat template structure consisting of user prompts/instructions and corresponding responses, we ventured into the realm of Bilingual Instruct Fine-tuning. This approach involved training the model to adeptly respond in either English or Kannada based on the language specified in the user prompt or instruction.\n\nChat Template\n\n```bash\n\u003c|user|\u003e\n{user prompt / instruction}\n\u003c|endoftext|\u003e\n\u003c|assistant|\u003e\n{response}\n\u003c|endoftext|\u003e\n```\n\nFor instance, given a user prompt like\n\n\"Give me 10 Study tips
1in Kannada,\"\n\n\u003e Response\n\nthe model seamlessly generates a response in Kannada, maintaining linguistic coherence. To enrich the training process, we amalgamated various instruction datasets, including [Alpaca Instruct](https://huggingface.co/datasets/tatsu-lab/alpaca), [Dolly Instruct](https://huggingface.co/datasets/c-s-ale/dolly-15k-instruction-alpaca-format), and more. Leveraging translation APIs such as Google, Azure, and a custom deployment of the IndicTrans2 model from [ai4bharat](https://ai4bharat.iitm.ac.in/), we crafted a comprehensive bilingual instruct dataset.\n\nThe dataset, now publicly available on Hugging Face [here](https://huggingface.co/datasets/Cognitive-Lab/Kannada-Instruct-dataset), encompasses diverse linguistic scenarios. During training, we implemented supervised fine-tuning with four distinct representations:\n\n1. Kannada Instruction â Kannada Output\n2. English Instruction â Kannada Output\n3. Kannada Instruction â English Output\n4. English Instruction â Kannada Output\n\nThis meticulous approach not only familiarized the model with responding in different languages but also laid the groundwork for mastering various cross-lingual tasks.\n\nThe weights of this finely-tuned model are accessible on Hugging Face, and for a hands-on experience, you can explore the 4-bit quantized version on [chat.cognitivelab.in](https://chat.cognitivelab.in/).\n\n## DPO Fine-tuning\n\nIn the culminating phase of our model refinement, we delved into the world of Direct Preference Optimization (DPO). This strategic choice, inspired by the success observed in various open-source models, aimed not only to align our model but also to drive improvements in benchmarks. Embarking on this experimental journey, we leveraged the [Anthropic/hh-rlhf](https://huggingface.co/datasets/Anthropic/hh-rlhf) dataset. Translating it to Kannada, we subjected the model to DPO fine-tuning, currently undergoing a comprehensive evaluation to gauge its performance impact.\n\n## Learnings and Conclusion: Navigating Challenges and Unveiling Potential\n\n**Scope of Improvement**\n\n- **Lack of World Knowledge:** The model, trained with a capped dataset of 1 billion Kannada tokens, exhibits occasional hallucinations in response to highly specific queries, signaling a scope for improvement in imparting broader world knowledge.\n- **Translation Challenges:** Notably, translation nuances surface when handling nouns like names and places. Addressing this challenge involves allocating more training data specifically for the translation phase, a key focus for future enhancements.\n- **Full Weight Fine-tuning Dilemma:** An observation surfaced regarding the model's slight overfitting to predict Kannada tokens due to full weight fine-tuning. Future iterations will strategically decide between full weight and Lora for continual pre-training based on extensive tests and evaluations.\n\n## What's Next: Evolution and Enrichment\n\n- **Addition of Romanized Kannada:** A pivotal expansion awaits with the incorporation of Romanized Kannada, enriching the model's linguistic versatility. An illustrative example: âKannada **ನನà³à²¨ ಹà³à²¸à²°à³ à²
ಮಾಬà³à²°à²¿**â (Nanna hesaru amÄbri).\n- **Continuous Learning and Model Refinement:** Building on the learnings from this model version, we commit to refining our data pipelines and fine-tuning stages. The subsequent iteration will scale the training dataset to approximately 10 to 15 billion Kannada tokens, optimizing data distribution across diverse stages for enhanced efficiency.\n\n## Acknowledgments: Standing on the Shoulders of Inspiring Projects\n\nOur journey has been shaped by inspiration drawn from impactful projects. Special mention goes to [Sarvam.ai](http://sarvam.ai/) and the illuminating project [Tamil Llama](https://github.com/abhinand5/tamil-llama) by [Abhinand Balachandran](https://www.linkedin.com/in/abhinand-05/). Their contributions have been instrumental in steering the course of our own endeavors.\n"])</script>
1<script>self.__next_f.push([1,"28:{\"title\":\"Introducing Ambari: Biligual Kannada English Large Language Model\",\"excerpt\":\"Ambari is a comprehensive natural language processing platform specifically designed for Indian languages, enabling developers to build sophisticated language applications with unprecedented accuracy and cultural context across 22+ Indian languages.\",\"coverImage\":\"/assets/blog/introducing-ambari/cover.png\",\"date\":\"2024-01-01T12:00:00.000Z\",\"contentType\":\"announcement\",\"author\":\"$29\",\"ogImage\":\"$2a\",\"slug\":\"introducing-ambari\",\"content\":\"$2b\"}\n39:{\"src\":\"/_next/static/media/grid.947ccc23.png\",\"height\":1100,\"width\":1100,\"blurDataURL\":\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAgAAAAICAMAAADz0U65AAAABlBMVEUqDyQMBwzhvROvAAAAAnRSTlMDFEdkSoYAAAAJcEhZcwAAFiUAABYlAUlSJPAAAAATSURBVHicY2CEAgYyACMDA0w7AAKTABJbPgamAAAAAElFTkSuQmCC\",\"blurWidth\":8,\"blurHeight\":8}\n3f:[]\n"])</script>
1<script>self.__next_f.push([1,"0:[[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/css/7cca8e2c5137bd71.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\"}],[\"$\",\"link\",\"1\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/css/ede4122d18cd4431.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\"}]],[\"$\",\"$L4\",null,{\"buildId\":\"csuyt7cj-0jFyxIywf9vM\",\"assetPrefix\":\"\",\"initialCanonicalUrl\":\"/blog\",\"initialTree\":[\"\",{\"children\":[\"blog\",{\"children\":[\"__PAGE__\",{}]}]},\"$undefined\",\"$undefined\",true],\"initialSeedData\":[\"\",{\"children\":[\"blog\",{\"children\":[\"__PAGE__\",{},[[\"$L5\",[\"$\",\"main\",null,{\"className\":\"overflow-hidden\",\"children\":[[\"$\",\"$L6\",null,{\"title\":\"Blog \u0026 Announcements\",\"subtitle\":\"Insights, tutorials, and updates from our team of AI researchers and engineers\",\"highlightText\":\"Announcements\",\"highlightBadge\":\"Latest Articles\",\"emoji\":\"âï¸\",\"height\":\"50vh\"}],[\"$\",\"div\",null,{\"id\":\"$undefined\",\"className\":\"relative py-8 lg:py-10 xl:py-12 lg:py-24 xl:py-32 pb-16 pt-12\",\"children\":[[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-8 flex items-end justify-between\",\"children\":[[\"$\",\"h2\",null,{\"className\":\"text-4xl font-bold tracking-tight lg:text-5xl\",\"children\":[[\"$\",\"span\",null,{\"className\":\"animate-gradient-text\",\"children\":\"Featured\"}],\" \"]}],[\"$\",\"$L7\",null,{\"href\":\"/blog\",\"className\":\"group flex items-center text-n-3 transition-colors hover:text-n-1\",\"children\":[\"View all\",[\"$\",\"svg\",null,{\"xmlns\":\"http://www.w3.org/2000/svg\",\"width\":24,\"height\":24,\"viewBox\":\"0 0 24 24\",\"fill\":\"none\",\"stroke\":\"currentColor\",\"strokeWidth\":2,\"strokeLinecap\":\"round\",\"strokeLinejoin\":\"round\",\"className\":\"lucide lucide-chevron-right ml-1 h-5 w-5 transition-transform group-hover:translate-x-1\",\"children\":[[\"$\",\"path\",\"mthhwq\",{\"d\":\"m9 18 6-6-6-6\"}],\"$undefined\"]}]]}]]}],[\"$\",\"div\",null,{\"className\":\"mb-6\",\"children\":[\"$\",\"$L8\",null,{\"format\":\"featured\",\"blog\":{\"title\":\"Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval\",\"excerpt\":\"We're excited to announce NetraEmbed and ColNetraEmbed, achieving 152% improvement over existing systems in multilingual document retrieval. Supporting 22 languages with state-of-the-art performance.\",\"coverImage\":\"/assets/blog/introducing-netraembed/cover.png\",\"date\":\"2025-12-08T10:00:00.000Z\",\"contentType\":\"announcement\",\"author\":{\"name\":\"CognitiveLab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"},\"ogImage\":{\"url\":\"/assets/blog/introducing-netraembed/cover.png\"},\"tags\":[\"Multilingual AI\",\"Document Retrieval\",\"Vision-Language Models\",\"RAG\"],\"slug\":\"introducing-netraembed\",\"content\":\"$9\"},\"mainFeature\":true}]}],[\"$\",\"div\",null,{\"className\":\"grid grid-cols-1 gap-6 sm:grid-cols-2 md:grid-cols-2 lg:grid-cols-3\",\"children\":[[\"$\",\"div\",\"introducing-nayana\",{\"className\":\"w-full\",\"children\":[\"$\",\"$L8\",null,{\"format\":\"featured\",\"blog\":{\"title\":\"CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana\",\"excerpt\":\"CognitiveLab is proud to be selected as a recipient of Meta's prestigious Llama Impact Grant 2024, accelerating our revolutionary Nayana projectâa multilingual (22 languages, 10 Indic), multimodal AI ecosystem that democratizes AI across global languages.\",\"coverImage\":\"/assets/blog/introducing-nayana/cover.png\",\"date\":\"2025-04-30T10:00:00.000Z\",\"contentType\":\"announcement\",\"author\":{\"name\":\"CognitiveLab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"},\"ogImage\":{\"url\":\"/assets/blog/introducing-nayana/cover.png\"},\"slug\":\"introducing-nayana\",\"content\":\"$a\"}}]}],[\"$\",\"div\",\"introducing-ai-engineering-academy\",{\"className\":\"w-full\",\"children\":[\"$\",\"$L8\",null,{\"format\":\"featured\",\"blog\":{\"title\":\"Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers\",\"excerpt\":\"AI Engineering Academy offers comprehensive training programs and resources to help aspiring engineers master the technical skills and practical knowledge needed to build, deploy, and maintain AI systems at scale.\",\"coverImage\":\"/assets/blog/introducing-ai-engineering-academy/cover.png\",\"date\":\"2024-12-11T12:00:00.000Z\",\"contentType\":\"blog\",\"author\":{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"},\"ogImage\":{\"url\":\"/assets/blog/introducing-ai-engineering-academy/cover.png\"},\"slug\":\"introducing-ai-engineering-academy\",\"content\":\"$b\"}}]}],[\"$\",\"div\",\"introducing-omniparse\",{\"className\":\"w-full\",\"children\":[\"$\",\"$L8\",null,{\"format\":\"featured\",\"blog\":{\"title\":\"Introducing Omniparse: Universal Data Parsing\",\"excerpt\":\"Omniparse is an advanced document parsing platform that uses AI to extract structured data from any document format, enabling businesses to automate document processing workflows with unprecedented accuracy and efficie
1ncy.\",\"coverImage\":\"/assets/blog/introducing-omniparse/cover.png\",\"date\":\"2024-06-27T12:00:00.000Z\",\"contentType\":\"blog\",\"author\":{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"},\"ogImage\":{\"url\":\"/assets/blog/introducing-omniparse/cover.png\"},\"slug\":\"introducing-omniparse\",\"content\":\"$c\"}}]}]]}]]}],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10\"}],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10\"}]]}],[\"$\",\"div\",null,{\"id\":\"$undefined\",\"className\":\"relative py-8 lg:py-10 xl:py-12 lg:py-24 xl:py-32 pb-20 pt-4\",\"children\":[[[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-8\",\"children\":[\"$\",\"h2\",null,{\"className\":\"text-4xl font-bold tracking-tight lg:text-5xl\",\"children\":[\"All\",\" \",[\"$\",\"span\",null,{\"className\":\"text-gradient-500\",\"children\":\"Content\"}]]}]}],[\"$\",\"$Ld\",null,{\"defaultValue\":\"all\",\"className\":\"w-full\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-6 overflow-auto pb-2\",\"children\":[\"$\",\"$Le\",null,{\"className\":\"inline-flex w-auto justify-start gap-4 overflow-visible rounded-lg border border-n-6 bg-n-7/50 p-1 backdrop-blur-sm\",\"children\":[[\"$\",\"$Lf\",null,{\"value\":\"all\",\"className\":\"rounded-md px-5 py-2\",\"children\":\"All\"}],[\"$\",\"$Lf\",null,{\"value\":\"blog\",\"className\":\"rounded-md px-5 py-2\",\"children\":\"Blog Posts\"}],[\"$\",\"$Lf\",null,{\"value\":\"announcements\",\"className\":\"rounded-md px-5 py-2\",\"children\":\"Announcements\"}]]}]}],[\"$\",\"$L10\",null,{\"value\":\"all\",\"className\":\"mt-0\",\"children\":[\"$\",\"div\",null,{\"className\":\"flex flex-col space-y-6\",\"children\":[[\"$\",\"$L8\",\"Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval\",{\"format\":\"featured\",\"blog\":\"$11\",\"mainFeature\":true}],[\"$\",\"$L8\",\"CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana\",{\"format\":\"featured\",\"blog\":\"$16\",\"mainFeature\":true}],[\"$\",\"$L8\",\"Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers\",{\"format\":\"featured\",\"blog\":\"$1a\",\"mainFeature\":true}],[\"$\",\"$L8\",\"Introducing Omniparse: Universal Data Parsing\",{\"format\":\"featured\",\"blog\":\"$1e\",\"mainFeature\":true}],[\"$\",\"$L8\",\"Introducing Indic LLM Leaderboard: Benchmarking Indian Language Models\",{\"format\":\"featured\",\"blog\":{\"title\":\"Introducing Indic LLM Leaderboard: Benchmarking Indian Language Models\",\"excerpt\":\"The Indic LLM Leaderboard is a comprehensive evaluation platform for language models across 22+ Indian languages, providing standardized metrics and culturally relevant evaluation tasks to accelerate progress in Indian language AI research.\",\"coverImage\":\"/assets/blog/introducing-indic-llm-leaderboard/cover.png\",\"date\":\"2024-04-25T12:00:00.000Z\",\"contentType\":\"announcement\",\"author\":{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"},\"ogImage\":{\"url\":\"/assets/blog/introducing-indic-llm-leaderboard/cover.png\"},\"slug\":\"introducing-indic-llm-leaderboard\",\"content\":\"$22\"},\"mainFeature\":true}],[\"$\",\"$L8\",\"Introducing Ambari: Biligual Kannada English Large Language Model\",{\"format\":\"featured\",\"blog\":{\"title\":\"Introducing Ambari: Biligual Kannada English Large Language Model\",\"excerpt\":\"Ambari is a comprehensive natural language processing platform specifically designed for Indian languages, enabling developers to build sophisticated language applications with unprecedented accuracy and cultural context across 22+ Indian languages.\",\"coverImage\":\"/assets/blog/introducing-ambari/cover.png\",\"date\":\"2024-01-01T12:00:00.000Z\",\"contentType\":\"announcement\",\"author\":{\"name\":\"Cognitivelab Team\",\"picture\":\"/assets/cognitivelab-logo.png\"},\"ogImage\":{\"url\":\"/assets/blog/introducing-ambari/cover.png\"},\"slug\":\"introducing-ambari\",\"content\":\"$23\"},\"mainFeature\":true}]]}]}],[\"$\",\"$L10\",null,{\"value\":\"blog\",\"className\":\"mt-0\",\"children\":[\"$\",\"div\",null,{\"className\":\"flex flex-col space-y-6\",\"children\":[[\"$\",\"$L8\",\"Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers\",{\"format\":\"featured\",\"blog\":\"$1a\",\"mainFeature\":true}],[\"$\",\"$L8\",\"Introducing Omniparse: Universal Data Parsing\",{\"format\":\"featured\",\"blog\":\"$1e\",\"mainFeature\":true}]]}]}],[\"$\",\"$L10\",null,{\"value\":\"announcements\",\"className\":\"mt-0\",\"children\":[\"$\",\"div\",null,{\"className\":\"flex flex-col space-y-6\",\"children\":[[\"$\",\"$L8\",\"Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval\",{\"format\":\"featured\",\"blog\":\"$11\",\"mainFeature\":true}],[\"$\",\"$L8\",\"CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana\",{\"format\":\"featured\",\"blog\":\"$16\",\"mainFeature\":true}],[\"$\",\"$L8\",\"Introducing Indic LLM Leaderboard: Benchmarking Indian Language Models\",{\"format\":\"featured\",\"blog\":\"$24\",\"mainFeature\":true}],[\"$\",\"$L8\",\"Introducing Ambari: Biligual Kannada English L
1arge Language Model\",{\"format\":\"featured\",\"blog\":\"$28\",\"mainFeature\":true}]]}]}]]}]]}],[\"$\",\"$L2c\",null,{}]],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10\"}],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10\"}]]}]]}]],null],null]},[\"$\",\"$L2d\",null,{\"parallelRouterKey\":\"children\",\"segmentPath\":[\"children\",\"blog\",\"children\"],\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L2e\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"notFoundStyles\":\"$undefined\",\"styles\":null}],null]},[[\"$\",\"html\",null,{\"lang\":\"en\",\"children\":[\"$\",\"body\",null,{\"className\":\"__className_f367f3 relative min-h-screen\",\"children\":[[\"$\",\"$L2f\",null,{\"children\":[\"$\",\"$L30\",null,{\"attribute\":\"class\",\"defaultTheme\":\"dark\",\"enableSystem\":true,\"disableTransitionOnChange\":true,\"children\":[[\"$\",\"$L31\",null,{}],[\"$\",\"main\",null,{\"children\":[\"$\",\"$L2d\",null,{\"parallelRouterKey\":\"children\",\"segmentPath\":[\"children\"],\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L2e\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":[\"$\",\"div\",null,{\"className\":\"overflow-hidden pt-[4.75rem] lg:pt-[5.25rem]\",\"children\":[[\"$\",\"$L32\",null,{}],[\"$\",\"$L33\",null,{}],[\"$\",\"$L34\",null,{}],[\"$\",\"$L35\",null,{}],[\"$\",\"$L36\",null,{}],[\"$\",\"$L37\",null,{}],[\"$\",\"div\",null,{\"id\":\"roadmap\",\"className\":\"relative py-8 lg:py-10 xl:py-12 overflow-hidden\",\"children\":[[[\"$\",\"div\",null,{\"className\":\"container md:pb-10\",\"children\":[[\"$\",\"h2\",null,{\"className\":\"font-base h2 mb-6 text-center text-3xl md:mb-8 md:text-5xl lg:text-6xl\",\"children\":[\"Blogs \u0026\",\" \",[\"$\",\"span\",null,{\"className\":\"animated-gradient-text metallic-text font-bold\",\"children\":[\"Announcements\",\" \"]}]]}],[\"$\",\"p\",null,{\"className\":\"mx-auto mb-8 max-w-2xl text-center text-base text-n-3 md:mb-10 md:text-lg\",\"children\":\"Stay updated with our latest blogs and announcements showcasing our innovative projects and research advancements.\"}],[\"$\",\"div\",null,{\"className\":\"relative grid gap-6 md:grid-cols-2 md:gap-8 lg:grid-cols-2 xl:grid-cols-2\",\"children\":[[\"$\",\"$L7\",\"0\",{\"href\":\"/blog/introducing-netraembed\",\"passHref\":true,\"children\":[\"$\",\"div\",null,{\"className\":\"cursor-pointer rounded-2xl bg-conic-gradient p-0.25 transition-all duration-300 hover:scale-[1.02] hover:shadow-lg\",\"children\":[\"$\",\"div\",null,{\"className\":\"relative h-full overflow-hidden rounded-[18px] bg-n-8 p-6 backdrop-blur-sm md:p-8\",\"children\":[[\"$\",\"div\",null,{\"className\":\"absolute left-0 top-0 max-w-full\",\"children\":[\"$\",\"$L38\",null,{\"className\":\"w-full\",\"src\":{\"src\":\"/_next/static/media/grid.947ccc23.png\",\"height\":1100,\"width\":1100,\"blurDataURL\":\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAgAAAAICAMAAADz0U65AAAABlBMVEUqDyQMBwzhvROvAAAAAnRSTlMDFEdkSoYAAAAJcEhZcwAAFiUAABYlAUlSJPAAAAATSURBVHicY2CEAgYyACMDA0w7AAKTABJbPgamAAAAAElFTkSuQmCC\",\"blurWidth\":8,\"blurHeight\":8},\"width\":550,\"height\":550,\"alt\":\"Grid\"}]}],[\"$\",\"div\",null,{\"className\":\"relative z-1 flex h-full flex-col\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-4 flex items-center justify-between\",\"children\":[\"$\",\"div\",null,{\"className\":\"tagline flex items-center \",\"children\":[[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M5 0.822266H1V12.8223H5\",\"stroke\":\"url(#brackets-left)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-left\",\"x1\":\"50%\",\"x2\":\"50%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#89F9E8\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}]}]]}],[\"$\",\"div\",null,{\"className\":\"mx-3 text-n-3\",\"children\":\"December 8, 2025\"}
1],[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M-2.98023e-08 0.822266H4V12.8223H-2.98023e-08\",\"stroke\":\"url(#brackets-right)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-right\",\"x1\":\"14.635%\",\"x2\":\"14.635%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#D87CEE\"}]]}]}]]}]]}]}],[\"$\",\"h4\",null,{\"className\":\"h4 mb-4 flex-grow\",\"children\":\"Introducing NetraEmbed - SoTA Multimodal Multilingual Document Retrieval\"}],[\"$\",\"p\",null,{\"className\":\"body-2 mb-4 line-clamp-3 text-n-4\",\"children\":\"We're excited to announce NetraEmbed and ColNetraEmbed, achieving 152% improvement over existing systems in multilingual document retrieval. Supporting 22 languages with state-of-the-art performance.\"}],[\"$\",\"span\",null,{\"className\":\"inline-block font-bold text-n-1 transition-all duration-300 group-hover:translate-x-2\",\"children\":\"Read More â\"}]]}]]}]}]}],[\"$\",\"$L7\",\"1\",{\"href\":\"/blog/introducing-nayana\",\"passHref\":true,\"children\":[\"$\",\"div\",null,{\"className\":\"cursor-pointer rounded-2xl bg-conic-gradient p-0.25 transition-all duration-300 hover:scale-[1.02] hover:shadow-lg\",\"children\":[\"$\",\"div\",null,{\"className\":\"relative h-full overflow-hidden rounded-[18px] bg-n-8 p-6 backdrop-blur-sm md:p-8\",\"children\":[[\"$\",\"div\",null,{\"className\":\"absolute left-0 top-0 max-w-full\",\"children\":[\"$\",\"$L38\",null,{\"className\":\"w-full\",\"src\":\"$39\",\"width\":550,\"height\":550,\"alt\":\"Grid\"}]}],[\"$\",\"div\",null,{\"className\":\"relative z-1 flex h-full flex-col\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-4 flex items-center justify-between\",\"children\":[\"$\",\"div\",null,{\"className\":\"tagline flex items-center \",\"children\":[[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M5 0.822266H1V12.8223H5\",\"stroke\":\"url(#brackets-left)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-left\",\"x1\":\"50%\",\"x2\":\"50%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#89F9E8\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}]}]]}],[\"$\",\"div\",null,{\"className\":\"mx-3 text-n-3\",\"children\":\"April 30, 2025\"}],[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M-2.98023e-08 0.822266H4V12.8223H-2.98023e-08\",\"stroke\":\"url(#brackets-right)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-right\",\"x1\":\"14.635%\",\"x2\":\"14.635%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#D87CEE\"}]]}]}]]}]]}]}],[\"$\",\"h4\",null,{\"className\":\"h4 mb-4 flex-grow\",\"children\":\"CognitiveLab Wins Meta's Llama Impact Grant 2024 for Project Nayana\"}],[\"$\",\"p\",null,{\"className\":\"body-2 mb-4 line-clamp-3 text-n-4\",\"children\":\"CognitiveLab is proud to be selected as a recipient of Meta's prestigious Llama Impact Grant 2024, accelerating our revolutionary Nayana projectâa multilingual (22 languages, 10 Indic), multimodal AI ecosystem that democratizes AI across global languages.\"}],[\"$\",\"span\",null,{\"className\":\"inline-block font-bold text-n-1 transition-all duration-300 group-hover:translate-x-2\",\"children\":\"Read More â\"}]]}]]}]}]}],[\"$\",\"$L7\",\"2\",{\"href\":\"/blog/introducing-ai-engineering-academy\",\"passHref\":true,\"children\":[\"$\",\"div\",null,{\"className\":\"cursor-pointer rounded-2xl bg-conic-gradient p-0.25 transition-all duration-300 hover:scale-[1.02] hover:shadow-lg\",\"children\":[\"$\",\"div\",null,{\"className\":\"relative h-full overflow-hidden rounded-[18px] bg-n-8 p-6 backdrop-blur-sm md:p-8\",\"children\":[[\"$\",\"div\",null,{\"className\":\"absolute left-0 top-0 max-w-full\",\"children\":[\"$\",\"$L38\",null,{\"className\":\"w-full\",\"src\":\"$39\",\"width\":550,\"height\"
1:550,\"alt\":\"Grid\"}]}],[\"$\",\"div\",null,{\"className\":\"relative z-1 flex h-full flex-col\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-4 flex items-center justify-between\",\"children\":[\"$\",\"div\",null,{\"className\":\"tagline flex items-center \",\"children\":[[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M5 0.822266H1V12.8223H5\",\"stroke\":\"url(#brackets-left)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-left\",\"x1\":\"50%\",\"x2\":\"50%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#89F9E8\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}]}]]}],[\"$\",\"div\",null,{\"className\":\"mx-3 text-n-3\",\"children\":\"December 11, 2024\"}],[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M-2.98023e-08 0.822266H4V12.8223H-2.98023e-08\",\"stroke\":\"url(#brackets-right)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-right\",\"x1\":\"14.635%\",\"x2\":\"14.635%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#D87CEE\"}]]}]}]]}]]}]}],[\"$\",\"h4\",null,{\"className\":\"h4 mb-4 flex-grow\",\"children\":\"Introducing AI Engineering Academy: Creating the Next Generation of AI Engineers\"}],[\"$\",\"p\",null,{\"className\":\"body-2 mb-4 line-clamp-3 text-n-4\",\"children\":\"AI Engineering Academy offers comprehensive training programs and resources to help aspiring engineers master the technical skills and practical knowledge needed to build, deploy, and maintain AI systems at scale.\"}],[\"$\",\"span\",null,{\"className\":\"inline-block font-bold text-n-1 transition-all duration-300 group-hover:translate-x-2\",\"children\":\"Read More â\"}]]}]]}]}]}],[\"$\",\"$L7\",\"3\",{\"href\":\"/blog/introducing-omniparse\",\"passHref\":true,\"children\":[\"$\",\"div\",null,{\"className\":\"cursor-pointer rounded-2xl bg-conic-gradient p-0.25 transition-all duration-300 hover:scale-[1.02] hover:shadow-lg\",\"children\":[\"$\",\"div\",null,{\"className\":\"relative h-full overflow-hidden rounded-[18px] bg-n-8 p-6 backdrop-blur-sm md:p-8\",\"children\":[[\"$\",\"div\",null,{\"className\":\"absolute left-0 top-0 max-w-full\",\"children\":[\"$\",\"$L38\",null,{\"className\":\"w-full\",\"src\":\"$39\",\"width\":550,\"height\":550,\"alt\":\"Grid\"}]}],[\"$\",\"div\",null,{\"className\":\"relative z-1 flex h-full flex-col\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-4 flex items-center justify-between\",\"children\":[\"$\",\"div\",null,{\"className\":\"tagline flex items-center \",\"children\":[[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M5 0.822266H1V12.8223H5\",\"stroke\":\"url(#brackets-left)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-left\",\"x1\":\"50%\",\"x2\":\"50%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#89F9E8\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}]}]]}],[\"$\",\"div\",null,{\"className\":\"mx-3 text-n-3\",\"children\":\"June 27, 2024\"}],[\"$\",\"svg\",null,{\"width\":\"5\",\"height\":\"14\",\"viewBox\":\"0 0 5 14\",\"fill\":\"none\",\"xmlns\":\"http://www.w3.org/2000/svg\",\"children\":[[\"$\",\"path\",null,{\"d\":\"M-2.98023e-08 0.822266H4V12.8223H-2.98023e-08\",\"stroke\":\"url(#brackets-right)\"}],[\"$\",\"defs\",null,{\"children\":[\"$\",\"linearGradient\",null,{\"id\":\"brackets-right\",\"x1\":\"14.635%\",\"x2\":\"14.635%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#D87CEE\"}]]}]}]]}]]}]}],[\"$\",\"h4\",null,{\"className\":\"h4 mb-4 flex-grow\",\"children\":\"Introducing Omniparse: Universal Data Parsing\"}],[\"$\",\"p\",null,{\"className\":\"body-2 mb-4 line-clamp-3 text-n-4\",\"children\":\"Omniparse is an advanced document parsing platform that uses AI to extract structured data from any document format, enabling businesses to automate document processing workflows with unprecedented accuracy and efficie
1ncy.\"}],[\"$\",\"span\",null,{\"className\":\"inline-block font-bold text-n-1 transition-all duration-300 group-hover:translate-x-2\",\"children\":\"Read More â\"}]]}]]}]}]}]]}],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute -left-[30.375rem] top-[18.25rem] w-[56.625rem] opacity-60 mix-blend-color-dodge\",\"children\":[\"$\",\"div\",null,{\"className\":\"absolute left-1/2 top-1/2 h-[58.85rem] w-[58.85rem] -translate-x-3/4 -translate-y-1/2\",\"children\":[\"$\",\"$L38\",null,{\"className\":\"w-full\",\"src\":{\"src\":\"/_next/static/media/gradient.15bf030e.png\",\"height\":1417,\"width\":1417,\"blurDataURL\":\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAgAAAAICAMAAADz0U65AAAAWlBMVEUuWYs1aaYxc6sZfqg0Uos1ToCyTrZQOHUAAACAPHAWm8OvPInBTr+qZsE9YZ1lQGtcap6WPWrER5heh82lS3wkj7p6bcGnO5JqUX4Rk7XHRq+LbKFJm9BDksL9gs4BAAAAG3RSTlMQhvGEgkcQCQER9oaF+oiD9Yj09fTriUnqh+mqTHGvAAAACXBIWXMAAAsTAAALEwEAmpwYAAAAQklEQVR4nBXGRxKAMAwAsU3DdkLovfz/mww6CRUzM1Ek9WMpu0A8lulqG+iGu9Y/zr/PvDUQfc6rA0kxnCEJKgCiH2EQAmpcsC38AAAAAElFTkSuQmCC\",\"blurWidth\":8,\"blurHeight\":8},\"width\":942,\"height\":942,\"alt\":\"Gradient\"}]}]}],[\"$\",\"div\",null,{\"className\":\"mt-12 flex justify-center md:mt-15 xl:mt-20\",\"children\":[\"$\",\"$L7\",null,{\"href\":\"/blog\",\"target\":\"_blank\",\"className\":\"button relative inline-flex items-center justify-center h-11 transition-colors hover:text-color-2 px-7 text-n-1 \",\"children\":[[\"$\",\"span\",null,{\"className\":\"relative z-10\",\"children\":\"More Blogs\"}],[[\"$\",\"svg\",null,{\"className\":\"absolute left-0 top-0\",\"width\":\"21\",\"height\":\"44\",\"viewBox\":\"0 0 21 44\",\"children\":[[\"$\",\"svg\",null,{\"className\":\"block\",\"width\":0,\"height\":0,\"children\":[\"$\",\"defs\",null,{\"children\":[[\"$\",\"linearGradient\",null,{\"id\":\"btn-left\",\"x1\":\"50%\",\"x2\":\"50%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#89F9E8\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-top\",\"x1\":\"100%\",\"x2\":\"0%\",\"y1\":\"50%\",\"y2\":\"50%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#D87CEE\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-bottom\",\"x1\":\"100%\",\"x2\":\"0%\",\"y1\":\"50%\",\"y2\":\"50%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#89F9E8\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-right\",\"x1\":\"14.635%\",\"x2\":\"14.635%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#D87CEE\"}]]}]]}]}],[\"$\",\"path\",null,{\"fill\":\"none\",\"stroke\":\"url(#btn-left)\",\"strokeWidth\":\"2\",\"d\":\"M21,43.00005 L8.11111,43.00005 C4.18375,43.00005 1,39.58105 1,35.36365 L1,8.63637 C1,4.41892 4.18375,1 8.11111,1 L21,1\"}]]}],[\"$\",\"svg\",null,{\"className\":\"absolute left-[1.3125rem] top-0 w-[calc(100%-2.625rem)]\",\"height\":\"44\",\"viewBox\":\"0 0 100 44\",\"preserveAspectRatio\":\"none\",\"fill\":\"none\",\"children\":[[\"$\",\"svg\",null,{\"className\":\"block\",\"width\":0,\"height\":0,\"children\":[\"$\",\"defs\",null,{\"children\":[[\"$\",\"linearGradient\",null,{\"id\":\"btn-left\",\"x1\":\"50%\",\"x2\":\"50%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#89F9E8\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-top\",\"x1\":\"100%\",\"x2\":\"0%\",\"y1\":\"50%\",\"y2\":\"50%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#D87CEE\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-bottom\",\"x1\":\"100%\",\"x2\":\"0%\",\"y1\":\"50%\",\"y2\":\"50%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#89F9E8\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-right\",\"x1\":\"14.635%\",\"x2\":\"14.635%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#D87CEE\"}]]}]]}]}],[[\"$\",\"polygon\",null,{\"fill\":\"url(#btn-top)\",\"fillRule\":\"nonzero\",\"points\":\"100 42 100 44 0 44 0 42\"}],[\"$\",\"polygon\",null,{\"fill\":\"url(#btn-bottom)\",\"fillRule\":\"nonzero\",\"points\":\"100 0 100 2 0 2 0 0\"}]]]}],[\"$\",\"svg\",null,{\"className\":\"absolute right-0 top-0\",\"width\":\"21\",\"height\"
1:\"44\",\"viewBox\":\"0 0 21 44\",\"children\":[[\"$\",\"svg\",null,{\"className\":\"block\",\"width\":0,\"height\":0,\"children\":[\"$\",\"defs\",null,{\"children\":[[\"$\",\"linearGradient\",null,{\"id\":\"btn-left\",\"x1\":\"50%\",\"x2\":\"50%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#89F9E8\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-top\",\"x1\":\"100%\",\"x2\":\"0%\",\"y1\":\"50%\",\"y2\":\"50%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#D87CEE\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#FACB7B\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-bottom\",\"x1\":\"100%\",\"x2\":\"0%\",\"y1\":\"50%\",\"y2\":\"50%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#89F9E8\"}]]}],[\"$\",\"linearGradient\",null,{\"id\":\"btn-right\",\"x1\":\"14.635%\",\"x2\":\"14.635%\",\"y1\":\"0%\",\"y2\":\"100%\",\"children\":[[\"$\",\"stop\",null,{\"offset\":\"0%\",\"stopColor\":\"#9099FC\"}],[\"$\",\"stop\",null,{\"offset\":\"100%\",\"stopColor\":\"#D87CEE\"}]]}]]}]}],[\"$\",\"path\",null,{\"fill\":\"none\",\"stroke\":\"url(#btn-right)\",\"strokeWidth\":\"2\",\"d\":\"M0,43.00005 L5.028,43.00005 L12.24,43.00005 C16.526,43.00005 20,39.58105 20,35.36365 L20,16.85855 C20,14.59295 18.978,12.44425 17.209,10.99335 L7.187,2.77111 C5.792,1.62675 4.034,1 2.217,1 L0,1\"}]]}]]]}]}]]}],[\"$\",\"$L2c\",null,{}]],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10\"}],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10\"}]]}],[\"$\",\"div\",null,{\"id\":\"$undefined\",\"className\":\"relative py-8 lg:py-10 xl:py-12 relative z-10\",\"children\":[[[\"$\",\"div\",null,{\"className\":\"container flex flex-col items-center text-center md:flex-row md:text-left\",\"children\":[[\"$\",\"div\",null,{\"className\":\"flex-1\",\"children\":[\"$\",\"h2\",null,{\"className\":\"h2\",\"children\":\"Startup Partners\"}]}],[\"$\",\"div\",null,{\"className\":\"\",\"children\":[\"$\",\"ul\",null,{\"className\":\"flex flex-wrap gap-6 md:gap-x-8\",\"children\":[[\"$\",\"li\",\"0\",{\"className\":\"flex h-[8.5rem] flex-1 items-center justify-center\",\"children\":[\"$\",\"div\",null,{\"className\":\"flex flex-col items-center\",\"children\":[[\"$\",\"$L38\",null,{\"src\":{\"src\":\"/_next/static/media/microsoft.ab117229.png\",\"height\":149,\"width\":500,\"blurDataURL\":\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAgAAAACCAMAAABSSm3fAAAAGFBMVEX////////////+/v7////////////////QSSPTAAAACHRSTlMoNMEHBgR4dUwnOk8AAAAJcEhZcwAACxMAAAsTAQCanBgAAAAaSURBVHicY2BiZ2VkYGBkZWBiY2FkYGRgAAABlAAkPG+pkgAAAABJRU5ErkJggg==\",\"blurWidth\":8,\"blurHeight\":2},\"width\":200,\"height\":50,\"alt\":\"company logo\"}],false]}]}],[\"$\",\"li\",\"1\",{\"className\":\"flex h-[8.5rem] flex-1 items-center justify-center\",\"children\":[\"$\",\"div\",null,{\"className\":\"flex flex-col items-center\",\"children\":[[\"$\",\"$L38\",null,{\"src\":{\"src\":\"/_next/static/media/meta-logo.4e64bc4a.png\",\"height\":312,\"width\":1549,\"blurDataURL\":\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAgAAAACCAMAAABSSm3fAAAAFVBMVEX////////////////////////+/v4IluoxAAAAB3RSTlNwY0xWkUE/59wjDwAAAAlwSFlzAAALEwAACxMBAJqcGAAAABpJREFUeJxjYGFiZmRgZWRjYGJgZmJkYWAAAAFwACOH8fPAAAAAAElFTkSuQmCC\",\"blurWidth\":8,\"blurHeight\":2},\"width\":200,\"height\":50,\"alt\":\"Meta\"}],[\"$\",\"p\",null,{\"className\":\"mt-2 text-center text-sm text-n-3\",\"children\":\"Llama impact grant 2024\"}]]}]}]]}]}]]}],[\"$\",\"$L2c\",null,{}]],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10\"}],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10\"}]]}],[\"$\",\"$L3a\",null,{}],[\"$\",\"div\",null,{\"id\":\"$undefined\",\"className\":\"relative py-8 lg:py-10 xl:py-12 pb-20 pt-12\",\"children\":[[[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mb-6 mt-12 text-center\",\"children\":[[\"$\",\"h2\",null,{\"className\":\"mb-4 text-center text-4xl font-bold text-n-1 lg:text-5xl\",\"children\":\"Opportunities\"}],[\"$\",\"p\",null,{\"className\":\"mx-auto max-w-2xl text-center text-lg text-n-3\",\"children\":\"Join our team or support our research\"}]]}],[\"$\",\"div\",null,{\"className\":\"mx-auto max-w-7xl\",\"children\":[\"$\",\"div\",null,{\"className\":\"grid gap-8 md:grid-cols-2\",\"children\":[[\"$\",\"div\",null,{\"className\":\"relative rounded-2xl border border-n-6/20 bg-n-8 p-6 md:p-8\",\"children\":[\"$\",\"div\",null,{\"className\":\"relative z-10\",\"children\":[[\"$\",\"h3\",null,{\"className\":\"mb-4 text-2xl font-bold text-n-1\",\"children\":\"Join our Team\"}],[\"$\",\"div\",null,{\"className\":\"mb-6 rounded-xl border border-n-6 bg-n-7/50 p-5\",\"children\":[[\"$\",\"h4\",null,{\"className\":\"mb-2 text-lg font-semibold text-n-1\",\"children\":\"Application Process\"}],[\"$\",\"p\",null,{\"className\":\"text-n-3\",\"children\":\"Please fill out the form below to show interest in our open positions. We will review your application and get back to you within 2-3 weeks.\"}]]}],[\"$\",\"div\",null,{\"className\":\"h-12 w-full rounded-md bg-gradient-to-r from-purple-500 via-blue-400 to-teal-300 p-[2px] hover:from-purple-600 hover:via-blue-500 hover:to-teal-400\",\"children\":[\"$\",\"a\",null,{\"href\":\"https://forms.gle/YXRGGGmc3N2bGq7m9\",\"target\":\"_blank\",\"className\":\"inline-flex items-center justify-center whitespace-nowrap transition-colors focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-ring disabled:pointer-events-none disabled:opacity-50 shadow px-4 py-2 h-11 w-full rounded-md bg-background text-base font-semibold text-white hover:bg-black block w-full text-center\",\"children\":\"Apply Now\"}]}]]}]}],[\"$\",\"div\",null,{\"className\":\"relative rounded-2xl border border-n-6/20 bg-n-8 p-6 md:p-8\",\"children\":[[\"$\",\"div\",null,{\"className\":\"absolute right-8 top-8\",\"children\":[\"$\",\"svg\",null,{\"xmlns\":\"http://www.w3.org/2000/svg\",\"width\":24,\"height\"
1:24,\"viewBox\":\"0 0 24 24\",\"fill\":\"none\",\"stroke\":\"currentColor\",\"strokeWidth\":2,\"strokeLinecap\":\"round\",\"strokeLinejoin\":\"round\",\"className\":\"lucide lucide-sparkles h-12 w-12 text-color-1\",\"children\":[[\"$\",\"path\",\"4pj2yx\",{\"d\":\"M9.937 15.5A2 2 0 0 0 8.5 14.063l-6.135-1.582a.5.5 0 0 1 0-.962L8.5 9.936A2 2 0 0 0 9.937 8.5l1.582-6.135a.5.5 0 0 1 .963 0L14.063 8.5A2 2 0 0 0 15.5 9.937l6.135 1.581a.5.5 0 0 1 0 .964L15.5 14.063a2 2 0 0 0-1.437 1.437l-1.582 6.135a.5.5 0 0 1-.963 0z\"}],[\"$\",\"path\",\"1olli1\",{\"d\":\"M20 3v4\"}],[\"$\",\"path\",\"1gvqau\",{\"d\":\"M22 5h-4\"}],[\"$\",\"path\",\"vumght\",{\"d\":\"M4 17v2\"}],[\"$\",\"path\",\"zchphs\",{\"d\":\"M5 18H3\"}],\"$undefined\"]}]}],[\"$\",\"div\",null,{\"className\":\"relative z-10\",\"children\":[[\"$\",\"h3\",null,{\"className\":\"mb-4 text-2xl font-bold text-n-1\",\"children\":\"Support Our Research\"}],[\"$\",\"p\",null,{\"className\":\"mb-6 text-n-3\",\"children\":\"If you like our research and would like to sponsor our projects and open source initiatives, please get in touch. Your sponsorship will greatly help us continue developing innovative solutions and advancing the field of AI.\"}],[\"$\",\"ul\",null,{\"className\":\"mb-6 space-y-3\",\"children\":[[\"$\",\"li\",null,{\"className\":\"flex items-start\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mr-3 flex h-6 w-6 items-center justify-center rounded-full bg-linear-gradient text-xs text-n-8\",\"children\":\"â\"}],[\"$\",\"span\",null,{\"className\":\"text-n-3\",\"children\":\"Support cutting-edge AI research\"}]]}],[\"$\",\"li\",null,{\"className\":\"flex items-start\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mr-3 flex h-6 w-6 items-center justify-center rounded-full bg-linear-gradient text-xs text-n-8\",\"children\":\"â\"}],[\"$\",\"span\",null,{\"className\":\"text-n-3\",\"children\":\"Contribute to open source development\"}]]}],[\"$\",\"li\",null,{\"className\":\"flex items-start\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mr-3 flex h-6 w-6 items-center justify-center rounded-full bg-linear-gradient text-xs text-n-8\",\"children\":\"â\"}],[\"$\",\"span\",null,{\"className\":\"text-n-3\",\"children\":\"Help make AI accessible to everyone\"}]]}]]}],[\"$\",\"div\",null,{\"className\":\"h-12 w-full rounded-md bg-gradient-to-r from-purple-500 via-blue-400 to-teal-300 p-[2px] hover:from-purple-600 hover:via-blue-500 hover:to-teal-400\",\"children\":[\"$\",\"a\",null,{\"href\":\"mailto:[email protected]\",\"className\":\"inline-flex items-center justify-center whitespace-nowrap transition-colors focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-ring disabled:pointer-events-none disabled:opacity-50 shadow px-4 py-2 h-11 w-full rounded-md bg-background text-base font-semibold text-white hover:bg-black block w-full text-center\",\"children\":\"Contact Research Team\"}]}]]}]]}]]}]}]]}],[\"$\",\"$L2c\",null,{}]],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute left-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:left-7.5 xl:left-10\"}],[\"$\",\"div\",null,{\"className\":\"pointer-events-none absolute right-5 top-0 hidden h-full w-0.25 bg-stroke-1 md:block lg:right-7.5 xl:right-10\"}]]}]]}],\"notFoundStyles\":[],\"styles\":null}]}],[\"$\",\"$L3b\",null,{}],null]}]}],[\"$\",\"$L3c\",null,{\"src\":\"https://plausible.io/js/script.file-downloads.outbound-links.js\",\"data-domain\":\"cognitivelab.in\"}],[\"$\",\"$L3c\",null,{\"id\":\"plausible-setup\",\"children\":\"\\n window.plausible = window.plausible || function() { \\n (window.plausible.q = window.plausible.q || []).push(arguments) \\n }\\n \"}]]}]}],null],null],\"couldBeIntercepted\":false,\"initialHead\":[false,\"$L3d\"],\"globalErrorComponent\":\"$3e\",\"missingSlots\":\"$W3f\"}]]\n"])</script>
1<script>self.__next_f.push([1,"3d:[[\"$\",\"meta\",\"0\",{\"name\":\"viewport\",\"content\":\"width=device-width, initial-scale=1\"}],[\"$\",\"meta\",\"1\",{\"charSet\":\"utf-8\"}],[\"$\",\"title\",\"2\",{\"children\":\"Blog | CognitiveLab\"}],[\"$\",\"meta\",\"3\",{\"name\":\"description\",\"content\":\"Latest articles, announcements and insights from our team of AI researchers and engineers\"}],[\"$\",\"meta\",\"4\",{\"property\":\"og:title\",\"content\":\"Blog | CognitiveLab\"}],[\"$\",\"meta\",\"5\",{\"property\":\"og:description\",\"content\":\"Latest articles, announcements and insights from our team of AI researchers and engineers\"}],[\"$\",\"meta\",\"6\",{\"property\":\"og:image\",\"content\":\"https://cognitivelab.in/assets/cognitivelab-og.png\"}],[\"$\",\"meta\",\"7\",{\"property\":\"og:image:width\",\"content\":\"1200\"}],[\"$\",\"meta\",\"8\",{\"property\":\"og:image:height\",\"content\":\"630\"}],[\"$\",\"meta\",\"9\",{\"property\":\"og:image:alt\",\"content\":\"CognitiveLab Blog\"}],[\"$\",\"meta\",\"10\",{\"name\":\"twitter:card\",\"content\":\"summary_large_image\"}],[\"$\",\"meta\",\"11\",{\"name\":\"twitter:title\",\"content\":\"Blog | CognitiveLab\"}],[\"$\",\"meta\",\"12\",{\"name\":\"twitter:description\",\"content\":\"Latest articles, announcements and insights from our team of AI researchers and engineers\"}],[\"$\",\"meta\",\"13\",{\"name\":\"twitter:image\",\"content\":\"https://cognitivelab.in/assets/cognitivelab-og.png\"}],[\"$\",\"link\",\"14\",{\"rel\":\"icon\",\"href\":\"/favicon.ico\",\"type\":\"image/x-icon\",\"sizes\":\"32x32\"}],[\"$\",\"meta\",\"15\",{\"name\":\"next-size-adjust\"}]]\n5:null\n"])</script>
1</body></html>
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.