PageSourceSearch

https://srpo.pages.dev/docs

html srpo.pages.dev collected 2026-10-03 05:58:27 UTC 25,700 bytes, 38 lines download raw bytes

1<!DOCTYPE html><html lang="en"><head><meta charSet="utf-8"/><meta name="viewport" content="width=device-width, initial-scale=1"/><link rel="preload" href="/_next/static/media/4cf2300e9c8272f7-s.p.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/media/78b756c5b9139e81-s.p.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/media/93f479601ee12b01-s.p.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/media/c4a2ca76cbcd952a-s.p.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="stylesheet" href="/_next/static/css/9e74eb8a61f65195.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/css/f055e431c58b907b.css" data-precedence="next"/><link rel="preload" as="script" fetchPriority="low" href="/_next/static/chunks/webpack-53f211bc9db111b9.js"/>
1<script src="/_next/static/chunks/4bd1b696-7b9071663a568a36.js" async=""></script>
1<script src="/_next/static/chunks/684-dc321eb11436ca36.js" async=""></script>
1<script src="/_next/static/chunks/main-app-037964b5d318d5f2.js" async=""></script>
1<script src="/_next/static/chunks/ee560e2c-c57569c2ac07079a.js" async=""></script>
1<script src="/_next/static/chunks/874-8c86f80752a840ac.js" async=""></script>
1<script src="/_next/static/chunks/app/layout-b4b4900cfb9381eb.js" async=""></script>
1<script src="/_next/static/chunks/app/not-found-13c945b8408818c2.js" async=""></script>
1<script src="/_next/static/chunks/693-7db0599a61dd7fa4.js" async=""></script>
1<script src="/_next/static/chunks/app/docs/page-3ad26dea2c732879.js" async=""></script>
1<meta name="next-size-adjust" content=""/><meta name="google-site-verification" content="5t-K1NUCKrtJ4ulTsBbFSlziOYxSdDzBByT5TeO6TI0"/><link rel="icon" href="/favicon.ico" type="image/x-icon"/><title>Docs | SRPO</title><meta name="description" content="Documentation for the SRPO project"/><meta name="keywords" content="multimodal reasoning,NEURIPS 2025,SRPO,SRPO framework,multimodal LLM,multimodal large language model reasoning,reflection-aware reinforcement learning,LLM with reflection mechanism,reasoning in multimodal AI,AI reasoning enhancement,deep learning for multimodal training,RL fine-tuning for language models,AI research paper,LLM architecture improvement,vision-language model with RL,how to improve multimodal LLM reasoning"/><link rel="canonical" href="https://srpo.pages.dev"/><meta property="og:title" content="SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware RL"/><meta property="og:description" content="A novel framework that enhances the reasoning capabilities of multimodal large language models"/><meta property="og:url" content="https://srpo.pages.dev"/><meta property="og:site_name" content="SRPO"/><meta property="og:image" content="https://srpo.pages.dev/og_image_motivation.png"/><meta property="og:image:width" content="1200"/><meta property="og:image:height" content="630"/><meta property="og:image:alt" content="SRPO - A novel framework that enhances the reasoning capabilities of multimodal large language models"/><meta name="twitter:card" content="summary_large_image"/><meta name="twitter:title" content="SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware RL"/><meta name="twitter:description" content="A novel framework that enhances the reasoning capabilities of multimodal large language models"/><meta name="twitter:image" content="https://srpo.pages.dev/og_image_motivation.png"/><meta name="twitter:image:width" content="1200"/><meta name="twitter:image:height" content="630"/><meta name="twitter:image:alt" content="SRPO - A novel framework that enhances the reasoning capabilities of multimodal large language models"/><link rel="icon" href="/favicon.ico" type="image/x-icon" sizes="188x256"/>
1<script>document.querySelectorAll('body link[rel="icon"], body link[rel="apple-touch-icon"]').forEach(el => document.head.appendChild(el))</script>
1<script src="/_next/static/chunks/polyfills-42372ed130431b0a.js" noModule=""></script>
1</head><body class="font-roboto __variable_188709 __variable_c5376c __variable_9a8899 __variable_10f679 antialiased"><nav class="nav-container fixed top-0 left-0 w-full z-50 transition-all duration-300 ease-in-out translate-y-0 bg-white/90 backdrop-blur-sm"><div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8"><div class="flex items-center justify-between h-16"><div class="md:hidden flex items-center"><button class="text-gray-700 hover:text-gray-900 focus:outline-none"><svg stroke="currentColor" fill="none" stroke-width="2" viewBox="0 0 24 24" stroke-linecap="round" stroke-linejoin="round" class="h-6 w-6" height="1em" width="1em" xmlns="http://www.w3.org/2000/svg"><line x1="3" y1="12" x2="21" y2="12"></line><line x1="3" y1="6" x2="21" y2="6"></line><line x1="3" y1="18" x2="21" y2="18"></line></svg></button></div><div class="hidden md:flex items-center space-x-8"><div class="relative group"><a class="text-gray-700 px-3 py-2 rounded-md text-sm font-medium transition-colors duration-200 flex items-center hover:text-gray-900" href="/">Paper<svg class="ml-1 h-4 w-4 text-gray-500 group-hover:text-gray-700 transition-transform duration-200" fill="none" viewBox="0 0 24 24" stroke="currentColor"><path stroke-linecap="round" stroke-linejoin="round" stroke-width="2" d="M19 9l-7 7-7-7"></path></svg></a><div class="absolute left-0 mt-0 bg-white border border-gray-200 rounded-lg shadow-lg py-1 z-50 min-w-[180px] transition-all duration-200 origin-top opacity-0 scale-95 pointer-events-none"><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/#abstraction">Abstract</a><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/#algorithm">Algorithm</a><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/#experiment">Experiment</a><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/#conclusion">Conclusion</a><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/#citeus">Cite Us</a></div></div><div class="relative group"><a class="text-gray-700 px-3 py-2 rounded-md text-sm font-medium transition-colors duration-200 flex items-center bg-blue-300 text-gray-900" href="/docs">Docs<svg class="ml-1 h-4 w-4 text-gray-500 group-hover:text-gray-700 transition-transform duration-200" fill="none" viewBox="0 0 24 24" stroke="currentColor"><path stroke-linecap="round" stroke-linejoin="round" stroke-width="2" d="M19 9l-7 7-7-7"></path></svg></a><div class="absolute left-0 mt-0 bg-white border border-gray-200 rounded-lg shadow-lg py-1 z-50 min-w-[180px] transition-all duration-200 origin-top opacity-0 scale-95 pointer-events-none"><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/docs/quick-start">Quick Start</a><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/docs/llm-sft">LLM-SFT</a><a class="block px-4 py-2 text-sm text-gray-700 hover:bg-gray-50 transition-colors duration-150" href="/docs/llm">LLM</a></div></div><div class="relative group"><a class="text-gray-700 px-3 py-2 rounded-md text-sm font-medium transition-colors duration-200 flex items-center hover:text-gray-900" href="/updates">Updates</a></div></div></div></div><div class="md:hidden hidden"><div class="px-2 pt-2 pb-3 space-y-1 sm:px-3"><div class="relative"><div class="flex items-center"><a class="flex-1 text-left px-3 py-2 rounded-md text-base font-medium text-gray-700 hover:text-gray-900 hover:bg-gray-50" href="/">Paper</a><button class="ml-2 p-1 text-gray-500 hover:text-gray-700 focus:outline-none" aria-label="Toggle submenu" type="button"><svg class="h-4 w-4 transition-transform duration-200 " fill="none" viewBox="0 0 24 24" stroke="currentColor"><path stroke-linecap="round" stroke-linejoin="round" stroke-width="2" d="M19 9l-7 7-7-7"></path></svg></button></div></div><div class="relative"><div class="flex items-center"><a class="flex-1 text-left px-3 py-2 rounded-md text-base font-medium bg-gray-200 text-gray-900" href="/docs">Docs</a><button class="ml-2 p-1 text-gray-500 hover:text-gray-700 focus:outline-none" aria-label="Toggle submenu" type="button"><svg class="h-4 w-4 transition-transform duration-200 " fill="none" viewBox="0 0 24 24" stroke="currentColor"><path stroke-linecap="round" stroke-linejoin="round" stroke-width="2" d="M19 9l-7 7-7-7"></path></svg></button></div></div><div class="relative"><div class="flex items-center"><a class="flex-1 text-left px-3 py-2 rounded-md text-base font-medium text-gray-700 hover:text-gray-900 hover:bg-gray-50" href="/updates">
1Updates</a></div></div></div></div></nav><div class="prose prose-zinc max-w-3xl mx-auto py-12"><h1 class="text-3xl sm:text-4xl font-bold text-zinc-900 mb-10 mt-6">SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware Reinforcement Learning</h1>
2<p class="mb-4 max-w-3/4">Welcome to the documentation for <strong>SRPO</strong>, a novel framework designed to enhance the reasoning capabilities of multimodal large language models (MLLMs) through reflection-aware reinforcement learning.</p>
3<h2 id="🚀-project-overview" class="text-2xl sm:text-3xl font-semibold mt-6 mb-3">🚀 Project Overview</h2>
4<p class="mb-4 max-w-3/4">SRPO introduces a reflection-aware RL pipeline that enables MLLMs to self-reflect, critique, and iteratively improve their reasoning on complex multimodal tasks.</p>
5<ul class="list-disc pl-5 mb-4">
6<li><strong>Paper:</strong> <a href="https://arxiv.org/abs/2506.01713" target="_blank" rel="noopener noreferrer" class="text-blue-600 hover:underline">Arxiv (coming soon)</a></li>
7<li><strong>Model:</strong> <a href="https://huggingface.co/SRPOMLLMs" target="_blank" rel="noopener noreferrer" class="text-blue-600 hover:underline">Hugging Face Models</a></li>
8<li><strong>Dataset:</strong> <a href="https://huggingface.co/datasets/SRPOMLLMs/srpo-sft-data" target="_blank" rel="noopener noreferrer" class="text-blue-600 hover:underline">Hugging Face Dataset</a></li>
9<li><strong>Project Page:</strong> <a href="https://srpo.pages.dev" target="_blank" rel="noopener noreferrer" class="text-blue-600 hover:underline">srpo.pages.dev</a></li>
10</ul>
11<h2 id="✨-key-features" class="text-2xl sm:text-3xl font-semibold mt-6 mb-3">✨ Key Features</h2>
12<ul class="list-disc pl-5 mb-4">
13<li><strong>Reflection-Driven Training:</strong> SRPO systematically generates high-quality, reflection-focused training data and employs a novel reward mechanism that explicitly incentivizes concise and effective self-reflection. This approach directly addresses the limitations of previous methods, such as insufficient data quality and a lack of self-reflective behavior for refining responses.</li>
14<li><strong>Enhanced Multimodal Reasoning:</strong> By encouraging models to self-reflect and critique their own outputs, SRPO significantly improves the reasoning capabilities of multimodal large language models.</li>
15<li><strong>State-of-the-Art Performance:</strong> Comprehensive experiments across multiple multimodal reasoning benchmarks demonstrate that SRPO surpasses existing state-of-the-art models in both reasoning accuracy and reflection quality.</li>
16<li><strong>Robustness Through Reflection:</strong> Our results highlight the critical role of reflection-driven training strategies in achieving robust and reliable multimodal reasoning.</li>
17</ul>
18<h2 id="📦-repository-structure" class="text-2xl sm:text-3xl font-semibold mt-6 mb-3">📦 Repository Structure</h2>
19<ul class="list-disc pl-5 mb-4">
20<li>See the <a href="/docs/llm" target="_blank" rel="noopener noreferrer" class="text-blue-600 hover:underline"><code class="rounded-md bg-neutral-100 px-1.5 py-0.5 text-sm font-mono text-neutral-800 border border-neutral-200">llm_engine</code></a> for unified LLM SDK (OpenAI, Azure, VLLM, etc.)</li>
21<li>See the <a href="/docs/llm-sft" target="_blank" rel="noopener noreferrer" class="text-blue-600 hover:underline"><code class="rounded-md bg-neutral-100 px-1.5 py-0.5 text-sm font-mono text-neutral-800 border border-neutral-200">llm_sft</code></a> for scripts for answer evaluation, reflection evaluation, and image description extraction</li>
22</ul>
23<h2 id="🏁-quick-start" class="text-2xl sm:text-3xl font-semibold mt-6 mb-3">🏁 Quick Start</h2>
24<p class="mb-4 max-w-3/4">See the <a href="/docs/quick-start" target="_blank" rel="noopener noreferrer" class="text-blue-600 hover:underline">Quick Start guide</a> for installation, data preparation, and running evaluation/reflection.</p>
25<h2 id="📄-citation" class="text-2xl sm:text-3xl font-semibold mt-6 mb-3">📄 Citation</h2>
26<p class="mb-4 max-w-3/4">If you use SRPO or this codebase, please cite our paper:</p>
27<pre><div class="relative my-6 group"><div class="absolute -top-3 left-4 px-3 py-1 bg-white text-neutral-600 text-xs font-medium rounded-t-lg border border-neutral-200 shadow-sm z-10">bibtex</div><button class="absolute right-4 top-4 z-10 rounded-lg bg-white/80 hover:bg-neutral-100 p-2 transition-all duration-200 opacity-0 group-hover:opacity-100 backdrop-blur-sm shadow-sm border border-neutral-200" aria-label="Copy code" title="Copy to clipboard"><svg xmlns="http://www.w3.org/2000/svg" width="16" height="16" viewB
27ox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-copy text-neutral-600 hover:text-neutral-900" aria-hidden="true"><rect width="14" height="14" x="8" y="8" rx="2" ry="2"></rect><path d="M4 16c-1.1 0-2-.9-2-2V4c0-1.1.9-2 2-2h10c1.1 0 2 .9 2 2"></path></svg></button><div class="relative overflow-hidden rounded-lg bg-white border border-neutral-200 shadow-sm"><div class="absolute inset-0 bg-gradient-to-b from-neutral-200/30 to-transparent opacity-50 pointer-events-none"></div><pre class="overflow-x-auto p-5 text-sm leading-relaxed scrollbar-thin scrollbar-thumb-neutral-300 scrollbar-track-neutral-100"><code class="hljs language-bibtex font-mono text-neutral-800">@misc{wan2025srpoenhancingmultimodalllm,
28      title={SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware Reinforcement Learning}, 
29      author={Zhongwei Wan and Zhihao Dou and Che Liu and Yu Zhang and Dongfei Cui and Qinjian Zhao and Hui Shen and Jing Xiong and Yi Xin and Yifan Jiang and Yangfan He and Mi Zhang and Shen Yan},
30      year={2025},
31      eprint={2506.01713},
32      archivePrefix={arXiv},
33      primaryClass={cs.CL},
34      url={https://arxiv.org/abs/2506.01713}, 
35}
36</code></pre></div></div></pre>
37<hr/>
38<p class="mb-4 max-w-3/4">For more details, explore the linked documentation pages or the codebase.</p></div><!--$--><!--/$--><!--$--><!--/$-->
38<script src="/_next/static/chunks/webpack-53f211bc9db111b9.js" async=""></script>
38<script>(self.__next_f=self.__next_f||[]).push([0])</script>
38<script>self.__next_f.push([1,"1:\"$Sreact.fragment\"\n2:I[7119,[\"844\",\"static/chunks/ee560e2c-c57569c2ac07079a.js\",\"874\",\"static/chunks/874-8c86f80752a840ac.js\",\"177\",\"static/chunks/app/layout-b4b4900cfb9381eb.js\"],\"default\"]\n3:I[7555,[],\"\"]\n4:I[1295,[],\"\"]\n5:I[6874,[\"874\",\"static/chunks/874-8c86f80752a840ac.js\",\"345\",\"static/chunks/app/not-found-13c945b8408818c2.js\"],\"\"]\n6:I[7922,[\"693\",\"static/chunks/693-7db0599a61dd7fa4.js\",\"40\",\"static/chunks/app/docs/page-3ad26dea2c732879.js\"],\"default\"]\n8:I[9665,[],\"MetadataBoundary\"]\na:I[9665,[],\"OutletBoundary\"]\nd:I[4911,[],\"AsyncMetadataOutlet\"]\nf:I[9665,[],\"ViewportBoundary\"]\n11:I[6614,[],\"\"]\n:HL[\"/_next/static/media/4cf2300e9c8272f7-s.p.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/media/78b756c5b9139e81-s.p.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/media/93f479601ee12b01-s.p.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/media/c4a2ca76cbcd952a-s.p.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/css/9e74eb8a61f65195.css\",\"style\"]\n:HL[\"/_next/static/css/f055e431c58b907b.css\",\"style\"]\n7:Tb02,"])</script>
38<script>self.__next_f.push([1,"# SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware Reinforcement Learning\n\nWelcome to the documentation for **SRPO**, a novel framework designed to enhance the reasoning capabilities of multimodal large language models (MLLMs) through reflection-aware reinforcement learning.\n\n## 🚀 Project Overview\nSRPO introduces a reflection-aware RL pipeline that enables MLLMs to self-reflect, critique, and iteratively improve their reasoning on complex multimodal tasks.\n\n- **Paper:** [Arxiv (coming soon)](https://arxiv.org/abs/2506.01713)\n- **Model:** [Hugging Face Models](https://huggingface.co/SRPOMLLMs)\n- **Dataset:** [Hugging Face Dataset](https://huggingface.co/datasets/SRPOMLLMs/srpo-sft-data)\n- **Project Page:** [srpo.pages.dev](https://srpo.pages.dev)\n\n## ✨ Key Features\n- **Reflection-Driven Training:** SRPO systematically generates high-quality, reflection-focused training data and employs a novel reward mechanism that explicitly incentivizes concise and effective self-reflection. This approach directly addresses the limitations of previous methods, such as insufficient data quality and a lack of self-reflective behavior for refining responses.\n- **Enhanced Multimodal Reasoning:** By encouraging models to self-reflect and critique their own outputs, SRPO significantly improves the reasoning capabilities of multimodal large language models.\n- **State-of-the-Art Performance:** Comprehensive experiments across multiple multimodal reasoning benchmarks demonstrate that SRPO surpasses existing state-of-the-art models in both reasoning accuracy and reflection quality.\n- **Robustness Through Reflection:** Our results highlight the critical role of reflection-driven training strategies in achieving robust and reliable multimodal reasoning.\n\n## 📦 Repository Structure\n- See the [`llm_engine`](/docs/llm) for unified LLM SDK (OpenAI, Azure, VLLM, etc.)\n- See the [`llm_sft`](/docs/llm-sft) for scripts for answer evaluation, reflection evaluation, and image description extraction\n\n## 🏁 Quick Start\nSee the [Quick Start guide](/docs/quick-start) for installation, data preparation, and running evaluation/reflection.\n\n## 📄 Citation\nIf you use SRPO or this codebase, please cite our paper:\n\n```bibtex\n@misc{wan2025srpoenhancingmultimodalllm,\n      title={SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware Reinforcement Learning}, \n      author={Zhongwei Wan and Zhihao Dou and Che Liu and Yu Zhang and Dongfei Cui and Qinjian Zhao and Hui Shen and Jing Xiong and Yi Xin and Yifan Jiang and Yangfan He and Mi Zhang and Shen Yan},\n      year={2025},\n      eprint={2506.01713},\n      archivePrefix={arXiv},\n      primaryClass={cs.CL},\n      url={https://arxiv.org/abs/2506.01713}, \n}\n```\n\n---\n\nFor more details, explore the linked documentation pages or the codebase.\n\n"])</script>
38<script>self.__next_f.push([1,"0:{\"P\":null,\"b\":\"Ij0dvvGjOARIE4tHIt21Z\",\"p\":\"\",\"c\":[\"\",\"docs\"],\"i\":false,\"f\":[[[\"\",{\"children\":[\"docs\",{\"children\":[\"__PAGE__\",{}]}]},\"$undefined\",\"$undefined\",true],[\"\",[\"$\",\"$1\",\"c\",{\"children\":[[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/css/9e74eb8a61f65195.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}]],[\"$\",\"html\",null,{\"lang\":\"en\",\"children\":[[\"$\",\"head\",null,{\"children\":[[\"$\",\"meta\",null,{\"name\":\"google-site-verification\",\"content\":\"5t-K1NUCKrtJ4ulTsBbFSlziOYxSdDzBByT5TeO6TI0\"}],[\"$\",\"link\",null,{\"rel\":\"icon\",\"href\":\"/favicon.ico\",\"type\":\"image/x-icon\"}]]}],[\"$\",\"body\",null,{\"className\":\"font-roboto __variable_188709 __variable_c5376c __variable_9a8899 __variable_10f679 antialiased\",\"children\":[[\"$\",\"$L2\",null,{}],[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":[[\"$\",\"div\",null,{\"className\":\"flex flex-col items-center justify-center min-h-screen bg-white text-gray-800\",\"children\":[[\"$\",\"h1\",null,{\"className\":\"text-4xl font-bold mb-4\",\"children\":\"404 - Page Not Found\"}],[\"$\",\"p\",null,{\"className\":\"text-lg mb-6\",\"children\":\"The page you are looking for does not exist.\"}],[\"$\",\"$L5\",null,{\"href\":\"/\",\"className\":\"text-blue-500 hover:underline\",\"children\":\"Go back to home\"}]]}],[]],\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]]}]]}],{\"children\":[\"docs\",[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}],{\"children\":[\"__PAGE__\",[\"$\",\"$1\",\"c\",{\"children\":[[\"$\",\"div\",null,{\"className\":\"prose prose-zinc max-w-3xl mx-auto py-12\",\"children\":[\"$\",\"$L6\",null,{\"content\":\"$7\"}]}],[\"$\",\"$L8\",null,{\"children\":\"$L9\"}],[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/css/f055e431c58b907b.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}]],[\"$\",\"$La\",null,{\"children\":[\"$Lb\",\"$Lc\",[\"$\",\"$Ld\",null,{\"promise\":\"$@e\"}]]}]]}],{},null,false]},null,false]},null,false],[\"$\",\"$1\",\"h\",{\"children\":[null,[\"$\",\"$1\",\"66kBCJlj46Izxn3J_6jUc\",{\"children\":[[\"$\",\"$Lf\",null,{\"children\":\"$L10\"}],[\"$\",\"meta\",null,{\"name\":\"next-size-adjust\",\"content\":\"\"}]]}],null]}],false]],\"m\":\"$undefined\",\"G\":[\"$11\",\"$undefined\"],\"s\":false,\"S\":true}\n"])</script>
38<script>self.__next_f.push([1,"12:\"$Sreact.suspense\"\n13:I[4911,[],\"AsyncMetadata\"]\n9:[\"$\",\"$12\",null,{\"fallback\":null,\"children\":[\"$\",\"$L13\",null,{\"promise\":\"$@14\"}]}]\n"])</script>
38<script>self.__next_f.push([1,"c:null\n"])</script>
38<script>self.__next_f.push([1,"10:[[\"$\",\"meta\",\"0\",{\"charSet\":\"utf-8\"}],[\"$\",\"meta\",\"1\",{\"name\":\"viewport\",\"content\":\"width=device-width, initial-scale=1\"}]]\nb:null\n"])</script>
38<script>self.__next_f.push([1,"14:{\"metadata\":[[\"$\",\"title\",\"0\",{\"children\":\"Docs | SRPO\"}],[\"$\",\"meta\",\"1\",{\"name\":\"description\",\"content\":\"Documentation for the SRPO project\"}],[\"$\",\"meta\",\"2\",{\"name\":\"keywords\",\"content\":\"multimodal reasoning,NEURIPS 2025,SRPO,SRPO framework,multimodal LLM,multimodal large language model reasoning,reflection-aware reinforcement learning,LLM with reflection mechanism,reasoning in multimodal AI,AI reasoning enhancement,deep learning for multimodal training,RL fine-tuning for language models,AI research paper,LLM architecture improvement,vision-language model with RL,how to improve multimodal LLM reasoning\"}],[\"$\",\"link\",\"3\",{\"rel\":\"canonical\",\"href\":\"https://srpo.pages.dev\"}],[\"$\",\"meta\",\"4\",{\"property\":\"og:title\",\"content\":\"SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware RL\"}],[\"$\",\"meta\",\"5\",{\"property\":\"og:description\",\"content\":\"A novel framework that enhances the reasoning capabilities of multimodal large language models\"}],[\"$\",\"meta\",\"6\",{\"property\":\"og:url\",\"content\":\"https://srpo.pages.dev\"}],[\"$\",\"meta\",\"7\",{\"property\":\"og:site_name\",\"content\":\"SRPO\"}],[\"$\",\"meta\",\"8\",{\"property\":\"og:image\",\"content\":\"https://srpo.pages.dev/og_image_motivation.png\"}],[\"$\",\"meta\",\"9\",{\"property\":\"og:image:width\",\"content\":\"1200\"}],[\"$\",\"meta\",\"10\",{\"property\":\"og:image:height\",\"content\":\"630\"}],[\"$\",\"meta\",\"11\",{\"property\":\"og:image:alt\",\"content\":\"SRPO - A novel framework that enhances the reasoning capabilities of multimodal large language models\"}],[\"$\",\"meta\",\"12\",{\"name\":\"twitter:card\",\"content\":\"summary_large_image\"}],[\"$\",\"meta\",\"13\",{\"name\":\"twitter:title\",\"content\":\"SRPO: Enhancing Multimodal LLM Reasoning via Reflection-Aware RL\"}],[\"$\",\"meta\",\"14\",{\"name\":\"twitter:description\",\"content\":\"A novel framework that enhances the reasoning capabilities of multimodal large language models\"}],[\"$\",\"meta\",\"15\",{\"name\":\"twitter:image\",\"content\":\"https://srpo.pages.dev/og_image_motivation.png\"}],[\"$\",\"meta\",\"16\",{\"name\":\"twitter:image:width\",\"content\":\"1200\"}],[\"$\",\"meta\",\"17\",{\"name\":\"twitter:image:height\",\"content\":\"630\"}],[\"$\",\"meta\",\"18\",{\"name\":\"twitter:image:alt\",\"content\":\"SRPO - A novel framework that enhances the reasoning capabilities of multimodal large language models\"}],[\"$\",\"link\",\"19\",{\"rel\":\"icon\",\"href\":\"/favicon.ico\",\"type\":\"image/x-icon\",\"sizes\":\"188x256\"}]],\"error\":null,\"digest\":\"$undefined\"}\n"])</script>
38<script>self.__next_f.push([1,"e:{\"metadata\":\"$14:metadata\",\"error\":null,\"digest\":\"$undefined\"}\n"])</script>
38<!-- Cloudflare Pages Analytics -->
vendor: 99 bytes, line 38
38<script defer src='https://static.cloudflareinsights.com/beacon.min.js' data-cf-beacon='{"token": "
38a3e454e4a35c441db5967f385b510a9e
vendor: 13 bytes, line 38
38"}'></script>
38<!-- Cloudflare Pages Analytics --></body></html>

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.