1:"$Sreact.fragment"
2:I[6529,["619","static/chunks/619-ba102abea3e3d0e4.js","177","static/chunks/app/layout-13fa3fd02f6eb5db.js"],"default"]
3:I[9766,[],""]
4:I[8924,[],""]
5:I[2619,["619","static/chunks/619-ba102abea3e3d0e4.js","953","static/chunks/app/blog/%5Bslug%5D/page-f25a122e9ccf798d.js"],""]
d:I[7150,[],""]
:HL["/_next/static/css/1d1f6bc532e5f43f.css","style"]
:HL["/_next/static/css/eb87e4f7aea490c6.css","style"]
0:{"P":null,"b":"DQJo8iKQxHubJM4JvsRbH","p":"","c":["","blog","qvq-72b-preview",""],"i":false,"f":[[["",{"children":["blog",{"children":[["slug","qvq-72b-preview","d"],{"children":["__PAGE__",{}]}]}]},"$undefined","$undefined",true],["",["$","$1","c",{"children":[[["$","link","0",{"rel":"stylesheet","href":"/_next/static/css/1d1f6bc532e5f43f.css","precedence":"next","crossOrigin":"$undefined","nonce":"$undefined"}]],["$","html",null,{"lang":"en","children":[["$","head",null,{"children":[["$","link",null,{"rel":"icon","type":"image/svg+xml","href":"/favicon.svg"}],["$","link",null,{"rel":"alternate icon","href":"/favicon.png"}]]}],["$","body",null,{"children":[["$","$L2",null,{}],["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]],"forbidden":"$undefined","unauthorized":"$undefined"}],["$","footer",null,{"children":["$","div",null,{"className":"container","children":[["$","div",null,{"className":"footer-content","children":[["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen LM"}],["$","p",null,{"children":"95 open Zen models across Zen3, Zen4, and Zen5. Chat, code, vision, audio, image, embeddings, rerankers, and safety. OpenAI- and Anthropic-compatible API."}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 5"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Nano (0.8B - 9B)"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Flash"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Mini"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 (default)"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Coder"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Pro"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Max"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 4"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 / Zen4.1"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Ultra / Max / Pro"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Mini / Thinking"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Coder / Pro / Flash"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 3 Multimodal"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Omni / VL / Web"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Nano / Guard"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Embedding / Reranker"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Image / ASR / TTS"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Resources"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/datasets","children":"Training Data"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://huggingface.co/zenlm","target":"_blank","rel":"noopener noreferrer","children":"HuggingFace"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://github.com/zenlm","target":"_blank","rel":"noopener noreferrer","children":"GitHub"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/research","children":"Research Papers"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://api.hanzo.ai","target":"_blank","rel":"noopener noreferrer","children":"Zen API"}]}]]}]]}]]}],"$L6"]}]}],"$L7","$L8"]}]]}]]}],{"children":["blog","$L9",{"children":[["slug","qvq-72b-preview","d"],"$La",{"children":["__PAGE__","$Lb",{},null,false]},null,false]},null,false]},null,false],"$Lc",false]],"m":"$undefined","G":["$d",[]],"s":false,"S":true}
e:I[7405,["619","static/chunks/619-ba102abea3e3d0e4.js","177","static/chunks/app/layout-13fa3fd02f6eb5db.js"],"default"]
10:I[4431,[],"OutletBoundary"]
12:I[5278,[],"AsyncMetadataOutlet"]
14:I[4431,[],"ViewportBoundary"]
16:I[4431,[],"MetadataBoundary"]
17:"$Sreact.suspense"
6:["$","div",null,{"className":"footer-bottom","children":["$","p",null,{"children":["© ",2026," Zen Authors. Open foundation models. Served on the Zen API."]}]}]
7:["$","$Le",null,{}]
8:["$","script",null,{"src":"/assets/js/main.js","async":true}]
9:["$","$1","c",{"children":[null,["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","forbidden":"$undefined","unauthorized":"$undefined"}]]}]
a:["$","$1","c",{"children":[null,["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","forbidden":"$undefined","unauthorized":"$undefined"}]]}]
b:["$","$1","c",{"children":["$Lf",[["$","link","0",{"rel":"stylesheet","href":"/_next/static/css/eb87e4f7aea490c6.css","precedence":"next","crossOrigin":"$undefined","nonce":"$undefined"}]],["$","$L10",null,{"children":["$L11",["$","$L12",null,{"promise":"$@13"}]]}]]}]
c:["$","$1","h",{"children":[null,[["$","$L14",null,{"children":"$L15"}],null],["$","$L16",null,{"children":["$","div",null,{"hidden":true,"children":["$","$17",null,{"fallback":null,"children":"$L18"}]}]}]]}]
19:T147e,<p><a href="https://github.com/QwenLM/zen-VL">GITHUB</a>
<a href="https://huggingface.co/Qwen">HUGGING FACE</a>
<a href="https://modelscope.cn/organization/qwen">MODELSCOPE</a>
<a href="https://www.kaggle.com/models/qwen-lm/qvq-72b-preview">KAGGLE</a>
<a href="https://huggingface.co/spaces/Qwen/QVQ-72B-Preview">DEMO</a>
<a href="https://discord.gg/yPEP2vHTu4">DISCORD</a></p>
<p>Language and vision intertwine in the human mind, shaping how we perceive and understand the world around us. Our ability to reason is deeply rooted in both linguistic thought and visual memory - but what happens when we extend these capabilities to AI? Today's large language models have demonstrated remarkable reasoning abilities, but we wondered: could they harness the power of visual understanding to reach new heights of cognitive capability?</p>
<p>Imagine an AI that can look at a complex physics problem, and methodically reason its way to a solution with the confidence of a master physicist. This vision inspired us to create QVQ - an open-weight model for multimodal reasoning, built upon zen-VL-72B. QVQ represents a significant leap forward in AI's capacity for visual understanding and complex problem-solving. QVQ achieves a score of 70.3 on MMMU and shows substantial improvements across math-related benchmarks compared to zen-VL-72B-Instruct. Through careful step-by-step reasoning, QVQ demonstrates enhanced capabilities in visual reasoning tasks, particularly excelling in domains that demand sophisticated analytical thinking.</p>
<h1 id="limitations">Limitations</h1>
<p><strong>QvQ-72B-Preview</strong> is an experimental research model developed by the Qwen team, focusing on enhancing visual reasoning capabilities. While it has demonstrated performance that exceeds expectations, there are several limitations to be aware of:</p>
<ol>
<li><strong>Language Mixing and Code-Switching</strong>: The model may mix languages or switch between them unexpectedly, affecting response clarity.</li>
<li><strong>Recursive Reasoning</strong>: The model may get stuck in circular logic patterns, producing verbose responses without reaching conclusions.</li>
<li><strong>Safety and Ethical Considerations</strong>: The model requires enhanced safety measures to ensure reliable and secure performance, and users should be cautious when deploying it.</li>
<li><strong>Performance and Benchmark Limitations</strong>: Although the model has shown improvements in visual reasoning, it cannot fully replace the capabilities of zen-VL-72B-Instruct. Additionally, during multi-step visual reasoning, the model may gradually lose focus on the image content, leading to hallucinations.</li>
</ol>
<h1 id="performance">Performance</h1>
<p>We evaluate QVQ-72B-Preview on 4 datasets, including:</p>
<ul>
<li>MMMU: A university-level multidisciplinary multimodal evaluation dataset designed to assess models' visual-related comprehensive understanding and reasoning capabilities.</li>
<li>MathVista: A mathematics-focused visual reasoning test set that evaluates capabilities such as logical reasoning with puzzle test graphics, algebraic reasoning with function graphs, and scientific reasoning with academic paper figures.</li>
<li>MathVision: A high-quality multimodal mathematical reasoning test set derived from real mathematics competitions, featuring greater problem diversity and subject breadth compared to MathVista.</li>
<li>OlympiadBench: An Olympic competition-level bilingual multimodal science benchmark test set containing 8,476 problems from Olympic mathematics and physics competitions, including the Chinese college entrance examination. Each problem comes with expert-level annotations detailing the step-by-step reasoning process.</li>
</ul>
<figure><img src="https://qianwen-res.oss-cn-beijing.aliyuncs.com/QVQ/QVQ.jpg#center" alt="" loading="lazy"></figure>
<p>In particular, QVQ-72B-Preview has achieved an impressive score of 70.3 on the MMMU benchmark, significantly outpacing its predecessor, zen-VL-72B-Instruct. Furthermore, in the remaining three benchmarks focused on mathematics and science problems, the model demonstrates exceptional performance, effectively closing the gap with the leading state-of-the-art o1 model.</p>
<h1 id="demo-cases">Demo Cases</h1>
<p>In the following section, we present several examples to illustrate the application of this new model in visual reasoning tasks.</p>
<p>{/* Interactive example: cases/1_1.json <em>/}
{/</em> Interactive example: cases/1_2.json <em>/}
{/</em> Interactive example: cases/1_3.json <em>/}
{/</em> Interactive example: cases/1_5.json <em>/}
{/</em> Interactive example: cases/1_6.json <em>/}
{/</em> Interactive example: cases/1_7.json */}</p>
<h1 id="next-step">Next Step</h1>
<p>As we progress towards achieving AGI, our vision is to develop a <strong>omni</strong> and <strong>smart</strong> model. To realize this goal, we are enhancing our vision-language foundation model with advanced capabilities for deep thinking and reasoning based on visual information. In the near future, we plan to integrate additional modalities into a unified model, making it even more intelligent and capable of addressing complex challenges and engaging in scientific exploration.</p>f:["$","main",null,{"children":["$","article",null,{"className":"blog-article","children":[["$","$L5",null,{"className":"blog-back","href":"/blog","children":"← Blog"}],["$","div",null,{"className":"blog-post-meta","children":["December 24, 2024"," ","·"," ",3," min read"]}],["$","h1",null,{"className":"blog-post-title","children":"QVQ: To See the World with Wisdom"}],["$","p",null,{"className":"blog-post-lede","children":"Language and vision intertwine in the human mind, shaping how we perceive and understand the world around us. Our ability to reason is deeply rooted in both linguistic thought and visual memory - but what happens when we extend these capabilities to AI? Today's large language models have demonstrate"}],["$","div",null,{"className":"blog-prose","dangerouslySetInnerHTML":{"__html":"$19"}}]]}]}]
15:[["$","meta","0",{"charSet":"utf-8"}],["$","meta","1",{"name":"viewport","content":"width=device-width, initial-scale=1"}]]
11:null
13:{"metadata":[["$","title","0",{"children":"QVQ: To See the World with Wisdom — Zen Blog"}],["$","meta","1",{"name":"description","content":"Language and vision intertwine in the human mind, shaping how we perceive and understand the world around us. Our ability to reason is deeply rooted in both linguistic thought and visual memory - but what happens when we extend these capabilities to AI? Today's large language models have demonstrate"}],["$","meta","2",{"name":"keywords","content":"AI, LLM, Agentic AI, Code Generation, Zen Coder, Multimodal, Open Source, Machine Learning"}]],"error":null,"digest":"$undefined"}
18:"$13:metadata"
