1:"$Sreact.fragment"
2:I[6529,["619","static/chunks/619-ba102abea3e3d0e4.js","177","static/chunks/app/layout-13fa3fd02f6eb5db.js"],"default"]
3:I[9766,[],""]
4:I[8924,[],""]
5:I[2619,["619","static/chunks/619-ba102abea3e3d0e4.js","953","static/chunks/app/blog/%5Bslug%5D/page-f25a122e9ccf798d.js"],""]
d:I[7150,[],""]
:HL["/_next/static/css/1d1f6bc532e5f43f.css","style"]
:HL["/_next/static/css/eb87e4f7aea490c6.css","style"]
0:{"P":null,"b":"DQJo8iKQxHubJM4JvsRbH","p":"","c":["","blog","qwen2.5-math",""],"i":false,"f":[[["",{"children":["blog",{"children":[["slug","qwen2.5-math","d"],{"children":["__PAGE__",{}]}]}]},"$undefined","$undefined",true],["",["$","$1","c",{"children":[[["$","link","0",{"rel":"stylesheet","href":"/_next/static/css/1d1f6bc532e5f43f.css","precedence":"next","crossOrigin":"$undefined","nonce":"$undefined"}]],["$","html",null,{"lang":"en","children":[["$","head",null,{"children":[["$","link",null,{"rel":"icon","type":"image/svg+xml","href":"/favicon.svg"}],["$","link",null,{"rel":"alternate icon","href":"/favicon.png"}]]}],["$","body",null,{"children":[["$","$L2",null,{}],["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]],"forbidden":"$undefined","unauthorized":"$undefined"}],["$","footer",null,{"children":["$","div",null,{"className":"container","children":[["$","div",null,{"className":"footer-content","children":[["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen LM"}],["$","p",null,{"children":"95 open Zen models across Zen3, Zen4, and Zen5. Chat, code, vision, audio, image, embeddings, rerankers, and safety. OpenAI- and Anthropic-compatible API."}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 5"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Nano (0.8B - 9B)"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Flash"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Mini"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 (default)"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Coder"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Pro"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Max"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 4"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 / Zen4.1"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Ultra / Max / Pro"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Mini / Thinking"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Coder / Pro / Flash"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 3 Multimodal"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Omni / VL / Web"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Nano / Guard"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Embedding / Reranker"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Image / ASR / TTS"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Resources"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/datasets","children":"Training Data"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://huggingface.co/zenlm","target":"_blank","rel":"noopener noreferrer","children":"HuggingFace"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://github.com/zenlm","target":"_blank","rel":"noopener noreferrer","children":"GitHub"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/research","children":"Research Papers"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://api.hanzo.ai","target":"_blank","rel":"noopener noreferrer","children":"Zen API"}]}]]}]]}]]}],"$L6"]}]}],"$L7","$L8"]}]]}]]}],{"children":["blog","$L9",{"children":[["slug","qwen2.5-math","d"],"$La",{"children":["__PAGE__","$Lb",{},null,false]},null,false]},null,false]},null,false],"$Lc",false]],"m":"$undefined","G":["$d",[]],"s":false,"S":true}
e:I[7405,["619","static/chunks/619-ba102abea3e3d0e4.js","177","static/chunks/app/layout-13fa3fd02f6eb5db.js"],"default"]
10:I[4431,[],"OutletBoundary"]
12:I[5278,[],"AsyncMetadataOutlet"]
14:I[4431,[],"ViewportBoundary"]
16:I[4431,[],"MetadataBoundary"]
17:"$Sreact.suspense"
6:["$","div",null,{"className":"footer-bottom","children":["$","p",null,{"children":["© ",2026," Zen Authors. Open foundation models. Served on the Zen API."]}]}]
7:["$","$Le",null,{}]
8:["$","script",null,{"src":"/assets/js/main.js","async":true}]
9:["$","$1","c",{"children":[null,["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","forbidden":"$undefined","unauthorized":"$undefined"}]]}]
a:["$","$1","c",{"children":[null,["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","forbidden":"$undefined","unauthorized":"$undefined"}]]}]
b:["$","$1","c",{"children":["$Lf",[["$","link","0",{"rel":"stylesheet","href":"/_next/static/css/eb87e4f7aea490c6.css","precedence":"next","crossOrigin":"$undefined","nonce":"$undefined"}]],["$","$L10",null,{"children":["$L11",["$","$L12",null,{"promise":"$@13"}]]}]]}]
c:["$","$1","h",{"children":[null,[["$","$L14",null,{"children":"$L15"}],null],["$","$L16",null,{"children":["$","div",null,{"hidden":true,"children":["$","$17",null,{"fallback":null,"children":"$L18"}]}]}]]}]
19:T2d1b,<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/2024-08-qwen3-math-72B.png#center" alt="" loading="lazy"></figure>
<p><a href="https://github.com/QwenLM/zen-Math">GITHUB</a>
<a href="https://huggingface.co/Qwen">HUGGING FACE</a>
<a href="https://modelscope.cn/organization/qwen">MODELSCOPE</a>
<a href="https://discord.gg/yPEP2vHTu4">DISCORD</a></p>
<blockquote>
<div>
<b>
🚨 zen-Math mainly supports solving English and Chinese math problems through CoT and TIR. We do not recommend using this series of models for other tasks.
</b>
</div>
</blockquote>
<h1 id="introduction">Introduction</h1>
<p>A month ago, we released the first series of mathematical LLMs - <a href="https://qwenlm.github.io/blog/qwen2-math/">zen-Math</a> - of our Qwen family. Today, we have upgraded it and open-sourced <strong>zen-Math</strong> series, including base models <strong>zen-Math-1.5B/7B/72B</strong>, instruction-tuned models <strong>zen-Math-1.5B/7B/72B-Instruct</strong>, and mathematical reward model <strong>zen-Math-RM-72B</strong>.</p>
<p>Unlike zen-Math series which only supports using Chain-of-Thought (CoT) to solve English math problems, zen-Math series is expanded to support using both CoT and Tool-integrated Reasoning (TIR) to solve math problems in both Chinese and English. The zen-Math series models have achieved significant performance improvements compared to the zen-Math series models on the Chinese and English mathematics benchmarks with CoT.</p>
<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/2024-08-qwen3-math-allsize.png#center" alt="" loading="lazy"></figure>
<p>While CoT plays a vital role in enhancing the reasoning capabilities of LLMs, it faces challenges in achieving computational accuracy and handling complex mathematical or algorithmic reasoning tasks, such as finding the roots of a quadratic equation or computing the eigenvalues of a matrix. TIR can further improve the model's proficiency in precise computation, symbolic manipulation, and algorithmic manipulation. zen-Math-1.5B/7B/72B-Instruct achieve 79.7, 85.3, and 87.8 respectively on the MATH benchmark using TIR.</p>
<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/qwen3-math-pipeline.jpeg#center" alt="" loading="lazy"></figure>
<h2 id="zen-math-base-models">zen-Math: Base Models</h2>
<p>The overall specialization pipelines of zen-Math and zen-Math are shown in the figure above. After training of zen-Math base models, we further upgrade them to zen-Math models through three primary avenues:</p>
<ol>
<li>
<p>Utilizing zen-Math-72B-Instruct models to synthesize additional high-quality mathematical pre-training data.</p>
</li>
<li>
<p>Aggregating more high-quality mathematical data, particularly in Chinese, from web sources, books, and codes across multiple recall cycles.</p>
</li>
<li>
<p>Leveraging the zen series base model for parameter initialization, which shows more powerful language understanding, code generation, and text reasoning capabilities.</p>
</li>
</ol>
<p>Ultimately, we construct <em>Qwen Math Corpus v2</em> for zen-Math-1.5B/7B/72B pre-training, maintaining a context length of 4K. Compared to <em>Qwen Math Corpus v1</em> used for zen-Math training, the total token count of <em>Qwen Math Corpus v2</em> has increased from 700B to over 1T.</p>
<p>We evaluate our zen-Math base models on three widely used English math benchmarks GSM8K, Math, and MMLU-STEM. In addition, we also evaluate three Chinese math benchmarks CMATH, GaoKao Math Cloze, and GaoKao Math QA. All evaluations are tested with few-shot chain-of-thought prompting.</p>
<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/qwen3-math-table.png#center" alt="" loading="lazy"></figure>
<p>Compared to zen-Math-1.5B/7B/72B, zen-Math-1.5B/7B/72B have achieved significant improvements on all benchmarks. For example, zen-Math-1.5B/7B/72B obtains 5.4, 5.0, 6.3 scores improvement on MATH, and 3.4, 12.2, 19.8 scores improvement on Gaokao Math QA.</p>
<h2 id="zen-math-instruct-instruction-tuned-models">zen-Math-Instruct: Instruction-Tuned Models</h2>
<p>Similar to zen-Math-Instruct, we train a math-specific reward model zen-Math-RM-72B based on zen-Math-72B. This RM is used for constructing the SFT data through Rejection Sampling and also in the reinforcement learning with Group Relative Policy Optimization (GRPO) after SFT.</p>
<p>In the development of zen-Math-Instruct, an additional iteration is conducted using the zen-Math-Instruct models and zen-Math-RM-72B to polish the quality of responses further during Rejection Sampling.</p>
<p>Compared with the post-training of zen-Math, we further introduced TIR data and SFT data in Chinese and English for zen post-training.</p>
<p>We evaluate zen-Math-Instruct on mathematical benchmarks in both English and Chinese. In addition to the widely-used benchmarks, such as GSM8K and Math, we also involve more exams that are more challenging to fully inspect the capabilities of zen-Math-Instruct, such as OlympiadBench, CollegeMath, GaoKao, AIME2024, and AMC2023. For Chinese mathematical benchmarks, we use CMATH, Gaokao (Chinese College Entrance Examination 2024), and CN Middle School 24 (China High School Entrance Examination 2024).</p>
<p>We report greedy, Maj@8 and RM@8 performance on all benchmarks in the zero-shot setting, except for the multi-choice benchmarks (including MMLU STEM and multiple-choice problems in GaoKao and CN Middle School 24) with a 5-shot setting.</p>
<p>The zen-Math-72B-Instruct model outperforms the zen-Math-72B-Instruct model by an average margin of 4.4 and 6.1 points in English and Chinese, respectively, establishing itself as the best open-source mathematical model currently available.</p>
<p>The flagship model, zen-Math-72B-Instruct, significantly outperforms both open-source models and leading closed-source models (e.g., GPT-4o, Gemini Math-Specialized 1.5 Pro). Under the TIR setting of RM@8, a high score of 92.9 was achieved on MATH.</p>
<p>With the aid of synthesized pre-training and supervised fine-tuning data from the 72B model, zen-Math-7B-Instruct surpasses zen-Math-Instruct 72B in performance. Under CoT and TIR settings, it achieves MATH scores of 83.6 and 85.3, respectively.</p>
<p>Even our smallest 1.5B model, achieves a MATH score of around 80 when utilizing the Python Interpreter, outperforming the majority of current models in this domain.</p>
<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/math_instruct_en.jpg#center" alt="" loading="lazy"></figure>
<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/math_instruct_zh.jpg#center" alt="" loading="lazy"></figure>
<p>In more complex mathematical competition evaluations such as AIME 2024 and AMC 2023, zen-Math-Instruct also performs well across various settings, including Greedy, Maj@64, RM@64, and RM@256.</p>
<p>With the support of the zen-Math-RM-72B, zen-Math-1.5B-Instruct, using the RM@256 in CoT mode, successfully solves 29 out of 40 problems on AMC 2023.</p>
<p>Moreover, zen-Math-72B-Instruct nearly achieves a perfect score in TIR mode, solving almost all the problems.</p>
<p>On the extremely difficult AIME 2024 benchmark, Claude3 Opus, GPT-4 Turbo, and Gemini 1.5 Pro manage to solve only 1 or 2 questions out of 30.</p>
<p>In contrast, zen-Math-72B-Instruct solves 9 problems in Greedy decoding CoT mode and 12 problems in TIR mode. With the help of the RM, zen-Math-7B-Instruct could even solve up to 21 problems, further demonstrating the outstanding mathematical problem-solving ability of zen-Math-Instruct.</p>
<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/math_instruct_aime.jpg#center" alt="" loading="lazy"></figure>
<h2 id="decontamination">Decontamination</h2>
<p>Decontamination is critical to ensuring unbiased model performance evaluation.</p>
<p>Following prior work zen, we exclude potentially contaminated training samples using 13-gram matching. To improve the accuracy of this matching process, we perform text normalization, removing irrelevant punctuation and symbols.</p>
<p>To further reduce false negatives, particularly for common mathematical expressions, we introduce an additional criterion: the ratio of the longest common subsequence must exceed <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mn>0.6</mn></mrow><annotation encoding="application/x-tex">0.6</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6444em;"></span><span class="mord">0.6</span></span></span></span> for a sample to be considered contaminated.</p>
<p>For pre-training data, we filter potentially contaminated samples against datasets such as GSM8K and MATH.
When dealing with post-training data, including SFT data, RM training data, and the RL query set, we exclude any potentially contaminated problems or solutions across all reported evaluation datasets. These evaluation datasets include GSM8K, MATH, Minerva Math, Gaokao 2023 En, Olympiad Bench, College Math, MMLU STEM, GaoKao, CMATH, CN Middle School 24, AIME 24, and AMC 23.</p>
<p>During the analysis of contaminated samples, we identify that some existing training datasets (e.g., the MATH training dataset) contain a significant proportion of problems that share highly similar concepts or structures with those found in test datasets.
Although these variations are not exact duplicates, they could potentially compromise the integrity of our evaluation.
Therefore, we continue to exclude such samples from the training corpora.</p>
<h2 id="demo">Demo</h2>
<p>We develop a demo that supports the TIR mode in <a href="https://github.com/QwenLM/Qwen-Agent">Qwen-Agent</a>, which allows running code locally to experience Tool-Integrated Reasoning capabilities of zen-Math.</p>
<figure><img src="http://qianwen-res.oss-cn-beijing.aliyuncs.com/zen/qwen3-math-example1.png#center" alt="" loading="lazy"></figure>
<p>Furthermore, we provide a multi-modal mathematic demo in <a href="https://huggingface.co/spaces/Qwen/zen-Math-Demo">Huggingface</a> and <a href="https://www.modelscope.cn/studios/qwen/Qwen-Math-demo">Modelscope</a>. This WebUI is based on zen-VL for OCR and zen-Math for mathematical reasoning. You can input either images, texts, or sketches of mathematical and arithmetic problems.</p>
<h2 id="summary">Summary</h2>
<p>We introduce zen-Math, which features several key technical highlights:</p>
<p>(1) Extensive using of synthesized mathematical data from zen-Math during the pre-training phase.</p>
<p>(2) Iterative generation of fine-tuning data and reinforcement training guided by the reward model during the post-training phase.</p>
<p>(3) Supporting for bilingual (English and Chinese) queries, along with chain-of-thought and tool-integrated reasoning capabilities.</p>
<p>As a result, zen-Math represents the most advanced open-source math model series to date.
The zen-Math-1.5B-Instruct model already surpasses most previous 70B math models, while the zen-Math-7B-Instruct matches the performance of zen-Math-72B-Instruct.
Our flagship model, zen-Math-7B-Instruct, outperforms zen-Math-72B-Instruct with an average score increase of 4.7 points across 7 tasks.</p>
<p>We hope that the advances we’ve made with specialized models like zen-Math will continue to strengthen the overall capabilities of the Qwen model and bring us closer to achieving artificial general intelligence.</p>f:["$","main",null,{"children":["$","article",null,{"className":"blog-article","children":[["$","$L5",null,{"className":"blog-back","href":"/blog","children":"← Blog"}],["$","div",null,{"className":"blog-post-meta","children":["September 18, 2024"," ","·"," ",7," min read"]}],["$","h1",null,{"className":"blog-post-title","children":"zen-Math: The world's leading open-sourced mathematical LLMs"}],["$","p",null,{"className":"blog-post-lede","children":"> <div align=\"center\"> > <b> > 🚨 zen-Math mainly supports solving English and Chinese math problems through CoT and TIR. We do not recommend using this series of models for other tasks. > </b> > </div>"}],["$","div",null,{"className":"blog-prose","dangerouslySetInnerHTML":{"__html":"$19"}}]]}]}]
15:[["$","meta","0",{"charSet":"utf-8"}],["$","meta","1",{"name":"viewport","content":"width=device-width, initial-scale=1"}]]
11:null
13:{"metadata":[["$","title","0",{"children":"zen-Math: The world's leading open-sourced mathematical LLMs — Zen Blog"}],["$","meta","1",{"name":"description","content":"> <div align=\"center\"> > <b> > 🚨 zen-Math mainly supports solving English and Chinese math problems through CoT and TIR. We do not recommend using this series of models for other tasks. > </b> > </div>"}],["$","meta","2",{"name":"keywords","content":"AI, LLM, Agentic AI, Code Generation, Zen Coder, Multimodal, Open Source, Machine Learning"}]],"error":null,"digest":"$undefined"}
18:"$13:metadata"
