1:"$Sreact.fragment"
2:I[6529,["619","static/chunks/619-ba102abea3e3d0e4.js","177","static/chunks/app/layout-13fa3fd02f6eb5db.js"],"default"]
3:I[9766,[],""]
4:I[8924,[],""]
5:I[2619,["619","static/chunks/619-ba102abea3e3d0e4.js","953","static/chunks/app/blog/%5Bslug%5D/page-f25a122e9ccf798d.js"],""]
e:I[7150,[],""]
:HL["/_next/static/css/1d1f6bc532e5f43f.css","style"]
:HL["/_next/static/css/eb87e4f7aea490c6.css","style"]
0:{"P":null,"b":"DQJo8iKQxHubJM4JvsRbH","p":"","c":["","blog","training-llms-collective-intelligence",""],"i":false,"f":[[["",{"children":["blog",{"children":[["slug","training-llms-collective-intelligence","d"],{"children":["__PAGE__",{}]}]}]},"$undefined","$undefined",true],["",["$","$1","c",{"children":[[["$","link","0",{"rel":"stylesheet","href":"/_next/static/css/1d1f6bc532e5f43f.css","precedence":"next","crossOrigin":"$undefined","nonce":"$undefined"}]],["$","html",null,{"lang":"en","children":[["$","head",null,{"children":[["$","link",null,{"rel":"icon","type":"image/svg+xml","href":"/favicon.svg"}],["$","link",null,{"rel":"alternate icon","href":"/favicon.png"}]]}],["$","body",null,{"children":[["$","$L2",null,{}],["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]],"forbidden":"$undefined","unauthorized":"$undefined"}],["$","footer",null,{"children":["$","div",null,{"className":"container","children":[["$","div",null,{"className":"footer-content","children":[["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen LM"}],["$","p",null,{"children":"95 open Zen models across Zen3, Zen4, and Zen5. Chat, code, vision, audio, image, embeddings, rerankers, and safety. OpenAI- and Anthropic-compatible API."}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 5"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Nano (0.8B - 9B)"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Flash"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Mini"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 (default)"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Coder"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Pro"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen5","children":"Zen5 Max"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 4"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 / Zen4.1"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Ultra / Max / Pro"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Mini / Thinking"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen4","children":"Zen4 Coder / Pro / Flash"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Zen 3 Multimodal"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Omni / VL / Web"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Nano / Guard"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Embedding / Reranker"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/models#zen3","children":"Zen3 Image / ASR / TTS"}]}]]}]]}],["$","div",null,{"className":"footer-section","children":[["$","h4",null,{"children":"Resources"}],["$","ul",null,{"children":[["$","li",null,{"children":["$","$L5",null,{"href":"/datasets","children":"Training Data"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://huggingface.co/zenlm","target":"_blank","rel":"noopener noreferrer","children":"HuggingFace"}]}],["$","li",null,{"children":["$","a",null,{"href":"https://github.com/zenlm","target":"_blank","rel":"noopener noreferrer","children":"GitHub"}]}],["$","li",null,{"children":["$","$L5",null,{"href":"/research","children":"Research Papers"}]}],["$","li",null,{"children":"$L6"}]]}]]}]]}],"$L7"]}]}],"$L8","$L9"]}]]}]]}],{"children":["blog","$La",{"children":[["slug","training-llms-collective-intelligence","d"],"$Lb",{"children":["__PAGE__","$Lc",{},null,false]},null,false]},null,false]},null,false],"$Ld",false]],"m":"$undefined","G":["$e",[]],"s":false,"S":true}
f:I[7405,["619","static/chunks/619-ba102abea3e3d0e4.js","177","static/chunks/app/layout-13fa3fd02f6eb5db.js"],"default"]
11:I[4431,[],"OutletBoundary"]
13:I[5278,[],"AsyncMetadataOutlet"]
15:I[4431,[],"ViewportBoundary"]
17:I[4431,[],"MetadataBoundary"]
18:"$Sreact.suspense"
6:["$","a",null,{"href":"https://api.hanzo.ai","target":"_blank","rel":"noopener noreferrer","children":"Zen API"}]
7:["$","div",null,{"className":"footer-bottom","children":["$","p",null,{"children":["© ",2026," Zen Authors. Open foundation models. Served on the Zen API."]}]}]
8:["$","$Lf",null,{}]
9:["$","script",null,{"src":"/assets/js/main.js","async":true}]
a:["$","$1","c",{"children":[null,["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","forbidden":"$undefined","unauthorized":"$undefined"}]]}]
b:["$","$1","c",{"children":[null,["$","$L3",null,{"parallelRouterKey":"children","error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L4",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","forbidden":"$undefined","unauthorized":"$undefined"}]]}]
c:["$","$1","c",{"children":["$L10",[["$","link","0",{"rel":"stylesheet","href":"/_next/static/css/eb87e4f7aea490c6.css","precedence":"next","crossOrigin":"$undefined","nonce":"$undefined"}]],["$","$L11",null,{"children":["$L12",["$","$L13",null,{"promise":"$@14"}]]}]]}]
d:["$","$1","h",{"children":[null,[["$","$L15",null,{"children":"$L16"}],null],["$","$L17",null,{"children":["$","div",null,{"hidden":true,"children":["$","$18",null,{"fallback":null,"children":"$L19"}]}]}]]}]
1a:Tc9a,<p>Language models are trained on text. That text represents the accumulated knowledge, reasoning, and creativity of countless individuals. Yet the curation process that selects training data receives surprisingly little attention.</p>
<h2 id="the-data-problem">The Data Problem</h2>
<p>Most large language models are trained on web scrapes filtered by simple heuristics. This approach has several issues:</p>
<ol>
<li><strong>Quality variance</strong>: Web content ranges from expert research to spam</li>
<li><strong>Hidden biases</strong>: Filtering decisions embed value judgments</li>
<li><strong>Provenance opacity</strong>: It's unclear what's included or excluded</li>
<li><strong>Legal ambiguity</strong>: Copyright and consent questions remain unresolved</li>
</ol>
<h2 id="our-approach-transparent-curation">Our Approach: Transparent Curation</h2>
<p>At Zen, we're taking a different path. Our data pipeline operates on three principles.</p>
<h3 id="explicit-criteria">Explicit Criteria</h3>
<p>Every filtering decision has documented rationale. When we exclude content, we record why. When we weight certain sources higher, we explain the reasoning. This creates an auditable trail.</p>
<h3 id="community-input">Community Input</h3>
<p>Data curation involves value judgments. What counts as "quality"? Which perspectives matter? These aren't purely technical questions. We're building mechanisms for community input into curation criteria.</p>
<h3 id="provenance-tracking">Provenance Tracking</h3>
<p>Each training example links to its source with metadata about:</p>
<ul>
<li>Original publication context</li>
<li>Author information (when available)</li>
<li>Licensing terms</li>
<li>Processing steps applied</li>
</ul>
<h2 id="technical-implementation">Technical Implementation</h2>
<p>We've developed a pipeline that processes documents through several stages:</p>
<pre><code>Source -> Extraction -> Deduplication -> Quality Scoring -> Attribution -> Storage
</code></pre>
<p>The quality scoring model itself is trained on human judgments, with explicit criteria:</p>
<ul>
<li>Factual accuracy (where verifiable)</li>
<li>Reasoning coherence</li>
<li>Writing clarity</li>
<li>Information density</li>
</ul>
<h2 id="early-results">Early Results</h2>
<p>Our initial corpus contains 2.3 trillion tokens with full provenance tracking. Early experiments suggest that careful curation can match larger but noisier datasets:</p>

























<table><thead><tr><th>Corpus</th><th>Tokens</th><th>Benchmark Score</th></tr></thead><tbody><tr><td>Web-raw</td><td>5T</td><td>72.3</td></tr><tr><td>Web-filtered</td><td>3T</td><td>74.1</td></tr><tr><td>Zen-curated</td><td>2.3T</td><td>75.8</td></tr></tbody></table>
<p>The numbers are preliminary, but the direction is clear: quality over quantity.</p>
<h2 id="what-this-means">What This Means</h2>
<p>Training on collective intelligence isn't just about scraping more data. It's about respecting the sources, maintaining transparency, and involving the community in decisions that shape AI behavior.</p>
<p>We'll publish our full curation methodology and tooling in the coming months.</p>
<hr>
<p><em>Zach Kelling is a co-founder of Zoo Labs Foundation.</em></p>10:["$","main",null,{"children":["$","article",null,{"className":"blog-article","children":[["$","$L5",null,{"className":"blog-back","href":"/blog","children":"← Blog"}],["$","div",null,{"className":"blog-post-meta","children":["July 21, 2021"," ","·"," ",2," min read"]}],["$","h1",null,{"className":"blog-post-title","children":"Training LLMs on Collective Intelligence"}],["$","p",null,{"className":"blog-post-lede","children":"How we're approaching training data curation to capture humanity's collective intelligence."}],["$","div",null,{"className":"blog-prose","dangerouslySetInnerHTML":{"__html":"$1a"}}]]}]}]
16:[["$","meta","0",{"charSet":"utf-8"}],["$","meta","1",{"name":"viewport","content":"width=device-width, initial-scale=1"}]]
12:null
14:{"metadata":[["$","title","0",{"children":"Training LLMs on Collective Intelligence — Zen Blog"}],["$","meta","1",{"name":"description","content":"How we're approaching training data curation to capture humanity's collective intelligence."}],["$","meta","2",{"name":"keywords","content":"AI, LLM, Agentic AI, Code Generation, Zen Coder, Multimodal, Open Source, Machine Learning"}]],"error":null,"digest":"$undefined"}
19:"$14:metadata"
