[{"data":1,"prerenderedAt":2337},["ShallowReactive",2],{"sanity-7iXP3sOrSyrwpMoyaopqutrI_r2a7XNibHtOh578tGQ":3,"sanity-Gp2TQ3KTRm3G-aQEd8uRROtS0gNWs2keg4oQkQ2mIc0":219},{"data":4,"sourceMap":139},{"_createdAt":5,"_id":6,"_rev":7,"_system":8,"_type":6,"_updatedAt":11,"defaultTitle":12,"footerMenu1":13,"footerMenu2":29,"footerMenu3":44,"gitHubUrl":53,"huggingFaceUrl":57,"linkedInUrl":49,"mainNavigation":59,"ogDescription":87,"ogImage":88,"ogImageUrl":93,"showSiteNotice":94,"siteNotice":95},"2026-01-21T14:58:26Z","siteSettingsSchema","mbou5Mk4jD84jdX2vzI6cY",{"base":9},{"id":6,"rev":10},"6X7ayGJgs0U64XMIw22UCy","2026-07-30T09:11:27Z","kaiko.ai",{"links":14,"title":28},[15,20,24],{"_key":16,"_type":17,"label":18,"url":19},"95cf27020263","navigationItem","Clinical AI Workspace","/clinical-ai-workspace",{"_key":21,"_type":17,"label":22,"url":23},"7b5bdfebb176","Research","/research",{"_key":25,"_type":17,"label":26,"url":27},"d311ee16a390","Insights Hub","/insights-hub","What we do",{"links":30,"title":43},[31,35,39],{"_key":32,"_type":17,"label":33,"url":34},"44a37ad9151d","About us","/about",{"_key":36,"_type":17,"label":37,"url":38},"193660568291","Careers","https://jobs.kaiko.ai/",{"_key":40,"_type":17,"label":41,"url":42},"aeb117deb742","Trust Center","https://trust.kaiko.ai/","Company",{"links":45,"title":58},[46,50,54],{"_key":47,"_type":17,"label":48,"url":49},"40c6bf52fdd1","LinkedIn","https://www.linkedin.com/company/kaiko-ai/",{"_key":51,"_type":17,"label":52,"url":53},"f79c6a57ca2f","GitHub","https://github.com/kaiko-ai",{"_key":55,"_type":17,"label":56,"url":57},"7ff3680a1f90","Hugging Face","https://huggingface.co/kaiko-ai","Community",[60,62,64],{"_key":61,"_type":17,"label":18,"url":19},"601985fd4221",{"_key":63,"_type":17,"label":22,"url":23},"a10cc775319c",{"_key":65,"_type":66,"children":67,"label":86},"3c77dba0a31e","navigationGroup",[68,71,73,77,79,82],{"_key":69,"_type":17,"label":70,"url":34},"4d07fcce85a7","About",{"_key":72,"_type":17,"label":26,"url":27},"66b87567d450",{"_key":74,"_type":17,"label":75,"url":76},"bf5d3c3b0672","Events","https://kaiko.ai/insights-hub?categories=event",{"_key":78,"_type":17,"label":37,"url":38},"b34eb970c9d9",{"_key":80,"_type":17,"label":81,"url":42},"0179b3449f54","Security & Compliance",{"_key":83,"_type":17,"label":84,"url":85},"9f1fac3acbae","Contact Us","mailto:info@kaiko.ai","Resources","Your Clinical AI assistant that supports across full patient care, combining context for deeper insights and reducing workload, safe and compliant, made in EU.",{"_type":89,"asset":90},"image",{"_ref":91,"_type":92},"image-d9426451401bfd3574519485220b90bce4571422-1200x630-jpg","reference","https://cdn.sanity.io/images/a5kt9um3/production/d9426451401bfd3574519485220b90bce4571422-1200x630.jpg?w=1200&h=630&fit=crop",true,{"badges":96,"link":125,"linkText":126,"message":127},[97,118],{"_key":98,"color":99,"label":117},"70498638b5b3",{"_type":100,"alpha":101,"hex":102,"hsl":103,"hsv":108,"rgb":112},"color",1,"#a8e2df",{"_type":104,"a":101,"h":105,"l":106,"s":107},"hslaColor",176.89655172413794,0.7725490196078431,0.4999999999999999,{"_type":109,"a":101,"h":105,"s":110,"v":111},"hsvaColor",0.2566371681415929,0.8862745098039215,{"_type":113,"a":101,"b":114,"g":115,"r":116},"rgbaColor",223,226,168,"Sept 30",{"_key":119,"color":120,"label":124},"b238b246731c",{"_type":100,"alpha":101,"hex":102,"hsl":121,"hsv":122,"rgb":123},{"_type":104,"a":101,"h":105,"l":106,"s":107},{"_type":109,"a":101,"h":105,"s":110,"v":111},{"_type":113,"a":101,"b":114,"g":115,"r":116},"In-person","https://kaiko.ai/insights-hub/Clinical-AI-Partnership-Event","More info",[128],{"_key":129,"_type":130,"children":131,"markDefs":137,"style":138},"ee062a592cfa","block",[132],{"_key":133,"_type":134,"marks":135,"text":136},"41b4b4d5d3ff","span",[],"Join our exclusive event during Zurich AI Festival, hear from CIOs, CMIOs, and hospital leaders",[],"normal",{"documents":140,"paths":144,"mappings":163},[141,143],{"_id":91,"_type":142},"sanity.imageAsset",{"_id":6,"_type":6},[145,146,147,148,149,150,151,152,153,154,155,156,157,158,159,160,161,162],"$['_createdAt']","$['_id']","$['_rev']","$['_system']","$['_type']","$['_updatedAt']","$['defaultTitle']","$['footerMenu1']","$['footerMenu2']","$['footerMenu3']","$['gitHubUrl']","$['huggingFaceUrl']","$['linkedInUrl']","$['mainNavigation']","$['ogDescription']","$['ogImage']","$['siteNotice']","$['url']",{"$['_createdAt']":164,"$['_id']":169,"$['_rev']":171,"$['_system']":174,"$['_type']":177,"$['_updatedAt']":180,"$['defaultTitle']":183,"$['footerMenu1']":186,"$['footerMenu2']":189,"$['footerMenu3']":192,"$['gitHubUrl']":195,"$['huggingFaceUrl']":198,"$['linkedInUrl']":201,"$['mainNavigation']":204,"$['ogDescription']":207,"$['ogImage']":210,"$['ogImageUrl']":213,"$['siteNotice']":216},{"source":165,"type":168},{"document":101,"path":166,"type":167},0,"documentValue","value",{"source":170,"type":168},{"document":101,"path":101,"type":167},{"source":172,"type":168},{"document":101,"path":173,"type":167},2,{"source":175,"type":168},{"document":101,"path":176,"type":167},3,{"source":178,"type":168},{"document":101,"path":179,"type":167},4,{"source":181,"type":168},{"document":101,"path":182,"type":167},5,{"source":184,"type":168},{"document":101,"path":185,"type":167},6,{"source":187,"type":168},{"document":101,"path":188,"type":167},7,{"source":190,"type":168},{"document":101,"path":191,"type":167},8,{"source":193,"type":168},{"document":101,"path":194,"type":167},9,{"source":196,"type":168},{"document":101,"path":197,"type":167},10,{"source":199,"type":168},{"document":101,"path":200,"type":167},11,{"source":202,"type":168},{"document":101,"path":203,"type":167},12,{"source":205,"type":168},{"document":101,"path":206,"type":167},13,{"source":208,"type":168},{"document":101,"path":209,"type":167},14,{"source":211,"type":168},{"document":101,"path":212,"type":167},15,{"source":214,"type":168},{"document":166,"path":215,"type":167},17,{"source":217,"type":168},{"document":101,"path":218,"type":167},16,{"data":220,"sourceMap":2253},{"_createdAt":221,"_id":222,"_rev":223,"_system":224,"_type":227,"_updatedAt":228,"area":229,"authors":231,"categories":235,"content":239,"date":1011,"featuredImage":1012,"isHidden":1015,"relatedPosts":1016,"slug":1023,"subHeading":1026,"suggestedPosts":1027,"title":2243,"topics":2244},"2026-09-02T07:41:36Z","2ebef6ad-1d7e-4dd2-8c40-e6ab0ebc69b9","2HnxnCjtcNi6fiTsCuvbIV",{"base":225},{"id":222,"rev":226},"ZLlcazqhY0lU8NzTPubaCV","post","2026-09-03T08:51:57Z",[230],"engineering",[232],{"_key":233,"_ref":234,"_type":92},"a7521ee36d77","c34dda2c-bcd5-4223-9fbe-f6fe2a64a68b",[236],{"_key":237,"_ref":238,"_type":92},"8e80265910b0","9613ea96-3dba-4c14-9ac9-2b5fe14eac53",[240,278,281,334,336,382,389,454,456,535,541,584,590,825,827,910,916,927,933,962,964,999],{"_key":241,"_type":242,"text":243},"9e4800b1174a","textSection",[244,253,270],{"_key":245,"_type":130,"children":246,"markDefs":251,"style":252},"cac1ad694d75",[247],{"_key":248,"_type":134,"marks":249,"text":250},"2b5a3d4647c1",[],"Summary",[],"h2",{"_key":254,"_type":130,"children":255,"markDefs":269,"style":138},"8efbe3542346",[256,260,265],{"_key":257,"_type":134,"marks":258,"text":259},"c478b1444472",[],"Making a frontier model excel at clinical work takes thousands of experiments, some running on hundreds of GPUs for weeks. At that scale, model development has to work like a production line. The ",{"_key":261,"_type":134,"marks":262,"text":264},"eee6ee8ce28b",[263],"em","AI Factory",{"_key":266,"_type":134,"marks":267,"text":268},"7e1404db9f3d",[]," is the platform we built to automate it, connecting diverse environments and workflows — from local development to distributed training — into a reproducible pipeline from source data to a post-trained model, while preserving the strict boundaries that medical data demands. This article introduces the AI Factory and shows how it accelerates research cycles, letting a small team develop frontier models for the clinic quickly and reliably.",[],{"_key":271,"_type":130,"children":272,"markDefs":277,"style":138},"0adab452f970",[273],{"_key":274,"_type":134,"marks":275,"text":276},"d16e987d8ee1",[],"",[],{"_key":279,"_type":280},"68e94cdea299","divider",{"_key":282,"_type":242,"text":283},"75866169f36d",[284,315],{"_key":285,"_type":130,"children":286,"markDefs":309,"style":138},"c90b2ab5542e",[287,291,296,300,305],{"_key":288,"_type":134,"marks":289,"text":290},"ede2eedbd9ba",[],"Frontier models are strong generalists. In the ",{"_key":292,"_type":134,"marks":293,"text":295},"1e569d2c1001",[294],"1ec288d94455","AMA's 2026 survey",{"_key":297,"_type":134,"marks":298,"text":299},"71bad5f7f81d",[],", four in five US physicians reported using AI in their practice, most often to summarize research and draft documentation. But clinical work also demands skills ",{"_key":301,"_type":134,"marks":302,"text":304},"e8b2879e6169",[303],"4ad737819d5e","not in the models' training data",{"_key":306,"_type":134,"marks":307,"text":308},"5a1e1ff5bc4d",[],": operating the tools and systems clinicians work with, like navigating a gigapixel pathology slide to describe the findings.",[310,313],{"_key":294,"_type":311,"href":312},"link","https://www.ama-assn.org/press-center/ama-press-releases/ama-ai-usage-among-doctors-doubles-confidence-technology-grows",{"_key":303,"_type":311,"href":314},"https://kaiko.ai/insights-hub/betting-Machine-God",{"_key":316,"_type":130,"children":317,"markDefs":331,"style":138},"9ae12db00b51",[318,322,327],{"_key":319,"_type":134,"marks":320,"text":321},"d08bea3c7c9f",[],"At kaiko, we close that gap by taking open frontier models and training them for clinical use. That work runs from clinicians describing their workflows, through collecting or synthesizing training data, to integrating it into the training mix. Every change requires controlled experiments, run at more than one scale and across different training stages. Ideas are never in short supply, so the pace is set by how quickly they can be turned into experiments that can be trusted. Earlier articles described the models we built and ",{"_key":323,"_type":134,"marks":324,"text":326},"61839d9137a4",[325],"f278a7b7a4db","how we train them at scale",{"_key":328,"_type":134,"marks":329,"text":330},"42eef6ef2526",[],"; this one focuses on the machinery that speeds up research cycles.",[332],{"_key":325,"_type":311,"href":333},"https://kaiko.ai/insights-hub/Training-clinical-reasoning-model-at-scale",{"_key":335,"_type":280},"24912a8e1ec5",{"_key":337,"_type":242,"text":338},"c735f287def7",[339,347,366,374],{"_key":340,"_type":130,"children":341,"markDefs":346,"style":252},"ccb0c534479e",[342],{"_key":343,"_type":134,"marks":344,"text":345},"4dd758be586c",[],"1. Enter the AI Factory",[],{"_key":348,"_type":130,"children":349,"markDefs":363,"style":138},"e49f8a7d1735",[350,354,359],{"_key":351,"_type":134,"marks":352,"text":353},"ae8c2139b335",[],"The AI Factory is the platform we built to make the experimentation loop fast and reliable. It turns a stream of changes into controlled experiments that can be trusted. The factory framing was inspired by ",{"_key":355,"_type":134,"marks":356,"text":358},"fceb272a9e5e",[357],"971e803b7aff","Poolside's Model Factory",{"_key":360,"_type":134,"marks":361,"text":362},"9aa65e101dc4",[],".",[364],{"_key":357,"_type":311,"href":365},"https://poolside.ai/blog/introducing-the-model-factory",{"_key":367,"_type":130,"children":368,"markDefs":373,"style":138},"ed44d7a64bd8",[369],{"_key":370,"_type":134,"marks":371,"text":372},"6d419c208001",[],"Consider the following example: a new dataset has been transferred from a data partner, and we want to see whether it improves the model. Integrating the new dataset into training is rarely a single step. The data has to be validated, filtered, deduplicated, copied to the compute cluster, included in the training mix, and so on.",[],{"_key":375,"_type":130,"children":376,"markDefs":381,"style":138},"dc925920b747",[377],{"_key":378,"_type":134,"marks":379,"text":380},"88e0b9473af6",[],"With the AI Factory, the whole cascade through data and training pipelines becomes a single change, versioned and reproducible (Figure 1). It can be a single pull request that declares the new dataset, plugs it into the existing pipeline, and adds it to the training mix. On merge, the platform takes over: it materializes the dataset while the CI builds and publishes the container image with the new training mix inside, and smoke tests the training pipeline overnight. The next training run automatically picks up the change, spins up a fresh cluster of GPU workers that find the dataset already waiting, and streams progress and results back to where it was started.",[],{"_key":383,"_type":384,"caption":385,"image":386},"8a629b9baf1b","imageSection","Figure 1: In the AI Factory, a single change can trigger the whole pipeline from source data to post-trained model. The platform materializes the datasets while the CI builds and publishes the image with the new training mix inside. The training run finds both waiting on a fresh cluster of GPU workers. Each step carries checks that can gate what follows.",{"_type":89,"asset":387},{"_ref":388,"_type":92},"image-0897c58f24caf45bc8630d8ca938d385f9a5c7ca-1022x302-svg",{"_key":390,"_type":242,"text":391},"1c116c00c85b",[392,400,408,422,434,446],{"_key":393,"_type":130,"children":394,"markDefs":399,"style":138},"3c9d2b816cc5",[395],{"_key":396,"_type":134,"marks":397,"text":398},"9db247de8368",[],"Automating the trip from a change to its results enables running more experiments. But the goal of each experiment is to answer a specific question, which takes more than automation. To attribute an outcome to a change, everything else has to hold still, which requires reproducibility. The result of an ML experiment depends on the code, data, dependencies, and runtime environment, and any one of them drifting between two runs can be enough to change the resulting model. A run that cannot be trusted to reflect exactly one change can produce misleading conclusions. Reproducibility is the foundation for turning GPU-hours into durable answers.",[],{"_key":401,"_type":130,"children":402,"markDefs":407,"style":138},"12a33f9ad5c9",[403],{"_key":404,"_type":134,"marks":405,"text":406},"61f3759ec2ab",[],"The AI Factory's automation and reproducibility rest on three design principles:",[],{"_key":409,"_type":130,"children":410,"level":101,"listItem":420,"markDefs":421,"style":138},"56e181518004",[411,416],{"_key":412,"_type":134,"marks":413,"text":415},"d08449922e5a",[414],"strong","The image is the unit of change.",{"_key":417,"_type":134,"marks":418,"text":419},"5ca78500ccbf",[]," Training code and its dependencies are baked into one container image, rebuilt and versioned automatically on every change. The image pins what runs.","bullet",[],{"_key":423,"_type":130,"children":424,"level":101,"listItem":420,"markDefs":433,"style":138},"70e76f000ac4",[425,429],{"_key":426,"_type":134,"marks":427,"text":428},"86d493ce61b1",[414],"One path works everywhere.",{"_key":430,"_type":134,"marks":431,"text":432},"328e8fdf6c9c",[]," A file path resolves to the same data on a GPU cluster, in a development or production deployment, and on a laptop. What it returns depends on access, not on location.",[],{"_key":435,"_type":130,"children":436,"level":101,"listItem":420,"markDefs":445,"style":138},"48b28a7148a5",[437,441],{"_key":438,"_type":134,"marks":439,"text":440},"51d5b9200a81",[414],"Every run gets its own cluster.",{"_key":442,"_type":134,"marks":443,"text":444},"63cd009be8c0",[]," A fresh cluster of GPU workers is created when a run starts and torn down when it ends. Each run gets a clean environment, isolated from all other runs.",[],{"_key":447,"_type":130,"children":448,"markDefs":453,"style":138},"08b14188618a",[449],{"_key":450,"_type":134,"marks":451,"text":452},"daa3b47135e6",[],"Section 2 introduces the components that implement these principles, and describes how they preserve the boundaries that medical data demands. Section 3 shows how the factory speeds up research, making experiments controlled and repeatable, and catching failures before they burn GPU-hours.",[],{"_key":455,"_type":280},"9fef189e0099",{"_key":457,"_type":242,"text":458},"064325f2e7f4",[459,467,475,483,492,500,519,527],{"_key":460,"_type":130,"children":461,"markDefs":466,"style":252},"c9a8966bea5c",[462],{"_key":463,"_type":134,"marks":464,"text":465},"5c07e3d5c0c5",[],"2. A tour of the machinery",[],{"_key":468,"_type":130,"children":469,"markDefs":474,"style":138},"51bf3da7dfbf",[470],{"_key":471,"_type":134,"marks":472,"text":473},"327a059bde86",[],"The AI Factory spans several environments. A change begins on a researcher's laptop and lands in a shared repository. From there, the data and training pipelines that pick it up are coordinated from the factory's control plane. The control plane is hosted on managed Kubernetes, while the training itself happens in separate GPU clusters. The change has to make that trip without quietly changing meaning. Wherever the job runs, the same code, dependencies, and data should produce the same result, even months later.",[],{"_key":476,"_type":130,"children":477,"markDefs":482,"style":138},"cd33c8b4d67f",[478],{"_key":479,"_type":134,"marks":480,"text":481},"f3d2ed4419d3",[],"Three gaps sit along the route: the change has to be integrated into tracked pipelines, the job has to cross from one cluster to another, and the same data has to reach every environment. The AI Factory bridges each gap with a dedicated component.",[],{"_key":484,"_type":130,"children":485,"markDefs":490,"style":491},"f777dc01f007",[486],{"_key":487,"_type":134,"marks":488,"text":489},"e369aa54270f",[],"2.1 The orchestrator",[],"h3",{"_key":493,"_type":130,"children":494,"markDefs":499,"style":138},"18c743999a4b",[495],{"_key":496,"_type":134,"marks":497,"text":498},"14a8bc1eaaaf",[],"The first gap sits between locally developed code and tracked pipelines. A tracked pipeline records what it read, what it produced, and how to run it again. This record is what makes a run reproducible, and only a tracked pipeline keeps it.",[],{"_key":501,"_type":130,"children":502,"markDefs":516,"style":138},"678125823aff",[503,507,512],{"_key":504,"_type":134,"marks":505,"text":506},"6d87df4e131a",[],"We bridge this gap with ",{"_key":508,"_type":134,"marks":509,"text":511},"36af51b37eaa",[510],"619e6e592e9c","Dagster",{"_key":513,"_type":134,"marks":514,"text":515},"c8f7a95d7a47",[],", the factory's central orchestrator. It coordinates work without doing the heavy lifting. All pipelines run through Dagster, from downloading and pre-processing data to submitting training runs; it keeps track of what should exist, launches the work that produces it, and records every run.",[517],{"_key":510,"_type":311,"href":518},"https://dagster.io/",{"_key":520,"_type":130,"children":521,"markDefs":526,"style":138},"e4a21f5d2e9d",[522],{"_key":523,"_type":134,"marks":524,"text":525},"f86ea1691037",[],"Dagster's central abstraction is the asset: something that should exist (e.g., a dataset or model), paired with the code that produces it. Every asset declares which other assets it is built from, so together they form a directed graph that traces the lineage of everything the factory produces.",[],{"_key":528,"_type":130,"children":529,"markDefs":534,"style":138},"17c2bb6c7a20",[530],{"_key":531,"_type":134,"marks":532,"text":533},"aa97e3f1390a",[],"Steps like general-purpose pre-processing are no longer part of a training job. Each step becomes an asset, so the code sits next to the data it produces and the output takes a name and place in the graph. The output outlives the job that made it, ready for other pipelines to build on and easy for any colleague to find. A new dataset enters the graph as a single pull request that declares the source and every step downstream as assets. As an example, Figure 2 shows the resulting graph for a public dataset: one asset downloads the raw files, the next cleans them and normalizes the schema, a third splits the data into training, validation, and test sets, and a final step converts each split into a training-ready format.",[],{"_key":536,"_type":384,"caption":537,"image":538},"752d9b22a679","Figure 2: The lineage of a public dataset in the Dagster asset graph: from the raw source, through cleaning and splitting, to a training-ready format per split. Every step is a tracked asset that records its latest materialization. Asset checks can validate the output of each.",{"_type":89,"asset":539},{"_ref":540,"_type":92},"image-c1bc48573015d38d0568d9a8ed5181187c5f01cf-6161x2696-png",{"_key":542,"_type":242,"text":543},"283df7f2e6e0",[544,552,560,568,576],{"_key":545,"_type":130,"children":546,"markDefs":551,"style":138},"3438e0620c45",[547],{"_key":548,"_type":134,"marks":549,"text":550},"ffce30ad7be1",[],"Assets carry checks: small tests that run every time the asset is rebuilt, asserting that the files, counts, and schema are as expected. A failed asset check can block everything downstream, providing cheap insurance against wasting hours of compute on a GPU cluster.",[],{"_key":553,"_type":130,"children":554,"markDefs":559,"style":138},"3a4d7ff2f9ea",[555],{"_key":556,"_type":134,"marks":557,"text":558},"942862eded70",[],"Not everything is an asset. Data is modeled as assets because a dataset has one current state worth tracking. In contrast, a Dagster job can model an experiment, which varies by settings and produces many outputs, like the checkpoints along the way. Schedules and sensors let the factory start work on its own, producing the same trace as any run started by hand.",[],{"_key":561,"_type":130,"children":562,"markDefs":567,"style":138},"aab4337b993b",[563],{"_key":564,"_type":134,"marks":565,"text":566},"bb456118405f",[],"Assets, jobs, checks, and schedules are all discovered from a single directory of definitions in our repository. The same directory loads on a laptop, in a development deployment, and in production. The pipeline a researcher runs locally is the same pipeline that runs in production. This orchestration underlies the automation and reproducibility that speed up the experimentation loop (Section 3).",[],{"_key":569,"_type":130,"children":570,"markDefs":575,"style":491},"182bc4a2809e",[571],{"_key":572,"_type":134,"marks":573,"text":574},"1898c7a56a1c",[],"2.2 Crossing clusters",[],{"_key":577,"_type":130,"children":578,"markDefs":583,"style":138},"d31180d07276",[579],{"_key":580,"_type":134,"marks":581,"text":582},"705ef40b327e",[],"The second gap sits at the boundary between clusters. While the control plane lives in one cloud, the heavy lifting happens in the GPU clusters of the compute plane. Every run has to cross that boundary, and the machinery we introduce in this section makes the crossing invisible. For the researcher, a run on hundreds of GPUs in another cluster is submitted like any other Dagster job and watched from the same user interface. Figure 3 traces the crossing end to end, from the run leaving Dagster, through the scheduler's queue, to the cluster created for it, and back.",[],{"_key":585,"_type":384,"caption":586,"image":587},"13e107534f5f","Figure 3: The path of one run across the two planes. Dagster submits a RayJob across the boundary, the scheduler admits it, an ephemeral Ray cluster pulls the image and runs the job, and logs and results stream back to Dagster.",{"_type":89,"asset":588},{"_ref":589,"_type":92},"image-baf8e751813a00be43c0cfea144a8114eee2fc77-1120x430-svg",{"_key":591,"_type":242,"text":592},"43f0ad62adff",[593,612,653,672,691,699,707,715,734,742,750,769,777,785,809,817],{"_key":594,"_type":130,"children":595,"markDefs":609,"style":138},"d2346e9c19b4",[596,600,605],{"_key":597,"_type":134,"marks":598,"text":599},"4c1d46861db1",[],"Dagster stays on the control-plane side of the boundary and acts as a client. It submits the run and watches it, but never ships code to the workers. Instead of installing code and dependencies at job start, a step that can fail or drift on any run, the CI bakes them into a container image, rebuilt and versioned on every change. The training itself runs on ",{"_key":601,"_type":134,"marks":602,"text":604},"167b5b57fc46",[603],"51bb14487c31","Ray",{"_key":606,"_type":134,"marks":607,"text":608},"30e8b28d2b7d",[],", a framework for distributing Python workloads across many machines, so what actually travels to the GPU cluster is a RayJob manifest that names the image, the entrypoint command, and the resources the run needs, such as the number of workers and GPUs.",[610],{"_key":603,"_type":311,"href":611},"https://www.ray.io/",{"_key":613,"_type":130,"children":614,"markDefs":646,"style":138},"f1c1e6f001dd",[615,619,624,628,633,637,642],{"_key":616,"_type":134,"marks":617,"text":618},"4144c5d055b3",[],"Our Dagster–Ray integration carries the run across the boundary. It layers two pieces on the open-source ",{"_key":620,"_type":134,"marks":621,"text":623},"b5cf8c45051d",[622],"555332a439ae","dagster-ray",{"_key":625,"_type":134,"marks":626,"text":627},"d55927b04125",[]," library. First, a builder validates a given run config and turns it into the RayJob manifest. Second, a ",{"_key":629,"_type":134,"marks":630,"text":632},"0fc3b1688cb3",[631],"5fdf9b2f8be3","Dagster Pipes",{"_key":634,"_type":134,"marks":635,"text":636},"0e607e77ab30",[]," client launches the external job and streams its logs and results back to the orchestrator. The choice of which cluster the job lands on is one field in the run configuration. The same client resolves the address of the chosen cluster and handles the token authentication against it. The connection itself runs over ",{"_key":638,"_type":134,"marks":639,"text":641},"50b85ae1cd1c",[640],"c83643da335c","Cilium ClusterMesh",{"_key":643,"_type":134,"marks":644,"text":645},"71df67a396fc",[],", which joins the two cluster networks so the orchestrator can reach the remote Ray endpoint as if it were a local service.",[647,649,651],{"_key":622,"_type":311,"href":648},"https://github.com/danielgafni/dagster-ray",{"_key":631,"_type":311,"href":650},"https://docs.dagster.io/integrations/external-pipelines",{"_key":640,"_type":311,"href":652},"https://cilium.io/",{"_key":654,"_type":130,"children":655,"markDefs":669,"style":138},"55cd91d0f608",[656,660,665],{"_key":657,"_type":134,"marks":658,"text":659},"6bd7a43d6298",[],"On the GPU cluster, ",{"_key":661,"_type":134,"marks":662,"text":664},"6ec5b9ced211",[663],"1b4f09cab969","KubeRay",{"_key":666,"_type":134,"marks":667,"text":668},"067dfc91b307",[]," expands the manifest into a fresh Ray cluster, with a head node and GPU workers created for this run and torn down when it finishes. Every run gets its own cluster, so one run cannot interfere with another through a leftover process or a stale cache. And since every run names its own image, two experiments can run different software stacks side by side.",[670],{"_key":663,"_type":311,"href":671},"https://github.com/ray-project/kuberay",{"_key":673,"_type":130,"children":674,"markDefs":688,"style":138},"aaf71d34dbf7",[675,679,684],{"_key":676,"_type":134,"marks":677,"text":678},"d9b7ed3e1d79",[],"We use the ",{"_key":680,"_type":134,"marks":681,"text":683},"95c69687931e",[682],"1e5823562e30","KAI Scheduler",{"_key":685,"_type":134,"marks":686,"text":687},"60ced660c8f9",[],", NVIDIA's open-source scheduler for GPU workloads, to decide when and where each run starts on the shared node pool. It follows written rules: a shared queue, explicit priorities, and preemption for lower-priority work. Gang scheduling places every worker of a job at once or makes the whole job wait, so a multi-node run never sits half-started, holding GPUs it cannot use. Once admitted, the pods pull the image from the registry, Ray runs the entrypoint command, and logs, events, and results stream back to Dagster over Pipes in real time.",[689],{"_key":682,"_type":311,"href":690},"https://github.com/NVIDIA/KAI-Scheduler",{"_key":692,"_type":130,"children":693,"markDefs":698,"style":138},"c0ae8a22c6fd",[694],{"_key":695,"_type":134,"marks":696,"text":697},"a62df20fecf4",[],"For the researcher submitting the job in the Dagster UI, the machinery remains invisible. The run has made the trip across clouds, through a queue, into a cluster of its own, and streamed the results (or failure) back to the same interface.",[],{"_key":700,"_type":130,"children":701,"markDefs":706,"style":491},"1f0f0e648436",[702],{"_key":703,"_type":134,"marks":704,"text":705},"9402b6337ace",[],"2.3 Identical data everywhere",[],{"_key":708,"_type":130,"children":709,"markDefs":714,"style":138},"ad5ea96863a0",[710],{"_key":711,"_type":134,"marks":712,"text":713},"81f0fea8009b",[],"The third gap sits between two competing demands on the data. For training throughput, it belongs close to the compute, ideally in the same data center. But a dataset must also be the same wherever it is read: on the GPU cluster, in the orchestrator, and on a laptop. Copying data into each environment meets the first demand and undermines the second, because every copy is a chance for versions to drift.",[],{"_key":716,"_type":130,"children":717,"markDefs":731,"style":138},"e4b81250e40c",[718,722,727],{"_key":719,"_type":134,"marks":720,"text":721},"7c503b60723d",[],"We close this gap with ",{"_key":723,"_type":134,"marks":724,"text":726},"ad3b4c7e5441",[725],"fd2e0b742d76","Hammerspace",{"_key":728,"_type":134,"marks":729,"text":730},"ceb65daa977f",[],", a layer that presents the storage across our sites as one file system. Instead of each environment keeping its own copy, they all mount one shared namespace. The orchestrator and GPU pods browse the same directory tree, and even a researcher working on their laptop can inspect the exact files a training run reads from the public tree. The tree mirrors the Dagster asset graph (e.g., Figure 2): a dataset's key in the orchestrator is its directory in the namespace, written once by the pipeline that materializes it and never copied anywhere by hand. One path works everywhere, so a dataset path copied from a run log resolves wherever it is pasted.",[732],{"_key":725,"_type":311,"href":733},"https://hammerspace.com/",{"_key":735,"_type":130,"children":736,"markDefs":741,"style":138},"bfcfb046c01e",[737],{"_key":738,"_type":134,"marks":739,"text":740},"a74946f3901a",[],"Underneath the shared namespace, a path says nothing about where the bytes live. Metadata nodes own the directory tree and decide where each file is stored, while data movers carry the bytes across sites when needed. To a job it all looks like ordinary network storage, mounted into its pods. Each Ray cluster starts with the directory tree in place, and reading it requires no library or special API, just a file path.",[],{"_key":743,"_type":130,"children":744,"markDefs":749,"style":138},"e2210df90f02",[745],{"_key":746,"_type":134,"marks":747,"text":748},"c99f105a5d21",[],"The namespace is identical across all environments, but the bytes stay local. At the GPU site, it is backed by fast storage in the same data center. Only background replication ever crosses the site link; every other environment reads the same files from its own local storage. Once a dataset has synced, a read never leaves the cluster. Freshly written data is the flip side: reads do not fail, since missing files are fetched on demand, but can be slow until the sync catches up.",[],{"_key":751,"_type":130,"children":752,"markDefs":766,"style":138},"d566c862a17f",[753,757,762],{"_key":754,"_type":134,"marks":755,"text":756},"cafe00c563b4",[],"Hammerspace ensures that even a modified or newly added dataset is already in place when a training run starts. The result is identical data everywhere, with fast local reads and no copies to drift. ",{"_key":758,"_type":134,"marks":759,"text":761},"de907346722a",[760],"81a03a77e1ce","Meta co-developed a parallel NFS deployment with Hammerspace",{"_key":763,"_type":134,"marks":764,"text":765},"a6ca682e1e26",[]," for the clusters that trained Llama 3, drawn by the same property: a change made anywhere becomes visible at once across all nodes and environments.",[767],{"_key":760,"_type":311,"href":768},"https://engineering.fb.com/2024/03/12/data-center-engineering/building-metas-genai-infrastructure/",{"_key":770,"_type":130,"children":771,"markDefs":776,"style":491},"3c4f3ef4024b",[772],{"_key":773,"_type":134,"marks":774,"text":775},"9128c30d223d",[],"2.4 Access control by construction",[],{"_key":778,"_type":130,"children":779,"markDefs":784,"style":138},"212dd1c00f0b",[780],{"_key":781,"_type":134,"marks":782,"text":783},"7b7fc3563e8d",[],"The previous sections described how a change is carried across environments. When the cargo is sensitive medical data, access control is essential. Access rules are mostly enforced outside the AI Factory, while the factory ensures alignment. Everything it produces stays within those rules by construction.",[],{"_key":786,"_type":130,"children":787,"markDefs":808,"style":138},"53b5a765d861",[788,792,796,800,804],{"_key":789,"_type":134,"marks":790,"text":791},"142ae1325773",[],"The alignment rests on two properties of every asset: its path and the identity that wrote it. The first segment of every asset key is ",{"_key":793,"_type":134,"marks":794,"text":795},"5b59992efc6e",[263],"public",{"_key":797,"_type":134,"marks":798,"text":799},"24f5992a816c",[]," or ",{"_key":801,"_type":134,"marks":802,"text":803},"b1194831a95e",[263],"private",{"_key":805,"_type":134,"marks":806,"text":807},"e7a1392df3bb",[],", and because the key corresponds to the storage path, the classification travels with the data. The private tree is reachable only by the pipelines and projects that own the data. In the trees the factory manages, write access belongs to pipeline identities, so datasets there are, by construction, produced by version-controlled code that has been reviewed and tested.",[],{"_key":810,"_type":130,"children":811,"markDefs":816,"style":138},"04f42d001459",[812],{"_key":813,"_type":134,"marks":814,"text":815},"b8452f4b8eba",[],"The most sensitive data never reaches the factory at all. Incoming data that may contain personal information lands in a restricted zone outside the shared namespace. Only pseudonymized data leaves that zone, crossing into the private tree, where the factory can pick it up.",[],{"_key":818,"_type":130,"children":819,"markDefs":824,"style":138},"6c30656992cd",[820],{"_key":821,"_type":134,"marks":822,"text":823},"718c18377323",[],"One piece of enforcement does live in the factory's code: pipelines derive the classification of an output from the classification of their inputs and refuse a mismatch, so a private dataset cannot end up in the public tree through a wrong prefix. Nobody has to remember which data is sensitive. Reads are decided by the path, writes by identity, wherever the job runs.",[],{"_key":826,"_type":280},"77923e6f3d28",{"_key":828,"_type":242,"text":829},"41e3d3d8a665",[830,838,846,854,862,870,878,886,894,902],{"_key":831,"_type":130,"children":832,"markDefs":837,"style":252},"0e11487a4369",[833],{"_key":834,"_type":134,"marks":835,"text":836},"f086b04862a3",[],"3. Speeding up research",[],{"_key":839,"_type":130,"children":840,"markDefs":845,"style":138},"94f677d3f6c7",[841],{"_key":842,"_type":134,"marks":843,"text":844},"8cebea666c26",[],"The AI Factory carries a change from a laptop into tracked pipelines, across cluster boundaries, and onto machines that already have the data waiting, with logs and results streaming back. This section shows what that machinery enables: experiments that are controlled and repeatable, and checks that can catch broken data, faulty images, and flaky hardware before they hit a production run.",[],{"_key":847,"_type":130,"children":848,"markDefs":853,"style":491},"0cc7420d8ae0",[849],{"_key":850,"_type":134,"marks":851,"text":852},"86678674e52d",[],"3.1 One change at a time",[],{"_key":855,"_type":130,"children":856,"markDefs":861,"style":138},"dae8378e99b6",[857],{"_key":858,"_type":134,"marks":859,"text":860},"9f8ff92698eb",[],"Suppose an experiment shows a curve that looks off compared to a baseline. Is it a side effect of the intended change, or something else that changed between the two runs? The code is usually version-controlled and easy to bisect. But ML experiments also depend on factors that can change silently between two runs, such as the data and code dependencies, which can alter training dynamics and hence the resulting model. In the worst case, hours go into retracing old runs, turning the day into a reproducibility quest.",[],{"_key":863,"_type":130,"children":864,"markDefs":869,"style":138},"c3efd6330ad7",[865],{"_key":866,"_type":134,"marks":867,"text":868},"5efb2dc5679b",[],"Repeatable experiments require pinning four moving parts: code, data, dependencies, and runtime environment. Following the principle that the training image is the unit of change (Section 2.2), code and dependencies, including the GPU libraries, arrive pinned inside the image, which the CI rebuilds whenever either moves and bumps its version in the projects that use it. Additional run configuration, which can include hundreds of hyperparameters, travels separately and is recorded with the run in the orchestrator. A dataset does not have to be tracked separately from its sources, since its lineage records which inputs and code produced it. Together, these mechanisms pin all four moving parts, and any past run can be re-executed with one click in the Dagster UI.",[],{"_key":871,"_type":130,"children":872,"markDefs":877,"style":138},"6c10786154c1",[873],{"_key":874,"_type":134,"marks":875,"text":876},"d6bc20c25132",[],"Besides reproducibility, versioning every input makes it easy to undo changes and to isolate the effect of each. Any change can be rolled back atomically by pointing at an older version of the code or data. To find the root cause of an unexpected result, like the suspicious curve from our previous example, we can vary each part individually. Changes that span multiple parts can be combined in a single pull request. Datasets are declared in the same repository as the code that consumes them, so one rebuild pins both jointly. Even state that does not exist yet, like a dataset declared but not yet materialized, arrives pinned at the same version as the code that will use it.",[],{"_key":879,"_type":130,"children":880,"markDefs":885,"style":138},"b37c168aabb0",[881],{"_key":882,"_type":134,"marks":883,"text":884},"dba3067e7cf0",[],"The differences between a run and its baseline are no longer a mystery but a finite list of candidates. The remaining variation comes from nondeterministic kernels and communication order, not from a silent change. Finding what changed becomes less of a quest and more of a controlled experiment, conducted one change at a time.",[],{"_key":887,"_type":130,"children":888,"markDefs":893,"style":491},"a9979cc10997",[889],{"_key":890,"_type":134,"marks":891,"text":892},"6fcbad3d2b5c",[],"3.2 Trust but verify",[],{"_key":895,"_type":130,"children":896,"markDefs":901,"style":138},"78548b627a93",[897],{"_key":898,"_type":134,"marks":899,"text":900},"92e91e915851",[],"Reproducibility establishes trust, but only after the fact. A failed run still has to be debugged and re-run. The factory also verifies ahead of time, checking data, training images, and cluster hardware before a fault can hit a production run.",[],{"_key":903,"_type":130,"children":904,"markDefs":909,"style":138},"bc509b8ff6bd",[905],{"_key":906,"_type":134,"marks":907,"text":908},"eaf3604ded8a",[],"The data side is handled by asset checks (Section 2.1), which we can now see in action. A check runs against the dataset an asset produces and asserts one property: whether the schema is the one the next stage expects, the required columns are present, the row counts fall in range, or the files are intact. What makes a check more than a warning is that it can be blocking. A failure stops everything downstream, so a broken dataset never becomes an input to training. Figure 4 shows the execution history for one dataset's checks, with a schema check that failed on two earlier runs and passed on the most recent. Those earlier failures are the mechanism doing its work.",[],{"_key":911,"_type":384,"caption":912,"image":913},"8a909ca263bd","Figure 4: Execution history for a dataset's asset checks in the Dagster UI. Both checks succeed on the latest run. On two earlier runs the schema check failed and stopped everything downstream instead of letting bad data through.",{"_type":89,"asset":914},{"_ref":915,"_type":92},"image-275e1713cbc089a5a8502077320315439cc8391d-2770x742-png",{"_key":917,"_type":242,"text":918},"abaa6ce4069e",[919],{"_key":920,"_type":130,"children":921,"markDefs":926,"style":138},"7d718f3652f3",[922],{"_key":923,"_type":134,"marks":924,"text":925},"d2d6f3ee174d",[],"Checking the data is not enough, because the training image itself can break in ways ordinary CI tests cannot catch. Some faults surface only on GPU hardware, when the image meets the driver and the workers connect across nodes. Nightly smoke tests exercise each new image end to end with short training runs, so a packaging or dependency fault shows up on a cheap overnight test rather than crashing a production run the next morning. The schedule is designed not to waste compute: an image that already has a green smoke test is skipped, so each image is verified exactly once. Figure 5 shows this in the schedule's tick history. A night after an image change requests one run, and a night without changes requests none.",[],{"_key":928,"_type":384,"caption":929,"image":930},"5bcaab86d9d9","Figure 5: Tick history for a nightly smoke schedule in the Dagster UI. A new image requests a short end-to-end training run, which can pass or fail. An unchanged image requests zero runs, so each image is verified exactly once.",{"_type":89,"asset":931},{"_ref":932,"_type":92},"image-c1bd0306dffaa959795e011994ee9c75bc405dff-3022x932-png",{"_key":934,"_type":242,"text":935},"00250dae9614",[936,954],{"_key":937,"_type":130,"children":938,"markDefs":951,"style":138},"81db03680bcd",[939,943,948],{"_key":940,"_type":134,"marks":941,"text":942},"0b250f7a2711",[],"The last piece to check is the hardware itself. On a large-scale training run that uses hundreds of GPUs over days or weeks, a single flaky GPU or a degraded link can waste hundreds of GPU-hours. Every hour, a scheduled probe measures the bandwidth between pairs of idle GPU nodes, preemptible below training priority so it never delays a real run. Training runs also carry a canary, which benchmarks the assigned GPUs, NVLink, and InfiniBand fabric before and during training, as described in ",{"_key":944,"_type":134,"marks":945,"text":947},"530631d976fe",[946],"9efb3beb7109","our previous article",{"_key":949,"_type":134,"marks":950,"text":362},"b91719606f88",[],[952],{"_key":946,"_type":311,"href":953},"https://kaiko.ai/insights-hub/Every-Run-Explains-Itself",{"_key":955,"_type":130,"children":956,"markDefs":961,"style":138},"7d1181f50347",[957],{"_key":958,"_type":134,"marks":959,"text":960},"c08c6858ef72",[],"A broken dataset stops at its check, a faulty image fails its smoke test overnight, and a flaky GPU surfaces before training starts. These checks change what researchers spend the day on and establish trust in the underlying infrastructure.",[],{"_key":963,"_type":280},"84715d2982fa",{"_key":965,"_type":242,"text":966},"ca304f77896c",[967,975,983,991],{"_key":968,"_type":130,"children":969,"markDefs":974,"style":252},"5b400782d21b",[970],{"_key":971,"_type":134,"marks":972,"text":973},"76c2b0d4681b",[],"4. Beyond training",[],{"_key":976,"_type":130,"children":977,"markDefs":982,"style":138},"ebc8dfb3fd88",[978],{"_key":979,"_type":134,"marks":980,"text":981},"db37a309290b",[],"Nothing in the AI Factory is specific to training. The same system can automate any workload that has to move reliably across environments. Automated evaluations can score every new checkpoint once a sensor detects it. Rejection sampling can turn model outputs into new training data, declared and versioned like any other asset. Batch inference over an entire archive can run as a downstream job, reading the trained model's weights directly from the shared namespace. Every workload runs on the same machinery, and the target cluster becomes just another field in the run configuration, a choice we plan to automate so each run lands where GPU capacity is available.",[],{"_key":984,"_type":130,"children":985,"markDefs":990,"style":138},"020f8d9e2db7",[986],{"_key":987,"_type":134,"marks":988,"text":989},"b0210e731f59",[],"No honest tour of the machinery ends without a word about the running costs. Every component needs maintenance, and faces a redesign when demands change. The image that pins everything is rebuilt on every change, however small, which requires fast builds. The shared namespace keeps data identical everywhere, but serves freshly written data slowly until a sync catches up. A fresh cluster per run adds minutes of startup, negligible for a long training run but real overhead for a quick experiment. And the factory itself requires continuous maintenance and coordination between platform and research teams. So far, the running costs have been worth it. The factory carries hundreds of jobs a month, from dataset materializations to nightly smoke tests, each traceable to the inputs and code that produced it.",[],{"_key":992,"_type":130,"children":993,"markDefs":998,"style":138},"96f9d3cfee32",[994],{"_key":995,"_type":134,"marks":996,"text":997},"08130d925556",[],"With the experimentation loop running on the AI Factory, the hours saved through automation and reproducibility go into improving the model. The principles that make it work are few: the image is the unit of change, one path works everywhere, and every run gets its own cluster. That foundation lets a small team develop frontier models for the clinic quickly and reliably, within the boundaries that medical data demands.",[],{"_key":1000,"_type":242,"text":1001},"c749c54e5f9b",[1002],{"_key":1003,"_type":130,"children":1004,"markDefs":1009,"style":1010},"891250f9ce51",[1005],{"_key":1006,"_type":134,"marks":1007,"text":1008},"6966271db60b",[],"The models and methods described are research prototypes and have not been approved or cleared as medical devices. They are not intended for clinical diagnosis or patient care.",[],"disclaimer","2026-09-03T07:41:00.000Z",{"_type":89,"asset":1013},{"_ref":1014,"_type":92},"image-fb2fe653ee9ffcbf7948d3378b109f9625892d2e-2011x939-png",false,[1017,1020],{"_key":1018,"_ref":1019,"_type":92},"41ab44fab4f2","5a60d1e9-dfb8-4767-aff8-1524bb04047e",{"_key":1021,"_ref":1022,"_type":92},"e8fce0bbe504","bfafc11e-e363-4983-9c72-822614883107",{"_type":1024,"current":1025},"slug","AI-Factory","Introducing our AI Factory and how it accelerates research cycles, allowing quick and reliable development of our frontier models.",[1028,1679],{"categories":1029,"content":1032,"date":1659,"featuredImage":1660,"slug":1663,"subHeading":1665,"title":1458,"topics":1666},[1030],{"_key":1031,"_ref":238,"_type":92},"4fe50cf56b15",[1033,1087,1089,1156,1158,1209,1215,1282,1284,1422,1428,1447,1449,1468,1474,1493,1495,1530,1536,1606,1608,1650],{"_key":1034,"_type":242,"text":1035},"69d51ba0eb7f",[1036,1043,1051,1063,1075],{"_key":1037,"_type":130,"children":1038,"markDefs":1042,"style":252},"dc546114bad8",[1039],{"_key":1040,"_type":134,"marks":1041,"text":250},"7ae30ff716a1",[],[],{"_key":1044,"_type":130,"children":1045,"markDefs":1050,"style":138},"35daf495c44a",[1046],{"_key":1047,"_type":134,"marks":1048,"text":1049},"fe2e6b919338",[],"On our journey toward training on a trillion tokens, we have learned what matters. In this post we share how we build:",[],{"_key":1052,"_type":130,"children":1053,"level":101,"listItem":420,"markDefs":1062,"style":138},"92d0a7bcc16f",[1054,1058],{"_key":1055,"_type":134,"marks":1056,"text":1057},"b439f69327b0",[414],"A stable and efficient training framework.",{"_key":1059,"_type":134,"marks":1060,"text":1061},"0b806bdb7063",[]," Enabling seamless training resumption after preemption or hardware failure, packing strategies to keep compute on real tokens, and end-to-end testing to catch regressions before they become expensive.",[],{"_key":1064,"_type":130,"children":1065,"level":101,"listItem":420,"markDefs":1074,"style":138},"15b3772e2b21",[1066,1070],{"_key":1067,"_type":134,"marks":1068,"text":1069},"f0908f44052a",[414],"Thorough data observability.",{"_key":1071,"_type":134,"marks":1072,"text":1073},"09d749fd6437",[]," We monitor what actually reaches the model, not just the recipe: the realized mixture and per-source statistics, collected live during the run.",[],{"_key":1076,"_type":130,"children":1077,"level":101,"listItem":420,"markDefs":1086,"style":138},"2045655f21e5",[1078,1082],{"_key":1079,"_type":134,"marks":1080,"text":1081},"c0869ece124f",[414],"Trustworthy benchmarks.",{"_key":1083,"_type":134,"marks":1084,"text":1085},"53af3e506dea",[]," We don't take a score at face value: aggressive decontamination of the training data against every benchmark, and recording the generations behind a number.",[],{"_key":1088,"_type":280},"705ca7d0a22e",{"_key":1090,"_type":242,"text":1091},"7c4a8e4216bf",[1092,1100,1108,1116,1124,1132,1140,1148],{"_key":1093,"_type":130,"children":1094,"markDefs":1099,"style":252},"970c73cf1b5b",[1095],{"_key":1096,"_type":134,"marks":1097,"text":1098},"a8bc1dde2f73",[],"Introduction",[],{"_key":1101,"_type":130,"children":1102,"markDefs":1107,"style":138},"ec6bcaa383e8",[1103],{"_key":1104,"_type":134,"marks":1105,"text":1106},"55a1bc359f8b",[],"Building frontier agentic reasoning models for clinical work means training a generalist model to navigate clinical workflows end-to-end: reasoning over long patient records, calling diagnostic tools, navigating radiology viewers, scrolling through whole-slide pathology images, and connecting what it learns from those tools back to a clinical question. These are tools that off-the-shelf foundation models never see during training and are not natively calibrated to operate.",[],{"_key":1109,"_type":130,"children":1110,"markDefs":1115,"style":138},"e1bbd08160e0",[1111],{"_key":1112,"_type":134,"marks":1113,"text":1114},"148332c41fcc",[],"In a little over a year, we moved from training dense ~7B parameter multimodal vision-language models on a few hundred million tokens to 100B-token runs on 10-100x larger MoE models, with a roadmap toward trillion-token-scale continued pre-training (CPT). In our approach, the model architecture is treated as fixed. By benchmarking several open weight models that reported data sources and performances transparently, we settled on a strong base. This leaves the data as the primary variable we control in CPT and post-training. In what follows, we will therefore take a data-centric view on the stack that we’ve built.",[],{"_key":1117,"_type":130,"children":1118,"markDefs":1123,"style":138},"0feb0297fe27",[1119],{"_key":1120,"_type":134,"marks":1121,"text":1122},"2f65f92a8251",[],"The first decision is the mixture itself: a three-way tug-of-war between installing new clinical knowledge, retaining the base's general skills, and warming up its latent ones. The rest is the engineering that delivers that mixture at scale:",[],{"_key":1125,"_type":130,"children":1126,"level":101,"listItem":420,"markDefs":1131,"style":138},"899d88d4a02c",[1127],{"_key":1128,"_type":134,"marks":1129,"text":1130},"cf5869a8c799",[],"a robust training framework for 100B+ MoE models that tracks not just training progress but the dynamics, hardware utilization, and data statistics of a run",[],{"_key":1133,"_type":130,"children":1134,"level":101,"listItem":420,"markDefs":1139,"style":138},"7ba6d98887b7",[1135],{"_key":1136,"_type":134,"marks":1137,"text":1138},"a0e19315943e",[],"efficient data streaming to focus the compute where it matters",[],{"_key":1141,"_type":130,"children":1142,"level":101,"listItem":420,"markDefs":1147,"style":138},"3fa8ad03a873",[1143],{"_key":1144,"_type":134,"marks":1145,"text":1146},"2a27f23ca539",[],"observability to track down any failure and resume from a given state across data, model, and hardware (and to see what the model actually trained on)",[],{"_key":1149,"_type":130,"children":1150,"level":101,"listItem":420,"markDefs":1155,"style":138},"4deb1bb0c509",[1151],{"_key":1152,"_type":134,"marks":1153,"text":1154},"3219294c411b",[],"an evaluation signal we can trust: broadly mapping the performance landscape we care about and rigorously decontaminating the test data",[],{"_key":1157,"_type":280},"43567e705a3f",{"_key":1159,"_type":242,"text":1160},"c634445876a4",[1161,1169,1186],{"_key":1162,"_type":130,"children":1163,"markDefs":1168,"style":252},"106f6f5434aa",[1164],{"_key":1165,"_type":134,"marks":1166,"text":1167},"4faf2a1ff06e",[],"Mid-training as a three-way tug-of-war",[],{"_key":1170,"_type":130,"children":1171,"markDefs":1185,"style":138},"85155a6b66aa",[1172,1176,1181],{"_key":1173,"_type":134,"marks":1174,"text":1175},"43cbeef9d998",[],"Rather than pre-training models from scratch, we start from capable models that have already been instruction-tuned and optimized for reasoning, and tool-use. We have previously verified that this yields stronger and better-performing models than starting from a checkpoint that was only pre-trained. However, a naïve continued pre-training on raw biomedical domain data would easily jeopardize the model’s learnt capabilities. To preserve these learnt skills, we craft a careful mix of raw, web-scale data (both text-only, image-captioning, and interleaved data), instruction-formatted Q&A and VQA, multi-turn conversations (with and without reasoning traces and images), as well as tool-use data. These instruction-formatted data components lean heavily on the general domain, while the raw, unformatted data is predominantly composed of domain-specific biomedical and clinical data intended to teach the model domain-specific knowledge. Technically, what we do is CPT, yet our data mixture moves this closer to what is often understood as mid-training, see e.g. in Liu et al. (2025)",{"_key":1177,"_type":134,"marks":1178,"text":1180},"cb53f06731d0",[1179],"sup","1",{"_key":1182,"_type":134,"marks":1183,"text":1184},"4bbd257e7830",[],". Conventional CPT would rely exclusively on in-domain data and is often accompanied by a sharp performance drop in orthogonal domains – something that we aim to prevent by carefully crafting a balanced data mix.",[],{"_key":1187,"_type":130,"children":1188,"markDefs":1206,"style":138},"3eab3a9d729e",[1189,1193,1198,1202],{"_key":1190,"_type":134,"marks":1191,"text":1192},"e7924949fda1",[],"Thus, we view our CPT stage as a three-way tug-of-war between warmup, skill retention, and the target domain shift; see also ",{"_key":1194,"_type":134,"marks":1195,"text":1197},"2d0705ed7503",[1196],"5e7f12df0c22","PRISM",{"_key":1199,"_type":134,"marks":1200,"text":1201},"0c8932053595",[1179],"2",{"_key":1203,"_type":134,"marks":1204,"text":1205},"e9e15b276cc4",[]," for an in-depth analysis of the mid-training paradigm. Here we will briefly outline each of these directions:",[1207],{"_key":1196,"_type":311,"href":1208},"https://arxiv.org/abs/2603.17074",{"_key":1210,"_type":384,"caption":1211,"image":1212},"3df31b978f06","Mid-training as a tug-of-war between three data types, each defined by which two properties it combines. The circles are the properties (biomedical content, web-scale volume, instruction/tool-use/reasoning format); each data type sits in a pairwise overlap: domain shift (biomedical + web-scale) reinforces clinical knowledge, retention (web-scale + instruction) guards the base's general skills, warmup (biomedical + instruction) wakes latent clinical skills. Each pulls the mixture toward it with a force set by its abundance, so warmup, scarce and expensive to obtain, pulls with the thinnest rope. The triple overlap, representing web-scale biomedical instruction data, is not readily available and we reserve this for a future blog post on synthetic data generation for post-training.",{"_type":89,"asset":1213},{"_ref":1214,"_type":92},"image-c6b872846dd7f33141ab065174d0ecad477d3940-728x462-svg",{"_key":1216,"_type":242,"text":1217},"01cd7e1f72a0",[1218,1234,1250,1274],{"_key":1219,"_type":130,"children":1220,"markDefs":1233,"style":138},"3fabde72f18d",[1221,1225,1229],{"_key":1222,"_type":134,"marks":1223,"text":1224},"13fd156cf704",[],"The base model already has latent clinical core skills (instruction following, medical reasoning, basic tool-use), but calibrated for consumer-facing chatbots rather than assisting in professional clinical workflows. ",{"_key":1226,"_type":134,"marks":1227,"text":1228},"b305ca4180bd",[414],"Warmup",{"_key":1230,"_type":134,"marks":1231,"text":1232},"e6936aee7d2e",[]," turns up the model's responsiveness on these skills by exposing it to clinical-shaped versions of them, e.g. reasoning traces injected into biomedical articles, or clinical VQA.",[],{"_key":1235,"_type":130,"children":1236,"markDefs":1249,"style":138},"f0253901d30c",[1237,1241,1245],{"_key":1238,"_type":134,"marks":1239,"text":1240},"c22fb66d7184",[],"The second direction is ",{"_key":1242,"_type":134,"marks":1243,"text":1244},"dff92df3f7ba",[414],"retention",{"_key":1246,"_type":134,"marks":1247,"text":1248},"24e91c1880f0",[],": The base already knows English and a few other languages, math, code, and has some broad \"world knowledge\". We can't afford to erode any of that while chasing clinical capability. Catastrophic forgetting is the classic version of this risk. Our mixture is shaped accordingly: A large fraction of the data mix is general instruction-formatted data that uplifts the skills which otherwise would erode under the heavy weight of biomedical data.",[],{"_key":1251,"_type":130,"children":1252,"markDefs":1273,"style":138},"d40f040fe896",[1253,1257,1261,1265,1269],{"_key":1254,"_type":134,"marks":1255,"text":1256},"65bc8e908242",[],"Finally, we adapt the model to the clinical domain by exposing it to the knowledge and structures of its target domain: biomedical terminology, the shape of EHRs, the literature, the conventions and guidelines of clinical reasoning. This is the ",{"_key":1258,"_type":134,"marks":1259,"text":1260},"410e5284309c",[414],"domain shift",{"_key":1262,"_type":134,"marks":1263,"text":1264},"26c2981fcf45",[]," we are targeting and what most people think of when they hear ",{"_key":1266,"_type":134,"marks":1267,"text":1268},"747daafbab57",[263],"continued pre-training",{"_key":1270,"_type":134,"marks":1271,"text":1272},"c17d0cac1ee9",[],": raw, web-scale domain data.",[],{"_key":1275,"_type":130,"children":1276,"markDefs":1281,"style":138},"ba6a77785636",[1277],{"_key":1278,"_type":134,"marks":1279,"text":1280},"f85eef49fe42",[],"Each data type pulls the model in a different direction and striking the right balance to meet the criteria of a production-grade clinical model requires detailed understanding of the intended uses of the model, but equally important is a thorough understanding of the data mix in training. We will come back to this momentarily.",[],{"_key":1283,"_type":280},"8360cdaf10a7",{"_key":1285,"_type":242,"text":1286},"3b0612e19a60",[1287,1295,1303,1311,1319,1331,1343,1351,1359,1367,1383,1395],{"_key":1288,"_type":130,"children":1289,"markDefs":1294,"style":252},"2de0048034bc",[1290],{"_key":1291,"_type":134,"marks":1292,"text":1293},"872be5594ca6",[],"Building the foundation",[],{"_key":1296,"_type":130,"children":1297,"markDefs":1302,"style":138},"e010aac9b769",[1298],{"_key":1299,"_type":134,"marks":1300,"text":1301},"cd7d3e113836",[],"Our aim for the training framework is that scaling up model size and data is a change of configuration, not a reinvention of the wheel. Everything above rests on it, so the infrastructure underneath has to be solid. Here’s what we concluded on.",[],{"_key":1304,"_type":130,"children":1305,"markDefs":1310,"style":252},"0d5e1577609d",[1306],{"_key":1307,"_type":134,"marks":1308,"text":1309},"58f0fbcd59a2",[],"Training framework and data loading",[],{"_key":1312,"_type":130,"children":1313,"markDefs":1318,"style":138},"2429fb33025d",[1314],{"_key":1315,"_type":134,"marks":1316,"text":1317},"8079b16ff675",[],"After evaluating the alternatives, we settled on building on top of NVIDIA's NeMo framework: Megatron-Bridge for training and Energon for data loading, respectively. What we needed was production-readiness at scale with enough integration surface to layer our own callbacks, logging, and recipe defaults on top, and both Megatron-Bridge and Energon fit: they are fully integrated from data loading, model training, and performance tracking, all the way to checkpoint resumption, and at the same time allow us to build on top, add custom training logic and logging where we need it. As a bonus, NVIDIA's NeMo comes with a range of low-level hardware accelerations that result in 5x faster training iterations compared to our old research training framework built on top of PyTorch Lightning.",[],{"_key":1320,"_type":130,"children":1321,"markDefs":1330,"style":138},"338523a84915",[1322,1326],{"_key":1323,"_type":134,"marks":1324,"text":1325},"f0a387393db2",[414],"Tests at every scale of integration.",{"_key":1327,"_type":134,"marks":1328,"text":1329},"3e7c8c20e52c",[]," While unit tests ensure that every component in the pipeline behaves as expected, unforeseen long-range interactions both within one team and across team boundaries can lead to undesirable outcomes. Therefore, we test end-to-end: For any change to the data or the code we trigger integration tests for the entire pipeline including short training runs and full iterations of the affected datasets to ensure continuity in our production training runs.",[],{"_key":1332,"_type":130,"children":1333,"markDefs":1342,"style":138},"9b4f97296995",[1334,1338],{"_key":1335,"_type":134,"marks":1336,"text":1337},"9e0baf1e6048",[414],"Byte-identical resumability.",{"_key":1339,"_type":134,"marks":1340,"text":1341},"f62d3f6b2bfe",[]," A multi-week run will be interrupted at some point, either by preemption, hardware failure, or a planned restart. This is where the NeMo universe shines: We persist the full dataloader state alongside every model checkpoint, thereby ensuring byte-identical behavior between the uninterrupted run and its continuation.",[],{"_key":1344,"_type":130,"children":1345,"markDefs":1350,"style":252},"2bf9ba0cb06b",[1346],{"_key":1347,"_type":134,"marks":1348,"text":1349},"d7a762e1c697",[],"Packing and token efficiency",[],{"_key":1352,"_type":130,"children":1353,"markDefs":1358,"style":138},"abbfbc797779",[1354],{"_key":1355,"_type":134,"marks":1356,"text":1357},"63d23b52b4e4",[],"Why production-readiness matters at this scale is best illustrated by putting numbers to it. One example that's easy to overlook is sequence packing.",[],{"_key":1360,"_type":130,"children":1361,"markDefs":1366,"style":138},"29813ce26ead",[1362],{"_key":1363,"_type":134,"marks":1364,"text":1365},"e6d75aae884f",[],"Biomedical and clinical corpora are wildly heterogeneous in document length: a PubMed abstract runs around 200 tokens, clinical notes vary broadly (500-3K), full papers and tool traces easily exceed 10K. Packing sequentially loaded samples from such a heterogeneous distribution into a standard pre-training sequence of length, say, 4k tokens, inevitably results in a substantial amount of padding tokens: long sequences are truncated, long and short sequences get combined into sub-optimal combinations. To alleviate this, we use a greedy Knapsack algorithm that selects packing candidates optimally from a pre-loaded buffer. Moreover, we choose a packing strategy that minimizes the padding content. Three packing regimes are worth comparing:",[],{"_key":1368,"_type":130,"children":1369,"level":101,"listItem":420,"markDefs":1382,"style":138},"374fc10da694",[1370,1374,1378],{"_key":1371,"_type":134,"marks":1372,"text":1373},"5f8528990b14",[414],"No packing.",{"_key":1375,"_type":134,"marks":1376,"text":1377},"6c7307634e50",[]," One document per sequence, padded out to the context length. On a biomedical document-length mix at 4K context, realistic padding waste is ",{"_key":1379,"_type":134,"marks":1380,"text":1381},"00595e8629fc",[414],"30-40%.",[],{"_key":1384,"_type":130,"children":1385,"level":101,"listItem":420,"markDefs":1394,"style":138},"575ade0b68a1",[1386,1390],{"_key":1387,"_type":134,"marks":1388,"text":1389},"3084a27232c9",[414],"Naïve packing.",{"_key":1391,"_type":134,"marks":1392,"text":1393},"aacf4e83e498",[]," Concatenate samples into one sequence, no attention masking. Padding waste drops to near zero, but attention across document boundaries can contaminate the gradient signal and has sub-optimal memory scaling.",[],{"_key":1396,"_type":130,"children":1397,"level":101,"listItem":420,"markDefs":1419,"style":138},"5a74c027d94a",[1398,1402,1406,1411,1415],{"_key":1399,"_type":134,"marks":1400,"text":1401},"7cc761332000",[414],"Boundary-aware packing.",{"_key":1403,"_type":134,"marks":1404,"text":1405},"74f97f6436e5",[]," Concatenate samples into a single sequence, but mask attention so each token only attends to others within its own document. This is the approach chosen by us and other frontier labs, see e.g. ",{"_key":1407,"_type":134,"marks":1408,"text":1410},"cd23054625a3",[1409],"1d1c311fb0db","DeepSeek-V4",{"_key":1412,"_type":134,"marks":1413,"text":1414},"21210b1e6afa",[1179],"3",{"_key":1416,"_type":134,"marks":1417,"text":1418},"89641ab631f8",[],". It guarantees near-zero padding waste and each document is treated as an individual sequence.",[1420],{"_key":1409,"_type":311,"href":1421},"https://arxiv.org/abs/2606.19348",{"_key":1423,"_type":384,"caption":1424,"image":1425},"445ea7d6238d","Top: six documents of different lengths. One document per sequence (left) leaves 30 - 40% of tokens as padding (note that for multi-node training, a global sequence length must be fixed); concatenating the documents (right) removes most of it, with a small residual if sequences are split only at paragraph or sentence boundaries. Bottom: the causal attention matrix over one packed sequence of three documents (A, B, C). Within-document attention (blue) is identical in both cases; naïve packing (left) additionally attends across document boundaries (gradient contamination and suboptimal memory scaling), while boundary-aware packing (right) masks those cross-document positions, so each document is treated as an individual sequence.",{"_type":89,"asset":1426},{"_ref":1427,"_type":92},"image-fd14a1c0d14abf45a5e9c08332953644f2cf7535-960x780-svg",{"_key":1429,"_type":242,"text":1430},"40929a7023b1",[1431],{"_key":1432,"_type":130,"children":1433,"markDefs":1446,"style":138},"d19cc193a915",[1434,1438,1442],{"_key":1435,"_type":134,"marks":1436,"text":1437},"20935da01ddb",[],"In GPU-hours this would mean the following. The public Qwen3.5-VL 122B-A10B SFT recipe runs at global batch size 36, with a maximum sequence length of 4,096, for 300k steps: roughly 44B training tokens, or about 2'100 GPU-hours on 48 H100s at 35% MFU. At a worst-case 40% padding waste, ~840 of those GPU-hours are spent processing padded positions that contribute no useful training signal. Scaling the same math up, a 100B-token run wastes around 1'900 GPU-hours per recipe ablation, and the 1T-token target wastes around 19'000. That's roughly ",{"_key":1439,"_type":134,"marks":1440,"text":1441},"92e681c9e784",[414],"$76K per run",{"_key":1443,"_type":134,"marks":1444,"text":1445},"ed6e374b63f6",[]," at typical H100 cloud pricing of around 4$/GPU-hour.",[],{"_key":1448,"_type":280},"9a8765b71e4c",{"_key":1450,"_type":242,"text":1451},"8d3ef62cd2fb",[1452,1460],{"_key":1453,"_type":130,"children":1454,"markDefs":1459,"style":252},"2f1d3ec018ee",[1455],{"_key":1456,"_type":134,"marks":1457,"text":1458},"449673190e49",[],"What did the model actually see?",[],{"_key":1461,"_type":130,"children":1462,"markDefs":1467,"style":138},"c935a02a3b9e",[1463],{"_key":1464,"_type":134,"marks":1465,"text":1466},"2aa229c788ff",[],"One subtlety matters here specifically since the realized and planned data mixtures can deviate substantially: The planned mix is a list of sources with their target shares, a set of filter parameters, and a training schedule. However, what reaches the model might deviate substantially from this planned mix. We illustrate this with the figure below (left panel). A declared (sample level) recipe would contain 71% web-scale biomedical text data; however, after transformations and filters are applied, sequences have been packed, and bad or uninformative samples removed (think of corrupted images that previously weren’t caught), the true token level mixture looks quite different, with a sharply decreased share of the web scale text dataset. Further looking at the loss token share per dataset, the tension relaxes but still strongly differs from the declared mix.",[],{"_key":1469,"_type":384,"caption":1470,"image":1471},"e7e060cc1dba","A real run (~20B tokens; sources anonymized), counted three ways. Left: the same realized data weighted by sample, by token, and by loss-token. By sample it matches the declared blend (71% biomedical text / 10% biomedical image-text / 19% skill additions); by token, biomedical text falls to 43% as token-dense skill data more than doubles its share; by loss-token (prompt, system, and image tokens masked out) biomedical text climbs back to 51%, while masked tool-use data collapses from 12% to 5%, far less than its token share suggests. Right: sources differ in size, so a small tool-use subset is cycled ~9 times before the largest corpus finishes a fifth of one pass.",{"_type":89,"asset":1472},{"_ref":1473,"_type":92},"image-22de86cabced38eb9772f42314b21a3c93b7acdb-1120x460-svg",{"_key":1475,"_type":242,"text":1476},"be90546e868a",[1477,1485],{"_key":1478,"_type":130,"children":1479,"markDefs":1484,"style":138},"89efc4513b11",[1480],{"_key":1481,"_type":134,"marks":1482,"text":1483},"9cf280f773d2",[],"Another aspect that is crucial to monitor is the number of times each dataset in the mix is seen. A first best judgement blend will be altered through the aforementioned dynamics and consequently we might be seeing the same samples more than intended. The right panel of the figure illustrates this, where a small tool-use dataset is heavily oversampled. The mechanism is as follows: While the dataset might have been correctly weighted by token share (tool-use data tends to come with fewer samples but very long sequences), many of the samples would exceed the context window of the initial stage of CPT already with the system message and user prompt, which depending on the training configuration may leave no tokens relevant for gradient calculation at all and thus are filtered on the fly. The resulting over-representation of the dataset in the mix requires a subsequent re-adjustment.",[],{"_key":1486,"_type":130,"children":1487,"markDefs":1492,"style":138},"4b03cf391aaf",[1488],{"_key":1489,"_type":134,"marks":1490,"text":1491},"335716809993",[],"The cost of being wrong grows with run size. At the trillion-token scale we're working toward, a 1% drift in the realized mixture is 10B tokens of unintended training spent on the wrong distribution (about the entire budget of a respectable open research run). Our token-level observability stack ensures that any drift is visible at training time and not hidden in a silent performance drift in evaluation.",[],{"_key":1494,"_type":280},"c70574fa1114",{"_key":1496,"_type":242,"text":1497},"78af5b6faf74",[1498,1506,1514,1522],{"_key":1499,"_type":130,"children":1500,"markDefs":1505,"style":252},"7aae45ba6e15",[1501],{"_key":1502,"_type":134,"marks":1503,"text":1504},"85ca49ba3bf3",[],"A trustworthy evaluation signal",[],{"_key":1507,"_type":130,"children":1508,"markDefs":1513,"style":138},"4be2a7958365",[1509],{"_key":1510,"_type":134,"marks":1511,"text":1512},"dd6e8aa47276",[],"Checkpoint validation is our main signal for whether the upstream work paid off, so it carries the same engineering investment as training and data.",[],{"_key":1515,"_type":130,"children":1516,"markDefs":1521,"style":138},"22e53ea0f1e8",[1517],{"_key":1518,"_type":134,"marks":1519,"text":1520},"f0c25d2f4d9c",[],"Contamination is a sharper risk for us than for a research model. A research model is judged on a fixed, public benchmark set it can decontaminate against; our models are judged by users on clinical tasks we may not have anticipated. Therefore, we evaluate against both critical public benchmarks and internal private ones, and keep adding and revising them as the product surface grows. One uncaught test-set leak turns that signal into a misleading one.",[],{"_key":1523,"_type":130,"children":1524,"markDefs":1529,"style":138},"d4e39d0546a2",[1525],{"_key":1526,"_type":134,"marks":1527,"text":1528},"77274cdcdeb2",[],"Therefore, we deduplicate every training source against every active benchmark. Exact matching runs on every pair; we add the costly fuzzy and semantic passes where source and benchmark draw on shared material. Overlap on the rest is tracked as a tripwire, escalating a pair if it climbs above a threshold.",[],{"_key":1531,"_type":384,"caption":1532,"image":1533},"a508836edf99","Leakage risk varies across (training source × benchmark) pairs, so we target the expensive matching rather than spread it evenly. Every pair is deduplicated by exact match; where a source and benchmark draw on shared material (high-risk pairs) we add the full fuzzy and semantic stack (dense-embedding matching for images). Overlap on the rest is tracked as a tripwire and escalated if it climbs. New sources and benchmarks are matched against everything active as they come online.",{"_type":89,"asset":1534},{"_ref":1535,"_type":92},"image-7ef14539e1371164e7cf6dbda271faf29106a01e-780x470-svg",{"_key":1537,"_type":242,"text":1538},"e04dea550fa1",[1539,1547,1559,1578,1598],{"_key":1540,"_type":130,"children":1541,"markDefs":1546,"style":138},"a74f34b0467f",[1542],{"_key":1543,"_type":134,"marks":1544,"text":1545},"a6f511a6172f",[],"A trustworthy signal needs two things: a benchmark the model has not already seen, and a way to read why a score moved. For the first, we escalate the matcher, each step catching what the one before it cannot:",[],{"_key":1548,"_type":130,"children":1549,"level":101,"listItem":420,"markDefs":1558,"style":138},"8355dc6045b8",[1550,1554],{"_key":1551,"_type":134,"marks":1552,"text":1553},"883ddbe07a8c",[414],"Exact match.",{"_key":1555,"_type":134,"marks":1556,"text":1557},"b97e43c51e87",[]," Verbatim copies fall out of a hash match, run on every (source × benchmark) pair.",[],{"_key":1560,"_type":130,"children":1561,"level":101,"listItem":420,"markDefs":1577,"style":138},"0dba7b474d3d",[1562,1566,1570,1574],{"_key":1563,"_type":134,"marks":1564,"text":1565},"5296869c12eb",[414],"Fuzzy matching.",{"_key":1567,"_type":134,"marks":1568,"text":1569},"4789fc6e165f",[]," Reformatted and reworded copies survive it: MinHash-LSH surfaces near-duplicates by Jaccard overlap",{"_key":1571,"_type":134,"marks":1572,"text":1573},"b025fbb6c976",[1179],"4",{"_key":1575,"_type":134,"marks":1576,"text":362},"5228c870ab07",[],[],{"_key":1579,"_type":130,"children":1580,"level":101,"listItem":420,"markDefs":1597,"style":138},"e717ab4975c6",[1581,1585,1589,1593],{"_key":1582,"_type":134,"marks":1583,"text":1584},"1c449255b409",[414],"Semantic matching.",{"_key":1586,"_type":134,"marks":1587,"text":1588},"407039cff2af",[]," What survives fuzzy is caught by embeddings: a cosine pass over text",{"_key":1590,"_type":134,"marks":1591,"text":1592},"9f4a886e94d7",[1179],"5",{"_key":1594,"_type":134,"marks":1595,"text":1596},"f0f305801ae8",[],", and dense image embeddings with approximate nearest-neighbour search that flag a training image answering a benchmark question even when its pixels differ. Both carry a higher false-positive cost, so we reserve them for pairs where a source and benchmark share material, as illustrated in the figure.",[],{"_key":1599,"_type":130,"children":1600,"markDefs":1605,"style":138},"a130dec621f4",[1601],{"_key":1602,"_type":134,"marks":1603,"text":1604},"0c39c1608f5c",[],"For the second point, we store all generations and not just the score. We pull actual generations from each checkpoint and inspect a few of them by hand and with an LLM judge at scale, because a drop in a benchmark has many possible causes a single number hides. A common example is a drop in performance, where the model gives the right answer in the wrong format, so the failure is instruction following, not knowledge. A score alone cannot tell the two apart.",[],{"_key":1607,"_type":280},"0fa9e390015a",{"_key":1609,"_type":242,"text":1610},"204aae08f9bd",[1611,1619,1627,1642],{"_key":1612,"_type":130,"children":1613,"markDefs":1618,"style":252},"dd7d700a4605",[1614],{"_key":1615,"_type":134,"marks":1616,"text":1617},"3111078cd748",[],"Conclusion and outlook",[],{"_key":1620,"_type":130,"children":1621,"markDefs":1626,"style":138},"d6a2d91f522f",[1622],{"_key":1623,"_type":134,"marks":1624,"text":1625},"4dd4e4410448",[],"Every stage in this post acts on the same data mixture: the framework and packing deliver it, observability records what the model actually trained on, and decontaminated evaluation shows whether it worked. Each component is built to hold as runs scale, so moving from today's 100B-token runs to the trillion-token target on 100B+ MoE models means changing the configuration, not rewriting the stack. What remains is to compose them into a single loop.",[],{"_key":1628,"_type":130,"children":1629,"markDefs":1641,"style":138},"8ac1185c23f4",[1630,1634,1637],{"_key":1631,"_type":134,"marks":1632,"text":1633},"1de81c4b8bf6",[],"That composition is what we are building toward in kaiko's ",{"_key":1635,"_type":134,"marks":1636,"text":264},"60d25eb0e4a7",[414],{"_key":1638,"_type":134,"marks":1639,"text":1640},"6552009fc270",[],", a pipeline where evaluation results feed back into the recipe directly, the recipe as a versioned, structured artifact and the observability primitives from the preceding sections making each stage's output legible to the next. The orchestration already runs on Dagster, workloads are distributed via Ray, and Kubernetes provisions and schedules the resources. The components exist today (framework, loader, packing, decontamination, a trustworthy evaluation signal, an emerging recipe artifact). Composing them into a single closed loop is the work ahead, and the subject of a future deep dive.",[],{"_key":1643,"_type":130,"children":1644,"markDefs":1649,"style":138},"f6c7f0283de9",[1645],{"_key":1646,"_type":134,"marks":1647,"text":1648},"1a31159132d3",[],"None of these practices are novel on their own. The literature has assembled most of the pieces. What rarely makes it into a paper are the steps behind a headline: preempted and crashed runs, the mixture drifting from the recipe, or a dataset silently cycled nine times. With this article, we want to shed some light on what it takes to move toward the trillion-token frontier.",[],{"_key":1651,"_type":1652,"references":1653},"81a6af91cf49","referencesSection",[1654,1655,1656,1657,1658],"Liu et al. 2025, Midtraining Bridges Pretraining and Posttraining Distributions, \nhttps://arxiv.org/abs/2510.14865","PRISM: Demystifying Retention and Interaction in Mid-Training, https://arxiv.org/abs/2603.17074","DeepSeek-V4: Towards Highly Efficient Million-Token Context Intelligence\nhttps://arxiv.org/abs/2606.19348","Lee et al. 2021 Deduplicating Training Data Makes Language Models Better, https://arxiv.org/abs/2107.06499 ","Abbas et al. 2023, SemDeDup: Data-efficient learning at web-scale through semantic deduplication, https://arxiv.org/abs/2303.09540","2026-07-09T16:12:54.910Z",{"_type":89,"asset":1661},{"_ref":1662,"_type":92},"image-d9a09026393fbffde93f88b44b36fe2f68e105d2-2010x938-png",{"_type":1024,"current":1664},"Training-clinical-reasoning-model-at-scale","Toward training at the trillion-token scale: Why reliability, stability, and observability matter and how we engineer them.",[1667,1670,1673,1676],{"_key":1668,"_ref":1669,"_type":92},"3b76795f501c","ffc879d7-9be3-47d1-bda5-5817c9d7cf8c",{"_key":1671,"_ref":1672,"_type":92},"c2dc7fdd784c","0e7a76c9-607b-4b29-b0f5-48940b955680",{"_key":1674,"_ref":1675,"_type":92},"a476af4cdeb6","038e94a3-7be4-458d-972c-d3b572c832a0",{"_key":1677,"_ref":1678,"_type":92},"279baccd7561","5f860d5a-1a0d-4041-ab25-eb07132ec4bd",{"categories":1680,"content":1683,"date":2229,"featuredImage":2230,"slug":2233,"subHeading":2235,"title":2236,"topics":2237},[1681],{"_key":1682,"_ref":238,"_type":92},"1e07a03563b1",[1684,1726,1728,1776,1778,1797,1803,1853,1859,1861,2001,2007,2018,2020,2062,2064,2106,2112,2114,2172,2174,2193,2195,2221],{"_key":1685,"_type":242,"text":1686},"54c03faa8c20",[1687,1694,1702,1714],{"_key":1688,"_type":130,"children":1689,"markDefs":1693,"style":252},"bb5dda5ea66a",[1690],{"_key":1691,"_type":134,"marks":1692,"text":250},"65039dbf0fc4",[],[],{"_key":1695,"_type":130,"children":1696,"markDefs":1701,"style":138},"59f86d52869f",[1697],{"_key":1698,"_type":134,"marks":1699,"text":1700},"a1ed96e0391b",[],"Training large multimodal models for clinical reasoning can involve hundreds of GPUs running for weeks. At that scale, small performance regressions are costly, and distributed failures can take hours to diagnose. This post describes two parts of our training infrastructure designed to address these problems:",[],{"_key":1703,"_type":130,"children":1704,"level":101,"listItem":420,"markDefs":1713,"style":138},"8d28d3d3106a",[1705,1709],{"_key":1706,"_type":134,"marks":1707,"text":1708},"9004312eba27",[414],"Performance diagnosis.",{"_key":1710,"_type":134,"marks":1711,"text":1712},"1328bf8cd8cf",[]," A profiler captures a few steps in kernel-level detail and reduces the trace to a named bottleneck. A lightweight monitor tracks the full run, from model FLOPs utilization (MFU) to per-collective communication time, so regressions are visible as they happen.",[],{"_key":1715,"_type":130,"children":1716,"level":101,"listItem":420,"markDefs":1725,"style":138},"e001fa666cdf",[1717,1721],{"_key":1718,"_type":134,"marks":1719,"text":1720},"c14ad6e9a547",[414],"Operational reliability.",{"_key":1722,"_type":134,"marks":1723,"text":1724},"21eeb8621045",[]," A canary tests the allocated compute, network, and storage before and periodically during training. A health monitor records progress on every rank and combines evidence from a distributed failure into one report, identifying either a likely originating rank or a shared fault domain.",[],{"_key":1727,"_type":280},"5056d2a20d08",{"_key":1729,"_type":242,"text":1730},"696a31ed110f",[1731,1738,1746,1754,1768],{"_key":1732,"_type":130,"children":1733,"markDefs":1737,"style":252},"0f84f22cd790",[1734],{"_key":1735,"_type":134,"marks":1736,"text":1098},"82eefd5d3cc0",[],[],{"_key":1739,"_type":130,"children":1740,"markDefs":1745,"style":138},"9bb70c46a1f8",[1741],{"_key":1742,"_type":134,"marks":1743,"text":1744},"f24d364a2561",[],"Training large multimodal models for clinical reasoning means running jobs on hundreds of GPUs across dozens of nodes, for days, sometimes weeks. At that scale, performance depends on every part of the distributed system. Training is synchronous, so a slow GPU, network link, or input pipeline can hold back the entire job. The same clusters run a steady stream of smaller experiments, so slowdowns and failures affect both large training runs and how fast we can test new ideas.",[],{"_key":1747,"_type":130,"children":1748,"markDefs":1753,"style":138},"91d9caf92250",[1749],{"_key":1750,"_type":134,"marks":1751,"text":1752},"9ad9ccf444a1",[],"Understanding why a run is slow requires visibility at different timescales. A profiler can explain a few steps in detail, but overhead and trace volume make continuous capture impractical. Dashboards track throughput and utilization over the full run but do not show where step time is lost. Cluster health checks can report that a node is available without revealing that a GPU is throttling, an interconnect is underperforming, or storage cannot keep up with the workload.",[],{"_key":1755,"_type":130,"children":1756,"markDefs":1767,"style":138},"9d3f9031320b",[1757,1761,1764],{"_key":1758,"_type":134,"marks":1759,"text":1760},"c41176e5f8e5",[],"Failures create a different problem. An error on one rank (one GPU process) often causes communication timeouts on others, leaving many secondary errors but little indication of the original fault. Diagnosing the incident can require comparing worker and cluster logs, scheduler events, dashboards, and hardware telemetry across the allocation. By the time the investigation begins, the scheduler may have already torn the job down and removed the most useful local state. None of this is rare at scale. Meta reported an unexpected interruption roughly every three hours during the 54-day pretraining of Llama 3 405B",{"_key":1762,"_type":134,"marks":1763,"text":1180},"4bb443e587f6",[1179],{"_key":1765,"_type":134,"marks":1766,"text":362},"e9883e24149e",[],[],{"_key":1769,"_type":130,"children":1770,"markDefs":1775,"style":138},"68c6e3580801",[1771],{"_key":1772,"_type":134,"marks":1773,"text":1774},"c5bee8d05815",[],"We built our own observability layer around a single principle: a run should explain itself. We record traces, timers, hardware counters, and per-rank heartbeats while the job is running and preserve them through teardown, so why a run is slow and why it failed is evident from what it recorded.",[],{"_key":1777,"_type":280},"1bd2a830a91d",{"_key":1779,"_type":242,"text":1780},"0f6d4d9cb9b6",[1781,1789],{"_key":1782,"_type":130,"children":1783,"markDefs":1788,"style":252},"09730133ced2",[1784],{"_key":1785,"_type":134,"marks":1786,"text":1787},"55a6e2ca325f",[],"Two pillars, four instruments",[],{"_key":1790,"_type":130,"children":1791,"markDefs":1796,"style":138},"865a4441cdac",[1792],{"_key":1793,"_type":134,"marks":1794,"text":1795},"9857101e73f3",[],"To make that possible, we use four components, grouped under performance and operational reliability (Figure 1).",[],{"_key":1798,"_type":384,"caption":1799,"image":1800},"27a16faf404c","Figure 1: The four components, grouped into two pillars. Left (performance): a profiler for deep, occasional captures, paired with a lightweight monitor that runs the whole job. Right (operational reliability): a canary that benchmarks the assigned hardware before and during the run, paired with a health monitor that watches every rank. All four attach to the training run and write to shared storage and the experiment tracker, where one diagnosis layer reads them.",{"_type":89,"asset":1801},{"_ref":1802,"_type":92},"image-9158682e1d202313b1dc67c462dfdf744fb4a163-3736x1444-webp",{"_key":1804,"_type":242,"text":1805},"fc8a6e3076ba",[1806,1822,1845],{"_key":1807,"_type":130,"children":1808,"markDefs":1821,"style":138},"a4292b632f40",[1809,1813,1817],{"_key":1810,"_type":134,"marks":1811,"text":1812},"1ae32d7d993c",[],"The ",{"_key":1814,"_type":134,"marks":1815,"text":1816},"07040119abed",[263],"profiler",{"_key":1818,"_type":134,"marks":1819,"text":1820},"bf8830cd677c",[]," records a short window of kernel, memory, and communication activity in full detail. The performance monitor tracks throughput, framework timers, and GPU counters for the entire run, providing one summary line per logging interval.",[],{"_key":1823,"_type":130,"children":1824,"markDefs":1844,"style":138},"bd4ba7d8e97c",[1825,1828,1832,1836,1840],{"_key":1826,"_type":134,"marks":1827,"text":1812},"156fd1eb5226",[],{"_key":1829,"_type":134,"marks":1830,"text":1831},"68c136d529fd",[263],"canary",{"_key":1833,"_type":134,"marks":1834,"text":1835},"aa2e05be67f9",[]," benchmarks compute, interconnect, and storage on the allocated nodes before training begins. It repeats a smaller set of checks during training. The ",{"_key":1837,"_type":134,"marks":1838,"text":1839},"b9a9896cf8ba",[263],"health monitor",{"_key":1841,"_type":134,"marks":1842,"text":1843},"b6831cdc19ec",[]," keeps a heartbeat and a watchdog on every rank throughout. It performs a few cheap in-memory updates per step, enough to tell at any instant whether a rank is still making progress. It collects failure evidence the moment one stops.",[],{"_key":1846,"_type":130,"children":1847,"markDefs":1852,"style":138},"fb6630c1f609",[1848],{"_key":1849,"_type":134,"marks":1850,"text":1851},"8ffe8b98277c",[],"The four are paired because neither kind is sufficient alone: profiling and full benchmarks are too heavy to leave on, while the continuous signals cannot explain what they detect. In practice, the monitor or canary notices a change, and the profiler or health monitor explains it. All four write their outputs to shared storage and the experiment tracker in formats the same analysis code reads through a single API, so after a slowdown or a crash the evidence is already in one place. Figure 2 shows when each component runs.",[],{"_key":1854,"_type":384,"caption":1855,"image":1856},"91fa62ccedeb","Figure 2: Time runs left to right across four lanes, one per component. The canary gates the start with a pre-flight check and re-checks periodically; the performance monitor emits one line every logging interval for the whole run; the profiler captures a short, deep window; the health monitor heartbeats continuously and, at the incident on the right, writes a report before the allocation is torn down.",{"_type":89,"asset":1857},{"_ref":1858,"_type":92},"image-4cd0c2c7d8d82b11e73a86dd23f47c1b6df71d21-3736x1141-webp",{"_key":1860,"_type":280},"8c6c3416d51b",{"_key":1862,"_type":242,"text":1863},"52f9bcdb2b9f",[1864,1872,1880,1888,1909,1917,1929,1941,1953,1965,1977,1985,1993],{"_key":1865,"_type":130,"children":1866,"markDefs":1871,"style":252},"618f66ef0523",[1867],{"_key":1868,"_type":134,"marks":1869,"text":1870},"f8f45a56cf73",[],"Anatomy of a slow step",[],{"_key":1873,"_type":130,"children":1874,"markDefs":1879,"style":138},"e9c99edb19f0",[1875],{"_key":1876,"_type":134,"marks":1877,"text":1878},"144642059b34",[],"A useful profile should end with a diagnosis, for example: non-overlapped NCCL work accounts for a third of this step, making it communication-bound. Producing that diagnosis automatically is harder than collecting the trace. A few steps of a multi-node mixture-of-experts model generate hundreds of thousands of events across dozens of GPU streams.",[],{"_key":1881,"_type":130,"children":1882,"markDefs":1887,"style":138},"55e272fcda60",[1883],{"_key":1884,"_type":134,"marks":1885,"text":1886},"4e4672aeb696",[],"We use the PyTorch profiler by default: it records a fixed set of steps and exports a Kineto trace, per-rank kernel summaries, a categorized memory timeline, and an allocator snapshot whose stack traces survive an out-of-memory crash. For system-level or individual-kernel analysis, we use Nsight Systems or Nsight Compute, with custom NVTX ranges marking the relevant model regions.",[],{"_key":1889,"_type":130,"children":1890,"markDefs":1906,"style":138},"20c7f504dbe2",[1891,1895,1900,1903],{"_key":1892,"_type":134,"marks":1893,"text":1894},"4cbaeeff1450",[],"Each trace is parsed on the node where it was produced and reduced to a few kilobytes. This includes GPU busy and idle time, the compute/NCCL/memcpy split, the kernels consuming the most device time, and peak memory per device. This is the same reduction problem behind Meta's ",{"_key":1896,"_type":134,"marks":1897,"text":1899},"f2de379ef9da",[1898],"678cbc6ee804","Holistic Trace Analysis",{"_key":1901,"_type":134,"marks":1902,"text":1201},"7d97747e0fcd",[1179],{"_key":1904,"_type":134,"marks":1905,"text":362},"c7bfbbaf5d77",[],[1907],{"_key":1898,"_type":311,"href":1908},"https://pytorch.org/blog/trace-analysis-for-masses/",{"_key":1910,"_type":130,"children":1911,"markDefs":1916,"style":138},"d5e1947bd6b3",[1912],{"_key":1913,"_type":134,"marks":1914,"text":1915},"51d172dfa2f4",[],"A priority-ordered classifier maps those measurements to a coarse bottleneck category:",[],{"_key":1918,"_type":130,"children":1919,"level":101,"listItem":420,"markDefs":1928,"style":138},"bcb1fc74bc9b",[1920,1924],{"_key":1921,"_type":134,"marks":1922,"text":1923},"7dde53ca0aa4",[414],"Launch-bound",{"_key":1925,"_type":134,"marks":1926,"text":1927},"5da3edc9327d",[]," — high idle time and many short kernels: the CPU is not issuing GPU work quickly enough.",[],{"_key":1930,"_type":130,"children":1931,"level":101,"listItem":420,"markDefs":1940,"style":138},"d6bd47e62137",[1932,1936],{"_key":1933,"_type":134,"marks":1934,"text":1935},"33d2f8e11e2d",[414],"Input-bound",{"_key":1937,"_type":134,"marks":1938,"text":1939},"32760946b48c",[]," — high idle time with normal kernel durations: the input pipeline is not keeping up.",[],{"_key":1942,"_type":130,"children":1943,"level":101,"listItem":420,"markDefs":1952,"style":138},"cc4bc25941a3",[1944,1948],{"_key":1945,"_type":134,"marks":1946,"text":1947},"70b91461536c",[414],"Communication-bound",{"_key":1949,"_type":134,"marks":1950,"text":1951},"e6288d308660",[]," — NCCL consumes a large share of\nbusy time; per-collective timings determine how much sits on the\ncritical path rather than overlapping compute.",[],{"_key":1954,"_type":130,"children":1955,"level":101,"listItem":420,"markDefs":1964,"style":138},"7a70cd0b81e6",[1956,1960],{"_key":1957,"_type":134,"marks":1958,"text":1959},"82f94cda2cef",[414],"Transfer-bound",{"_key":1961,"_type":134,"marks":1962,"text":1963},"7c433ea6f4ea",[]," — host-device copies consume a large share of the step.",[],{"_key":1966,"_type":130,"children":1967,"level":101,"listItem":420,"markDefs":1976,"style":138},"4cff56eab01e",[1968,1972],{"_key":1969,"_type":134,"marks":1970,"text":1971},"7f9b6ff56d72",[414],"Compute-bound",{"_key":1973,"_type":134,"marks":1974,"text":1975},"233545ec4c24",[]," — none of the previous conditions applies.",[],{"_key":1978,"_type":130,"children":1979,"markDefs":1984,"style":138},"7911ce340f48",[1980],{"_key":1981,"_type":134,"marks":1982,"text":1983},"32b53ce37496",[],"These labels describe the step as a whole. A compute-bound step can still contain kernels limited by memory bandwidth, occupancy, or instruction dependencies. Nsight Compute provides that finer distinction. Communication time also needs cross-rank context. If one rank, pipeline stage, or expert receives more work, faster ranks spend longer waiting in collectives, and an imbalance elsewhere ends up counted as NCCL time. We separate the two by comparing compute and input time across ranks.",[],{"_key":1986,"_type":130,"children":1987,"markDefs":1992,"style":138},"eba6045565f4",[1988],{"_key":1989,"_type":134,"marks":1990,"text":1991},"c567e2e11eb1",[],"The classifier always returns its supporting measurements, and its thresholds are versioned and tested so an engineer can inspect or override a verdict without reopening the full trace.",[],{"_key":1994,"_type":130,"children":1995,"markDefs":2000,"style":138},"26595fcbea51",[1996],{"_key":1997,"_type":134,"marks":1998,"text":1999},"3c1930560e44",[],"Figure 3 shows one of our training runs before and after acting on a communication-bound verdict. The optimized step is shorter, and its compute share rises because communication and idle time fall relative to the shorter total.",[],{"_key":2002,"_type":384,"caption":2003,"image":2004},"e1ed637ac962","Figure 3: One training step before and after acting on a communication-bound verdict, here by overlapping communication with compute. Top: the step as two lanes, compute and communication; each block is a group of kernels, and the gaps between blocks are idle time. In the baseline, communication stalls compute; optimized, it runs underneath compute instead, so the step finishes ~32% sooner. Bottom: the same two steps as rings, inner ring by category and outer ring by the largest kernels within it. These are single operations such as one matmul or one all-reduce, showing which specific kernels dominate the category. Each is normalized to its own step so they show where time goes rather than absolute duration.",{"_type":89,"asset":2005},{"_ref":2006,"_type":92},"image-9ddaca3358b7b71933392cb13376f6f7d38434a3-3760x1960-webp",{"_key":2008,"_type":242,"text":2009},"5f12e195cf48",[2010],{"_key":2011,"_type":130,"children":2012,"markDefs":2017,"style":138},"283280d49f9f",[2013],{"_key":2014,"_type":134,"marks":2015,"text":2016},"2483316e4596",[],"We use the verdict to choose the next experiment. Communication bottlenecks lead to overlap or collective changes. Launch bottlenecks lead to fusion or graph capture. Input bottlenecks lead to the loader and storage path. When an individual operation remains expensive, we write a Triton or CUDA replacement and keep it in an in-house kernel library that the training stack pulls from instead of the stock op. Every change is re-profiled before it is kept.",[],{"_key":2019,"_type":280},"647d6f88bcf0",{"_key":2021,"_type":242,"text":2022},"96733b40086b",[2023,2031,2046,2054],{"_key":2024,"_type":130,"children":2025,"markDefs":2030,"style":252},"d283294a9a4e",[2026],{"_key":2027,"_type":134,"marks":2028,"text":2029},"9edecb7e2c25",[],"Monitoring performance over a full run",[],{"_key":2032,"_type":130,"children":2033,"markDefs":2045,"style":138},"a395aa7291e0",[2034,2038,2041],{"_key":2035,"_type":134,"marks":2036,"text":2037},"de04678ace26",[],"A profiler samples a few steps by design: depending on the configuration, profiling adds roughly 5–30% overhead, and on a healthy run most captured data is redundant. So we pair it with a lightweight performance monitor that runs for the full job. The key metric over that horizon is model FLOPs utilization (MFU)",{"_key":2039,"_type":134,"marks":2040,"text":1414},"fdc4a04ae716",[1179],{"_key":2042,"_type":134,"marks":2043,"text":2044},"e626ecc04460",[],": an MFU of 50% means the achieved model throughput is roughly half of the GPU’s theoretical peak at the training precision. We use it to track workload over time and compare related configurations. It is more informative than GPU utilization alone since a GPU can stay fully utilized while spending time on communication or inefficient kernels.",[],{"_key":2047,"_type":130,"children":2048,"markDefs":2053,"style":138},"e586309b668f",[2049],{"_key":2050,"_type":134,"marks":2051,"text":2052},"baddcc5096ad",[],"MFU on its own cannot explain a regression. To locate the cause, the monitor fuses hardware counters from DCGM (SM and tensor-core activity, occupancy, memory) with the framework's timers (compute, communication, pipeline bubbles, data loading) and adds timers where the framework is too coarse.",[],{"_key":2055,"_type":130,"children":2056,"markDefs":2061,"style":138},"50d7ff3711d6",[2057],{"_key":2058,"_type":134,"marks":2059,"text":2060},"8a97e5a49069",[],"Communication is where those extra timers matter most. An NCCL call returns to the host almost immediately while the exchange runs asynchronously on the GPU's communication stream, so a host-side timer captures little more than the kernel launch. Instead, the monitor brackets each collective with CUDA events on its own stream and reads back the elapsed device time. This captures how long the exchange really took and how much overlapped with compute rather than stalling it. Tagged with its process group (data, pipeline, expert, or tensor parallel), this turns communication from one opaque bar into per-collective time attributed to the part of the parallel configuration that produced it.",[],{"_key":2063,"_type":280},"f3cd2adc6584",{"_key":2065,"_type":242,"text":2066},"8c48a49dafcb",[2067,2075,2090,2098],{"_key":2068,"_type":130,"children":2069,"markDefs":2074,"style":252},"dda896ec3d95",[2070],{"_key":2071,"_type":134,"marks":2072,"text":2073},"adf172999fbc",[],"The hardware is guilty until measured",[],{"_key":2076,"_type":130,"children":2077,"markDefs":2089,"style":138},"9ef641154fd8",[2078,2082,2085],{"_key":2079,"_type":134,"marks":2080,"text":2081},"79b9730e1280",[],"A node can pass cluster health checks yet run below expected performance: a GPU throttling under sustained load, an NVLink or PCIe link operating below expected width, or a storage mount delivering insufficient throughput. Because each step includes synchronous collectives that all ranks must complete, the slowest affected rank can reduce throughput for the whole job",{"_key":2083,"_type":134,"marks":2084,"text":1573},"2ab3c4dcdcf0",[1179],{"_key":2086,"_type":134,"marks":2087,"text":2088},"5276eac7a7ac",[],", and nothing crashes or reports unhealthy while this happens.",[],{"_key":2091,"_type":130,"children":2092,"markDefs":2097,"style":138},"4396e8b23fbb",[2093],{"_key":2094,"_type":134,"marks":2095,"text":2096},"c968b3e3bb01",[],"The canary tests the nodes and topology assigned to the run by executing inside the training process group itself. Before the first step, it benchmarks per-GPU compute throughput, intra-node and inter-node collective bandwidth, and storage read throughput, and records clocks, thermals, throttling state, and hardware error counters. Results are reported per rank, and the lowest-performing rank determines whether the allocation passes. A failed pre-flight check blocks training from starting, allowing the job to be rescheduled so a bad node costs minutes to reschedule around rather than days of degraded throughput.",[],{"_key":2099,"_type":130,"children":2100,"markDefs":2105,"style":138},"72e47f9087dc",[2101],{"_key":2102,"_type":134,"marks":2103,"text":2104},"e2077ce155cb",[],"During training, it periodically repeats a smaller set of checks. Measurements taken under load are usually lower than idle pre-flight results. Two mechanisms sit on top: hard thresholds relaxed to a fraction of the pre-flight value, which catch outright failure, and an exponentially weighted moving average of each measurement. This average flags sustained deviation from the pre-flight distribution and catches gradual degradation, such as a link losing bandwidth over hours, long before the relaxed threshold fires. Figure 4 shows one such event.",[],{"_key":2107,"_type":384,"caption":2108,"image":2109},"675a9c9c9694","Figure 4: Each dot is a periodic in-flight measurement of collective bandwidth, as a fraction of the pre-flight baseline (vertical axis) over the course of the run (horizontal axis). Because individual samples are noisy, the canary tracks their exponentially weighted moving average (the trend line). As a link slowly degrades, that trend crosses its expected band and flags the problem hours before the measurement would trip the relaxed hard threshold; the gap between the two is the early warning.",{"_type":89,"asset":2110},{"_ref":2111,"_type":92},"image-2068d3130e582185c5991ac53453eac6f0b12390-3736x1490-webp",{"_key":2113,"_type":280},"44ea4a42f0e0",{"_key":2115,"_type":242,"text":2116},"0a333f75aad2",[2117,2125,2133,2141,2149,2157],{"_key":2118,"_type":130,"children":2119,"markDefs":2124,"style":252},"1d9a4c0493af",[2120],{"_key":2121,"_type":134,"marks":2122,"text":2123},"ba25c4469b93",[],"When a run dies at 3am",[],{"_key":2126,"_type":130,"children":2127,"markDefs":2132,"style":138},"4ec45a2e04f1",[2128],{"_key":2129,"_type":134,"marks":2130,"text":2131},"06bcc6cb43c9",[],"Distributed failures produce secondary errors that often obscure the original fault. In an asymmetric failure, one rank fails first and the remaining ranks time out later in a collective. In a correlated failure, several ranks fail together because they share a host, switch, storage dependency, or scheduler event. In both cases the allocation may be torn down before the relevant local state is retained.",[],{"_key":2134,"_type":130,"children":2135,"markDefs":2140,"style":138},"f0acc5bc135d",[2136],{"_key":2137,"_type":134,"marks":2138,"text":2139},"2dfe92fea179",[],"The health monitor runs on every rank. Background threads record the current step, phase, and timestamp, sample GPU state into a ring buffer, and run a progress watchdog. The training path performs a few in-memory updates and no synchronous I/O.",[],{"_key":2142,"_type":130,"children":2143,"markDefs":2148,"style":138},"b5fa30b7c509",[2144],{"_key":2145,"_type":134,"marks":2146,"text":2147},"518876f941ce",[],"When a rank crashes, or the watchdog sees progress stop, the monitor collects every thread's stack, recent GPU state, PyTorch's NCCL flight-recorder entries, the other ranks' latest heartbeats, and the local exception, and writes them into one structured incident report on shared storage. A rule-based classifier labels the common cases: out of memory, software exception, data-pipeline stall, storage timeout, collective mismatch or deadlock, straggler, node failure.",[],{"_key":2150,"_type":130,"children":2151,"markDefs":2156,"style":138},"15a40e9b48c2",[2152],{"_key":2153,"_type":134,"marks":2154,"text":2155},"b4fd24940604",[],"Two real crashes show what this looks like in practice. One run died at step 62; the raw stack said only that a data-loader iterator had failed, and the report unwrapped the worker's traceback to the underlying stale file handle, so the cause was the shared filesystem briefly dropping its mounts mid-read. Another run died at step 2000 with an NCCL error in its logs; the report showed the CUDA allocations behind a collective failing because the GPU had run out of memory. The raw logs of the two runs look similar, and the reports separated them into a storage fault and an out-of-memory failure.",[],{"_key":2158,"_type":130,"children":2159,"markDefs":2171,"style":138},"23ea6700f670",[2160,2164,2167],{"_key":2161,"_type":134,"marks":2162,"text":2163},"c375f8a21792",[],"Cross-rank timing is most useful for asymmetric failures. The rank whose heartbeat stops first, carrying an exception the others do not share, is usually the source, and the timeouts that follow trace back to it. For correlated failures, the report groups affected ranks by shared host, switch, or dependency, since there may be no originating rank. In one incident, every rank reported a communication timeout, but one had stopped heartbeating about thirty seconds before the others inside the expert-routing all-to-all dispatch",{"_key":2165,"_type":134,"marks":2166,"text":1592},"328448631b88",[1179],{"_key":2168,"_type":134,"marks":2169,"text":2170},"a43031ac1f91",[],". The ranks exchange data-dependent amounts of expert traffic and must first agree on the exchange sizes. Padding to a fixed expert capacity made the sizes static and removed the synchronization point where the run hung.",[],{"_key":2173,"_type":280},"45c7f253a685",{"_key":2175,"_type":242,"text":2176},"afac641382cf",[2177,2185],{"_key":2178,"_type":130,"children":2179,"markDefs":2184,"style":252},"ecc9b1b371d6",[2180],{"_key":2181,"_type":134,"marks":2182,"text":2183},"e842f7940824",[],"Runs that read their own reports",[],{"_key":2186,"_type":130,"children":2187,"markDefs":2192,"style":138},"59d79b3c142f",[2188],{"_key":2189,"_type":134,"marks":2190,"text":2191},"b08e6ebfda99",[],"All of this was built for people first, but the same artifacts are exposed as tools over the Model Context Protocol so diagnosis can be invoked programmatically too. An agent can pull a crash report and apply the same odd-one-out triage, trigger the on-node trace analysis, and return the verdict with its evidence. It can also watch throughput and flag a sustained regression. The instruments supply the measurements, the agent summarizes them, and any action that costs cluster time still goes through a human. The practical effect is that routine questions no longer require an engineer to open a raw trace.",[],{"_key":2194,"_type":280},"97c50caa5910",{"_key":2196,"_type":242,"text":2197},"a98a1e58d342",[2198,2205,2213],{"_key":2199,"_type":130,"children":2200,"markDefs":2204,"style":252},"50036cb64539",[2201],{"_key":2202,"_type":134,"marks":2203,"text":1617},"5f4ca35ba542",[],[],{"_key":2206,"_type":130,"children":2207,"markDefs":2212,"style":138},"bbad1fbf3f1b",[2208],{"_key":2209,"_type":134,"marks":2210,"text":2211},"0c8b5722b1f6",[],"The tools presented in this post address two recurring ways a run wastes the hardware it holds: running below its capacity, and failing in a way that is slow to diagnose. The profiler and performance monitor show where step time goes and whether performance changes during a run. The canary verifies the assigned hardware before training and detects degradation while the job runs. The health monitor records per-rank progress and preserves enough evidence to distinguish an originating error from the failures that follow it.",[],{"_key":2214,"_type":130,"children":2215,"markDefs":2220,"style":138},"c2230ee3cd7e",[2216],{"_key":2217,"_type":134,"marks":2218,"text":2219},"962cef98a990",[],"Folding them into the closed loop of our AI Factory is the work ahead and the subject of a future post.",[],{"_key":2222,"_type":1652,"references":2223},"3fdfd4cf995b",[2224,2225,2226,2227,2228],"Dubey et al. 2024. The Llama 3 Herd of Models. arXiv:2407.21783.","PyTorch Trace Analysis for the Masses, https://pytorch.org/blog/trace-analysis-for-masses/ ","Chowdhery et al. 2022. PaLM: Scaling Language Modeling with Pathways. arXiv:2204.02311.","Jiang et al. 2024. MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs. USENIX NSDI 2024. arXiv:2402.15627.","DeepSeek-AI 2024. DeepSeek-V3 Technical Report. arXiv:2412.19437.","2026-07-29T13:00:00.000Z",{"_type":89,"asset":2231},{"_ref":2232,"_type":92},"image-57eb27fab7979b0e35b5be4803e54af4367c9ce6-2010x938-png",{"_type":1024,"current":2234},"Every-Run-Explains-Itself","How we instrument large training runs to reveal where performance is lost, trace failures to their source, and verify the hardware beneath them.","Every run explains itself",[2238,2241],{"_key":2239,"_ref":2240,"_type":92},"dc9c8fc1e4c9","f006e877-46a9-4d49-9d08-ad783336cf59",{"_key":2242,"_ref":1678,"_type":92},"d7df9b62181d","Inside the Medical AI Factory",[2245,2247,2250],{"_key":2246,"_ref":1678,"_type":92},"936cec153639",{"_key":2248,"_ref":2249,"_type":92},"bb0fbacd7d8f","87071b12-f012-4898-9928-e78a41e6f88e",{"_key":2251,"_ref":2252,"_type":92},"c20679ac4829","88cf361a-c42a-4eca-9903-4758b4e3de98",{"documents":2254,"paths":2258,"mappings":2270},[2255,2256,2257],{"_id":222,"_type":227},{"_id":1019,"_type":227},{"_id":1022,"_type":227},[145,146,147,148,149,150,2259,2260,2261,2262,2263,2264,2265,2266,2267,2268,2269],"$['area']","$['authors']","$['categories']","$['content']","$['date']","$['featuredImage']","$['relatedPosts']","$['slug']","$['subHeading']","$['title']","$['topics']",{"$['_createdAt']":2271,"$['_id']":2273,"$['_rev']":2275,"$['_system']":2277,"$['_type']":2279,"$['_updatedAt']":2281,"$['area']":2283,"$['authors']":2285,"$['categories']":2287,"$['content']":2289,"$['date']":2291,"$['featuredImage']":2293,"$['relatedPosts']":2295,"$['slug']":2297,"$['subHeading']":2299,"$['suggestedPosts'][0]['categories']":2301,"$['suggestedPosts'][0]['content']":2303,"$['suggestedPosts'][0]['date']":2305,"$['suggestedPosts'][0]['featuredImage']":2307,"$['suggestedPosts'][0]['slug']":2309,"$['suggestedPosts'][0]['subHeading']":2311,"$['suggestedPosts'][0]['title']":2313,"$['suggestedPosts'][0]['topics']":2315,"$['suggestedPosts'][1]['categories']":2317,"$['suggestedPosts'][1]['content']":2319,"$['suggestedPosts'][1]['date']":2321,"$['suggestedPosts'][1]['featuredImage']":2323,"$['suggestedPosts'][1]['slug']":2325,"$['suggestedPosts'][1]['subHeading']":2327,"$['suggestedPosts'][1]['title']":2329,"$['suggestedPosts'][1]['topics']":2331,"$['title']":2333,"$['topics']":2335},{"source":2272,"type":168},{"document":166,"path":166,"type":167},{"source":2274,"type":168},{"document":166,"path":101,"type":167},{"source":2276,"type":168},{"document":166,"path":173,"type":167},{"source":2278,"type":168},{"document":166,"path":176,"type":167},{"source":2280,"type":168},{"document":166,"path":179,"type":167},{"source":2282,"type":168},{"document":166,"path":182,"type":167},{"source":2284,"type":168},{"document":166,"path":185,"type":167},{"source":2286,"type":168},{"document":166,"path":188,"type":167},{"source":2288,"type":168},{"document":166,"path":191,"type":167},{"source":2290,"type":168},{"document":166,"path":194,"type":167},{"source":2292,"type":168},{"document":166,"path":197,"type":167},{"source":2294,"type":168},{"document":166,"path":200,"type":167},{"source":2296,"type":168},{"document":166,"path":203,"type":167},{"source":2298,"type":168},{"document":166,"path":206,"type":167},{"source":2300,"type":168},{"document":166,"path":209,"type":167},{"source":2302,"type":168},{"document":101,"path":191,"type":167},{"source":2304,"type":168},{"document":101,"path":194,"type":167},{"source":2306,"type":168},{"document":101,"path":197,"type":167},{"source":2308,"type":168},{"document":101,"path":200,"type":167},{"source":2310,"type":168},{"document":101,"path":206,"type":167},{"source":2312,"type":168},{"document":101,"path":209,"type":167},{"source":2314,"type":168},{"document":101,"path":212,"type":167},{"source":2316,"type":168},{"document":101,"path":218,"type":167},{"source":2318,"type":168},{"document":173,"path":191,"type":167},{"source":2320,"type":168},{"document":173,"path":194,"type":167},{"source":2322,"type":168},{"document":173,"path":197,"type":167},{"source":2324,"type":168},{"document":173,"path":200,"type":167},{"source":2326,"type":168},{"document":173,"path":206,"type":167},{"source":2328,"type":168},{"document":173,"path":209,"type":167},{"source":2330,"type":168},{"document":173,"path":212,"type":167},{"source":2332,"type":168},{"document":173,"path":218,"type":167},{"source":2334,"type":168},{"document":166,"path":212,"type":167},{"source":2336,"type":168},{"document":166,"path":218,"type":167},1788425554335]