[{"data":1,"prerenderedAt":2248},["ShallowReactive",2],{"sanity-7iXP3sOrSyrwpMoyaopqutrI_r2a7XNibHtOh578tGQ":3,"sanity-KTAcbUy3h4KQ1mawecZK6AJRI0xh4hlQANNjSRj_GgE":219},{"data":4,"sourceMap":139},{"_createdAt":5,"_id":6,"_rev":7,"_system":8,"_type":6,"_updatedAt":11,"defaultTitle":12,"footerMenu1":13,"footerMenu2":29,"footerMenu3":44,"gitHubUrl":53,"huggingFaceUrl":57,"linkedInUrl":49,"mainNavigation":59,"ogDescription":87,"ogImage":88,"ogImageUrl":93,"showSiteNotice":94,"siteNotice":95},"2026-01-21T14:58:26Z","siteSettingsSchema","mbou5Mk4jD84jdX2vzI6cY",{"base":9},{"id":6,"rev":10},"6X7ayGJgs0U64XMIw22UCy","2026-07-30T09:11:27Z","kaiko.ai",{"links":14,"title":28},[15,20,24],{"_key":16,"_type":17,"label":18,"url":19},"95cf27020263","navigationItem","Clinical AI Workspace","/clinical-ai-workspace",{"_key":21,"_type":17,"label":22,"url":23},"7b5bdfebb176","Research","/research",{"_key":25,"_type":17,"label":26,"url":27},"d311ee16a390","Insights Hub","/insights-hub","What we do",{"links":30,"title":43},[31,35,39],{"_key":32,"_type":17,"label":33,"url":34},"44a37ad9151d","About us","/about",{"_key":36,"_type":17,"label":37,"url":38},"193660568291","Careers","https://jobs.kaiko.ai/",{"_key":40,"_type":17,"label":41,"url":42},"aeb117deb742","Trust Center","https://trust.kaiko.ai/","Company",{"links":45,"title":58},[46,50,54],{"_key":47,"_type":17,"label":48,"url":49},"40c6bf52fdd1","LinkedIn","https://www.linkedin.com/company/kaiko-ai/",{"_key":51,"_type":17,"label":52,"url":53},"f79c6a57ca2f","GitHub","https://github.com/kaiko-ai",{"_key":55,"_type":17,"label":56,"url":57},"7ff3680a1f90","Hugging Face","https://huggingface.co/kaiko-ai","Community",[60,62,64],{"_key":61,"_type":17,"label":18,"url":19},"601985fd4221",{"_key":63,"_type":17,"label":22,"url":23},"a10cc775319c",{"_key":65,"_type":66,"children":67,"label":86},"3c77dba0a31e","navigationGroup",[68,71,73,77,79,82],{"_key":69,"_type":17,"label":70,"url":34},"4d07fcce85a7","About",{"_key":72,"_type":17,"label":26,"url":27},"66b87567d450",{"_key":74,"_type":17,"label":75,"url":76},"bf5d3c3b0672","Events","https://kaiko.ai/insights-hub?categories=event",{"_key":78,"_type":17,"label":37,"url":38},"b34eb970c9d9",{"_key":80,"_type":17,"label":81,"url":42},"0179b3449f54","Security & Compliance",{"_key":83,"_type":17,"label":84,"url":85},"9f1fac3acbae","Contact Us","mailto:info@kaiko.ai","Resources","Your Clinical AI assistant that supports across full patient care, combining context for deeper insights and reducing workload, safe and compliant, made in EU.",{"_type":89,"asset":90},"image",{"_ref":91,"_type":92},"image-d9426451401bfd3574519485220b90bce4571422-1200x630-jpg","reference","https://cdn.sanity.io/images/a5kt9um3/production/d9426451401bfd3574519485220b90bce4571422-1200x630.jpg?w=1200&h=630&fit=crop",true,{"badges":96,"link":125,"linkText":126,"message":127},[97,118],{"_key":98,"color":99,"label":117},"70498638b5b3",{"_type":100,"alpha":101,"hex":102,"hsl":103,"hsv":108,"rgb":112},"color",1,"#a8e2df",{"_type":104,"a":101,"h":105,"l":106,"s":107},"hslaColor",176.89655172413794,0.7725490196078431,0.4999999999999999,{"_type":109,"a":101,"h":105,"s":110,"v":111},"hsvaColor",0.2566371681415929,0.8862745098039215,{"_type":113,"a":101,"b":114,"g":115,"r":116},"rgbaColor",223,226,168,"Sept 30",{"_key":119,"color":120,"label":124},"b238b246731c",{"_type":100,"alpha":101,"hex":102,"hsl":121,"hsv":122,"rgb":123},{"_type":104,"a":101,"h":105,"l":106,"s":107},{"_type":109,"a":101,"h":105,"s":110,"v":111},{"_type":113,"a":101,"b":114,"g":115,"r":116},"In-person","https://kaiko.ai/insights-hub/Clinical-AI-Partnership-Event","More info",[128],{"_key":129,"_type":130,"children":131,"markDefs":137,"style":138},"ee062a592cfa","block",[132],{"_key":133,"_type":134,"marks":135,"text":136},"41b4b4d5d3ff","span",[],"Join our exclusive event during Zurich AI Festival, hear from CIOs, CMIOs, and hospital leaders",[],"normal",{"documents":140,"paths":144,"mappings":163},[141,143],{"_id":91,"_type":142},"sanity.imageAsset",{"_id":6,"_type":6},[145,146,147,148,149,150,151,152,153,154,155,156,157,158,159,160,161,162],"$['_createdAt']","$['_id']","$['_rev']","$['_system']","$['_type']","$['_updatedAt']","$['defaultTitle']","$['footerMenu1']","$['footerMenu2']","$['footerMenu3']","$['gitHubUrl']","$['huggingFaceUrl']","$['linkedInUrl']","$['mainNavigation']","$['ogDescription']","$['ogImage']","$['siteNotice']","$['url']",{"$['_createdAt']":164,"$['_id']":169,"$['_rev']":171,"$['_system']":174,"$['_type']":177,"$['_updatedAt']":180,"$['defaultTitle']":183,"$['footerMenu1']":186,"$['footerMenu2']":189,"$['footerMenu3']":192,"$['gitHubUrl']":195,"$['huggingFaceUrl']":198,"$['linkedInUrl']":201,"$['mainNavigation']":204,"$['ogDescription']":207,"$['ogImage']":210,"$['ogImageUrl']":213,"$['siteNotice']":216},{"source":165,"type":168},{"document":101,"path":166,"type":167},0,"documentValue","value",{"source":170,"type":168},{"document":101,"path":101,"type":167},{"source":172,"type":168},{"document":101,"path":173,"type":167},2,{"source":175,"type":168},{"document":101,"path":176,"type":167},3,{"source":178,"type":168},{"document":101,"path":179,"type":167},4,{"source":181,"type":168},{"document":101,"path":182,"type":167},5,{"source":184,"type":168},{"document":101,"path":185,"type":167},6,{"source":187,"type":168},{"document":101,"path":188,"type":167},7,{"source":190,"type":168},{"document":101,"path":191,"type":167},8,{"source":193,"type":168},{"document":101,"path":194,"type":167},9,{"source":196,"type":168},{"document":101,"path":197,"type":167},10,{"source":199,"type":168},{"document":101,"path":200,"type":167},11,{"source":202,"type":168},{"document":101,"path":203,"type":167},12,{"source":205,"type":168},{"document":101,"path":206,"type":167},13,{"source":208,"type":168},{"document":101,"path":209,"type":167},14,{"source":211,"type":168},{"document":101,"path":212,"type":167},15,{"source":214,"type":168},{"document":166,"path":215,"type":167},17,{"source":217,"type":168},{"document":101,"path":218,"type":167},16,{"data":220,"sourceMap":2164},{"_createdAt":221,"_id":222,"_rev":223,"_system":224,"_type":227,"_updatedAt":228,"area":229,"authors":231,"categories":235,"content":239,"date":804,"featuredImage":805,"isHidden":808,"relatedPosts":809,"slug":816,"subHeading":819,"suggestedPosts":820,"title":2157,"topics":2158},"2026-07-29T13:20:41Z","bfafc11e-e363-4983-9c72-822614883107","6X7ayGJgs0U64XMIw0OHcO",{"base":225},{"id":222,"rev":226},"HCcsE6imGnTNXGDBy6mMVk","post","2026-07-29T15:17:18Z",[230],"engineering",[232],{"_key":233,"_ref":234,"_type":92},"8d8addbe718d","6a0ae6e1-567d-4330-8de3-d05158a5bf28",[236],{"_key":237,"_ref":238,"_type":92},"1e07a03563b1","9613ea96-3dba-4c14-9ac9-2b5fe14eac53",[240,287,290,342,344,363,370,421,427,429,571,577,588,590,633,635,678,684,686,745,747,766,768,795],{"_key":241,"_type":242,"text":243},"54c03faa8c20","textSection",[244,253,261,275],{"_key":245,"_type":130,"children":246,"markDefs":251,"style":252},"bb5dda5ea66a",[247],{"_key":248,"_type":134,"marks":249,"text":250},"65039dbf0fc4",[],"Summary",[],"h2",{"_key":254,"_type":130,"children":255,"markDefs":260,"style":138},"59f86d52869f",[256],{"_key":257,"_type":134,"marks":258,"text":259},"a1ed96e0391b",[],"Training large multimodal models for clinical reasoning can involve hundreds of GPUs running for weeks. At that scale, small performance regressions are costly, and distributed failures can take hours to diagnose. This post describes two parts of our training infrastructure designed to address these problems:",[],{"_key":262,"_type":130,"children":263,"level":101,"listItem":273,"markDefs":274,"style":138},"8d28d3d3106a",[264,269],{"_key":265,"_type":134,"marks":266,"text":268},"9004312eba27",[267],"strong","Performance diagnosis.",{"_key":270,"_type":134,"marks":271,"text":272},"1328bf8cd8cf",[]," A profiler captures a few steps in kernel-level detail and reduces the trace to a named bottleneck. A lightweight monitor tracks the full run, from model FLOPs utilization (MFU) to per-collective communication time, so regressions are visible as they happen.","bullet",[],{"_key":276,"_type":130,"children":277,"level":101,"listItem":273,"markDefs":286,"style":138},"e001fa666cdf",[278,282],{"_key":279,"_type":134,"marks":280,"text":281},"c14ad6e9a547",[267],"Operational reliability.",{"_key":283,"_type":134,"marks":284,"text":285},"21eeb8621045",[]," A canary tests the allocated compute, network, and storage before and periodically during training. A health monitor records progress on every rank and combines evidence from a distributed failure into one report, identifying either a likely originating rank or a shared fault domain.",[],{"_key":288,"_type":289},"5056d2a20d08","divider",{"_key":291,"_type":242,"text":292},"696a31ed110f",[293,301,309,317,334],{"_key":294,"_type":130,"children":295,"markDefs":300,"style":252},"0f84f22cd790",[296],{"_key":297,"_type":134,"marks":298,"text":299},"82eefd5d3cc0",[],"Introduction",[],{"_key":302,"_type":130,"children":303,"markDefs":308,"style":138},"9bb70c46a1f8",[304],{"_key":305,"_type":134,"marks":306,"text":307},"f24d364a2561",[],"Training large multimodal models for clinical reasoning means running jobs on hundreds of GPUs across dozens of nodes, for days, sometimes weeks. At that scale, performance depends on every part of the distributed system. Training is synchronous, so a slow GPU, network link, or input pipeline can hold back the entire job. The same clusters run a steady stream of smaller experiments, so slowdowns and failures affect both large training runs and how fast we can test new ideas.",[],{"_key":310,"_type":130,"children":311,"markDefs":316,"style":138},"91d9caf92250",[312],{"_key":313,"_type":134,"marks":314,"text":315},"9ad9ccf444a1",[],"Understanding why a run is slow requires visibility at different timescales. A profiler can explain a few steps in detail, but overhead and trace volume make continuous capture impractical. Dashboards track throughput and utilization over the full run but do not show where step time is lost. Cluster health checks can report that a node is available without revealing that a GPU is throttling, an interconnect is underperforming, or storage cannot keep up with the workload.",[],{"_key":318,"_type":130,"children":319,"markDefs":333,"style":138},"9d3f9031320b",[320,324,329],{"_key":321,"_type":134,"marks":322,"text":323},"c41176e5f8e5",[],"Failures create a different problem. An error on one rank (one GPU process) often causes communication timeouts on others, leaving many secondary errors but little indication of the original fault. Diagnosing the incident can require comparing worker and cluster logs, scheduler events, dashboards, and hardware telemetry across the allocation. By the time the investigation begins, the scheduler may have already torn the job down and removed the most useful local state. None of this is rare at scale. Meta reported an unexpected interruption roughly every three hours during the 54-day pretraining of Llama 3 405B",{"_key":325,"_type":134,"marks":326,"text":328},"4bb443e587f6",[327],"sup","1",{"_key":330,"_type":134,"marks":331,"text":332},"e9883e24149e",[],".",[],{"_key":335,"_type":130,"children":336,"markDefs":341,"style":138},"68c6e3580801",[337],{"_key":338,"_type":134,"marks":339,"text":340},"c5bee8d05815",[],"We built our own observability layer around a single principle: a run should explain itself. We record traces, timers, hardware counters, and per-rank heartbeats while the job is running and preserve them through teardown, so why a run is slow and why it failed is evident from what it recorded.",[],{"_key":343,"_type":289},"1bd2a830a91d",{"_key":345,"_type":242,"text":346},"0f6d4d9cb9b6",[347,355],{"_key":348,"_type":130,"children":349,"markDefs":354,"style":252},"09730133ced2",[350],{"_key":351,"_type":134,"marks":352,"text":353},"55a6e2ca325f",[],"Two pillars, four instruments",[],{"_key":356,"_type":130,"children":357,"markDefs":362,"style":138},"865a4441cdac",[358],{"_key":359,"_type":134,"marks":360,"text":361},"9857101e73f3",[],"To make that possible, we use four components, grouped under performance and operational reliability (Figure 1).",[],{"_key":364,"_type":365,"caption":366,"image":367},"27a16faf404c","imageSection","Figure 1: The four components, grouped into two pillars. Left (performance): a profiler for deep, occasional captures, paired with a lightweight monitor that runs the whole job. Right (operational reliability): a canary that benchmarks the assigned hardware before and during the run, paired with a health monitor that watches every rank. All four attach to the training run and write to shared storage and the experiment tracker, where one diagnosis layer reads them.",{"_type":89,"asset":368},{"_ref":369,"_type":92},"image-9158682e1d202313b1dc67c462dfdf744fb4a163-3736x1444-webp",{"_key":371,"_type":242,"text":372},"fc8a6e3076ba",[373,390,413],{"_key":374,"_type":130,"children":375,"markDefs":389,"style":138},"a4292b632f40",[376,380,385],{"_key":377,"_type":134,"marks":378,"text":379},"1ae32d7d993c",[],"The ",{"_key":381,"_type":134,"marks":382,"text":384},"07040119abed",[383],"em","profiler",{"_key":386,"_type":134,"marks":387,"text":388},"bf8830cd677c",[]," records a short window of kernel, memory, and communication activity in full detail. The performance monitor tracks throughput, framework timers, and GPU counters for the entire run, providing one summary line per logging interval.",[],{"_key":391,"_type":130,"children":392,"markDefs":412,"style":138},"bd4ba7d8e97c",[393,396,400,404,408],{"_key":394,"_type":134,"marks":395,"text":379},"156fd1eb5226",[],{"_key":397,"_type":134,"marks":398,"text":399},"68c136d529fd",[383],"canary",{"_key":401,"_type":134,"marks":402,"text":403},"aa2e05be67f9",[]," benchmarks compute, interconnect, and storage on the allocated nodes before training begins. It repeats a smaller set of checks during training. The ",{"_key":405,"_type":134,"marks":406,"text":407},"b9a9896cf8ba",[383],"health monitor",{"_key":409,"_type":134,"marks":410,"text":411},"b6831cdc19ec",[]," keeps a heartbeat and a watchdog on every rank throughout. It performs a few cheap in-memory updates per step, enough to tell at any instant whether a rank is still making progress. It collects failure evidence the moment one stops.",[],{"_key":414,"_type":130,"children":415,"markDefs":420,"style":138},"fb6630c1f609",[416],{"_key":417,"_type":134,"marks":418,"text":419},"8ffe8b98277c",[],"The four are paired because neither kind is sufficient alone: profiling and full benchmarks are too heavy to leave on, while the continuous signals cannot explain what they detect. In practice, the monitor or canary notices a change, and the profiler or health monitor explains it. All four write their outputs to shared storage and the experiment tracker in formats the same analysis code reads through a single API, so after a slowdown or a crash the evidence is already in one place. Figure 2 shows when each component runs.",[],{"_key":422,"_type":365,"caption":423,"image":424},"91fa62ccedeb","Figure 2: Time runs left to right across four lanes, one per component. The canary gates the start with a pre-flight check and re-checks periodically; the performance monitor emits one line every logging interval for the whole run; the profiler captures a short, deep window; the health monitor heartbeats continuously and, at the incident on the right, writes a report before the allocation is torn down.",{"_type":89,"asset":425},{"_ref":426,"_type":92},"image-4cd0c2c7d8d82b11e73a86dd23f47c1b6df71d21-3736x1141-webp",{"_key":428,"_type":289},"8c6c3416d51b",{"_key":430,"_type":242,"text":431},"52f9bcdb2b9f",[432,440,448,456,479,487,499,511,523,535,547,555,563],{"_key":433,"_type":130,"children":434,"markDefs":439,"style":252},"618f66ef0523",[435],{"_key":436,"_type":134,"marks":437,"text":438},"f8f45a56cf73",[],"Anatomy of a slow step",[],{"_key":441,"_type":130,"children":442,"markDefs":447,"style":138},"e9c99edb19f0",[443],{"_key":444,"_type":134,"marks":445,"text":446},"144642059b34",[],"A useful profile should end with a diagnosis, for example: non-overlapped NCCL work accounts for a third of this step, making it communication-bound. Producing that diagnosis automatically is harder than collecting the trace. A few steps of a multi-node mixture-of-experts model generate hundreds of thousands of events across dozens of GPU streams.",[],{"_key":449,"_type":130,"children":450,"markDefs":455,"style":138},"55e272fcda60",[451],{"_key":452,"_type":134,"marks":453,"text":454},"4e4672aeb696",[],"We use the PyTorch profiler by default: it records a fixed set of steps and exports a Kineto trace, per-rank kernel summaries, a categorized memory timeline, and an allocator snapshot whose stack traces survive an out-of-memory crash. For system-level or individual-kernel analysis, we use Nsight Systems or Nsight Compute, with custom NVTX ranges marking the relevant model regions.",[],{"_key":457,"_type":130,"children":458,"markDefs":475,"style":138},"20c7f504dbe2",[459,463,468,472],{"_key":460,"_type":134,"marks":461,"text":462},"4cbaeeff1450",[],"Each trace is parsed on the node where it was produced and reduced to a few kilobytes. This includes GPU busy and idle time, the compute/NCCL/memcpy split, the kernels consuming the most device time, and peak memory per device. This is the same reduction problem behind Meta's ",{"_key":464,"_type":134,"marks":465,"text":467},"f2de379ef9da",[466],"678cbc6ee804","Holistic Trace Analysis",{"_key":469,"_type":134,"marks":470,"text":471},"7d97747e0fcd",[327],"2",{"_key":473,"_type":134,"marks":474,"text":332},"c7bfbbaf5d77",[],[476],{"_key":466,"_type":477,"href":478},"link","https://pytorch.org/blog/trace-analysis-for-masses/",{"_key":480,"_type":130,"children":481,"markDefs":486,"style":138},"d5e1947bd6b3",[482],{"_key":483,"_type":134,"marks":484,"text":485},"51d172dfa2f4",[],"A priority-ordered classifier maps those measurements to a coarse bottleneck category:",[],{"_key":488,"_type":130,"children":489,"level":101,"listItem":273,"markDefs":498,"style":138},"bcb1fc74bc9b",[490,494],{"_key":491,"_type":134,"marks":492,"text":493},"7dde53ca0aa4",[267],"Launch-bound",{"_key":495,"_type":134,"marks":496,"text":497},"5da3edc9327d",[]," — high idle time and many short kernels: the CPU is not issuing GPU work quickly enough.",[],{"_key":500,"_type":130,"children":501,"level":101,"listItem":273,"markDefs":510,"style":138},"d6bd47e62137",[502,506],{"_key":503,"_type":134,"marks":504,"text":505},"33d2f8e11e2d",[267],"Input-bound",{"_key":507,"_type":134,"marks":508,"text":509},"32760946b48c",[]," — high idle time with normal kernel durations: the input pipeline is not keeping up.",[],{"_key":512,"_type":130,"children":513,"level":101,"listItem":273,"markDefs":522,"style":138},"cc4bc25941a3",[514,518],{"_key":515,"_type":134,"marks":516,"text":517},"70b91461536c",[267],"Communication-bound",{"_key":519,"_type":134,"marks":520,"text":521},"e6288d308660",[]," — NCCL consumes a large share of\nbusy time; per-collective timings determine how much sits on the\ncritical path rather than overlapping compute.",[],{"_key":524,"_type":130,"children":525,"level":101,"listItem":273,"markDefs":534,"style":138},"7a70cd0b81e6",[526,530],{"_key":527,"_type":134,"marks":528,"text":529},"82f94cda2cef",[267],"Transfer-bound",{"_key":531,"_type":134,"marks":532,"text":533},"7c433ea6f4ea",[]," — host-device copies consume a large share of the step.",[],{"_key":536,"_type":130,"children":537,"level":101,"listItem":273,"markDefs":546,"style":138},"4cff56eab01e",[538,542],{"_key":539,"_type":134,"marks":540,"text":541},"7f9b6ff56d72",[267],"Compute-bound",{"_key":543,"_type":134,"marks":544,"text":545},"233545ec4c24",[]," — none of the previous conditions applies.",[],{"_key":548,"_type":130,"children":549,"markDefs":554,"style":138},"7911ce340f48",[550],{"_key":551,"_type":134,"marks":552,"text":553},"32b53ce37496",[],"These labels describe the step as a whole. A compute-bound step can still contain kernels limited by memory bandwidth, occupancy, or instruction dependencies. Nsight Compute provides that finer distinction. Communication time also needs cross-rank context. If one rank, pipeline stage, or expert receives more work, faster ranks spend longer waiting in collectives, and an imbalance elsewhere ends up counted as NCCL time. We separate the two by comparing compute and input time across ranks.",[],{"_key":556,"_type":130,"children":557,"markDefs":562,"style":138},"eba6045565f4",[558],{"_key":559,"_type":134,"marks":560,"text":561},"c567e2e11eb1",[],"The classifier always returns its supporting measurements, and its thresholds are versioned and tested so an engineer can inspect or override a verdict without reopening the full trace.",[],{"_key":564,"_type":130,"children":565,"markDefs":570,"style":138},"26595fcbea51",[566],{"_key":567,"_type":134,"marks":568,"text":569},"3c1930560e44",[],"Figure 3 shows one of our training runs before and after acting on a communication-bound verdict. The optimized step is shorter, and its compute share rises because communication and idle time fall relative to the shorter total.",[],{"_key":572,"_type":365,"caption":573,"image":574},"e1ed637ac962","Figure 3: One training step before and after acting on a communication-bound verdict, here by overlapping communication with compute. Top: the step as two lanes, compute and communication; each block is a group of kernels, and the gaps between blocks are idle time. In the baseline, communication stalls compute; optimized, it runs underneath compute instead, so the step finishes ~32% sooner. Bottom: the same two steps as rings, inner ring by category and outer ring by the largest kernels within it. These are single operations such as one matmul or one all-reduce, showing which specific kernels dominate the category. Each is normalized to its own step so they show where time goes rather than absolute duration.",{"_type":89,"asset":575},{"_ref":576,"_type":92},"image-9ddaca3358b7b71933392cb13376f6f7d38434a3-3760x1960-webp",{"_key":578,"_type":242,"text":579},"5f12e195cf48",[580],{"_key":581,"_type":130,"children":582,"markDefs":587,"style":138},"283280d49f9f",[583],{"_key":584,"_type":134,"marks":585,"text":586},"2483316e4596",[],"We use the verdict to choose the next experiment. Communication bottlenecks lead to overlap or collective changes. Launch bottlenecks lead to fusion or graph capture. Input bottlenecks lead to the loader and storage path. When an individual operation remains expensive, we write a Triton or CUDA replacement and keep it in an in-house kernel library that the training stack pulls from instead of the stock op. Every change is re-profiled before it is kept.",[],{"_key":589,"_type":289},"647d6f88bcf0",{"_key":591,"_type":242,"text":592},"96733b40086b",[593,601,617,625],{"_key":594,"_type":130,"children":595,"markDefs":600,"style":252},"d283294a9a4e",[596],{"_key":597,"_type":134,"marks":598,"text":599},"9edecb7e2c25",[],"Monitoring performance over a full run",[],{"_key":602,"_type":130,"children":603,"markDefs":616,"style":138},"a395aa7291e0",[604,608,612],{"_key":605,"_type":134,"marks":606,"text":607},"de04678ace26",[],"A profiler samples a few steps by design: depending on the configuration, profiling adds roughly 5–30% overhead, and on a healthy run most captured data is redundant. So we pair it with a lightweight performance monitor that runs for the full job. The key metric over that horizon is model FLOPs utilization (MFU)",{"_key":609,"_type":134,"marks":610,"text":611},"fdc4a04ae716",[327],"3",{"_key":613,"_type":134,"marks":614,"text":615},"e626ecc04460",[],": an MFU of 50% means the achieved model throughput is roughly half of the GPU’s theoretical peak at the training precision. We use it to track workload over time and compare related configurations. It is more informative than GPU utilization alone since a GPU can stay fully utilized while spending time on communication or inefficient kernels.",[],{"_key":618,"_type":130,"children":619,"markDefs":624,"style":138},"e586309b668f",[620],{"_key":621,"_type":134,"marks":622,"text":623},"baddcc5096ad",[],"MFU on its own cannot explain a regression. To locate the cause, the monitor fuses hardware counters from DCGM (SM and tensor-core activity, occupancy, memory) with the framework's timers (compute, communication, pipeline bubbles, data loading) and adds timers where the framework is too coarse.",[],{"_key":626,"_type":130,"children":627,"markDefs":632,"style":138},"50d7ff3711d6",[628],{"_key":629,"_type":134,"marks":630,"text":631},"8a97e5a49069",[],"Communication is where those extra timers matter most. An NCCL call returns to the host almost immediately while the exchange runs asynchronously on the GPU's communication stream, so a host-side timer captures little more than the kernel launch. Instead, the monitor brackets each collective with CUDA events on its own stream and reads back the elapsed device time. This captures how long the exchange really took and how much overlapped with compute rather than stalling it. Tagged with its process group (data, pipeline, expert, or tensor parallel), this turns communication from one opaque bar into per-collective time attributed to the part of the parallel configuration that produced it.",[],{"_key":634,"_type":289},"f3cd2adc6584",{"_key":636,"_type":242,"text":637},"8c48a49dafcb",[638,646,662,670],{"_key":639,"_type":130,"children":640,"markDefs":645,"style":252},"dda896ec3d95",[641],{"_key":642,"_type":134,"marks":643,"text":644},"adf172999fbc",[],"The hardware is guilty until measured",[],{"_key":647,"_type":130,"children":648,"markDefs":661,"style":138},"9ef641154fd8",[649,653,657],{"_key":650,"_type":134,"marks":651,"text":652},"79b9730e1280",[],"A node can pass cluster health checks yet run below expected performance: a GPU throttling under sustained load, an NVLink or PCIe link operating below expected width, or a storage mount delivering insufficient throughput. Because each step includes synchronous collectives that all ranks must complete, the slowest affected rank can reduce throughput for the whole job",{"_key":654,"_type":134,"marks":655,"text":656},"2ab3c4dcdcf0",[327],"4",{"_key":658,"_type":134,"marks":659,"text":660},"5276eac7a7ac",[],", and nothing crashes or reports unhealthy while this happens.",[],{"_key":663,"_type":130,"children":664,"markDefs":669,"style":138},"4396e8b23fbb",[665],{"_key":666,"_type":134,"marks":667,"text":668},"c968b3e3bb01",[],"The canary tests the nodes and topology assigned to the run by executing inside the training process group itself. Before the first step, it benchmarks per-GPU compute throughput, intra-node and inter-node collective bandwidth, and storage read throughput, and records clocks, thermals, throttling state, and hardware error counters. Results are reported per rank, and the lowest-performing rank determines whether the allocation passes. A failed pre-flight check blocks training from starting, allowing the job to be rescheduled so a bad node costs minutes to reschedule around rather than days of degraded throughput.",[],{"_key":671,"_type":130,"children":672,"markDefs":677,"style":138},"72e47f9087dc",[673],{"_key":674,"_type":134,"marks":675,"text":676},"e2077ce155cb",[],"During training, it periodically repeats a smaller set of checks. Measurements taken under load are usually lower than idle pre-flight results. Two mechanisms sit on top: hard thresholds relaxed to a fraction of the pre-flight value, which catch outright failure, and an exponentially weighted moving average of each measurement. This average flags sustained deviation from the pre-flight distribution and catches gradual degradation, such as a link losing bandwidth over hours, long before the relaxed threshold fires. Figure 4 shows one such event.",[],{"_key":679,"_type":365,"caption":680,"image":681},"675a9c9c9694","Figure 4: Each dot is a periodic in-flight measurement of collective bandwidth, as a fraction of the pre-flight baseline (vertical axis) over the course of the run (horizontal axis). Because individual samples are noisy, the canary tracks their exponentially weighted moving average (the trend line). As a link slowly degrades, that trend crosses its expected band and flags the problem hours before the measurement would trip the relaxed hard threshold; the gap between the two is the early warning.",{"_type":89,"asset":682},{"_ref":683,"_type":92},"image-2068d3130e582185c5991ac53453eac6f0b12390-3736x1490-webp",{"_key":685,"_type":289},"44ea4a42f0e0",{"_key":687,"_type":242,"text":688},"0a333f75aad2",[689,697,705,713,721,729],{"_key":690,"_type":130,"children":691,"markDefs":696,"style":252},"1d9a4c0493af",[692],{"_key":693,"_type":134,"marks":694,"text":695},"ba25c4469b93",[],"When a run dies at 3am",[],{"_key":698,"_type":130,"children":699,"markDefs":704,"style":138},"4ec45a2e04f1",[700],{"_key":701,"_type":134,"marks":702,"text":703},"06bcc6cb43c9",[],"Distributed failures produce secondary errors that often obscure the original fault. In an asymmetric failure, one rank fails first and the remaining ranks time out later in a collective. In a correlated failure, several ranks fail together because they share a host, switch, storage dependency, or scheduler event. In both cases the allocation may be torn down before the relevant local state is retained.",[],{"_key":706,"_type":130,"children":707,"markDefs":712,"style":138},"f0acc5bc135d",[708],{"_key":709,"_type":134,"marks":710,"text":711},"2dfe92fea179",[],"The health monitor runs on every rank. Background threads record the current step, phase, and timestamp, sample GPU state into a ring buffer, and run a progress watchdog. The training path performs a few in-memory updates and no synchronous I/O.",[],{"_key":714,"_type":130,"children":715,"markDefs":720,"style":138},"b5fa30b7c509",[716],{"_key":717,"_type":134,"marks":718,"text":719},"518876f941ce",[],"When a rank crashes, or the watchdog sees progress stop, the monitor collects every thread's stack, recent GPU state, PyTorch's NCCL flight-recorder entries, the other ranks' latest heartbeats, and the local exception, and writes them into one structured incident report on shared storage. A rule-based classifier labels the common cases: out of memory, software exception, data-pipeline stall, storage timeout, collective mismatch or deadlock, straggler, node failure.",[],{"_key":722,"_type":130,"children":723,"markDefs":728,"style":138},"15a40e9b48c2",[724],{"_key":725,"_type":134,"marks":726,"text":727},"b4fd24940604",[],"Two real crashes show what this looks like in practice. One run died at step 62; the raw stack said only that a data-loader iterator had failed, and the report unwrapped the worker's traceback to the underlying stale file handle, so the cause was the shared filesystem briefly dropping its mounts mid-read. Another run died at step 2000 with an NCCL error in its logs; the report showed the CUDA allocations behind a collective failing because the GPU had run out of memory. The raw logs of the two runs look similar, and the reports separated them into a storage fault and an out-of-memory failure.",[],{"_key":730,"_type":130,"children":731,"markDefs":744,"style":138},"23ea6700f670",[732,736,740],{"_key":733,"_type":134,"marks":734,"text":735},"c375f8a21792",[],"Cross-rank timing is most useful for asymmetric failures. The rank whose heartbeat stops first, carrying an exception the others do not share, is usually the source, and the timeouts that follow trace back to it. For correlated failures, the report groups affected ranks by shared host, switch, or dependency, since there may be no originating rank. In one incident, every rank reported a communication timeout, but one had stopped heartbeating about thirty seconds before the others inside the expert-routing all-to-all dispatch",{"_key":737,"_type":134,"marks":738,"text":739},"328448631b88",[327],"5",{"_key":741,"_type":134,"marks":742,"text":743},"a43031ac1f91",[],". The ranks exchange data-dependent amounts of expert traffic and must first agree on the exchange sizes. Padding to a fixed expert capacity made the sizes static and removed the synchronization point where the run hung.",[],{"_key":746,"_type":289},"45c7f253a685",{"_key":748,"_type":242,"text":749},"afac641382cf",[750,758],{"_key":751,"_type":130,"children":752,"markDefs":757,"style":252},"ecc9b1b371d6",[753],{"_key":754,"_type":134,"marks":755,"text":756},"e842f7940824",[],"Runs that read their own reports",[],{"_key":759,"_type":130,"children":760,"markDefs":765,"style":138},"59d79b3c142f",[761],{"_key":762,"_type":134,"marks":763,"text":764},"b08e6ebfda99",[],"All of this was built for people first, but the same artifacts are exposed as tools over the Model Context Protocol so diagnosis can be invoked programmatically too. An agent can pull a crash report and apply the same odd-one-out triage, trigger the on-node trace analysis, and return the verdict with its evidence. It can also watch throughput and flag a sustained regression. The instruments supply the measurements, the agent summarizes them, and any action that costs cluster time still goes through a human. The practical effect is that routine questions no longer require an engineer to open a raw trace.",[],{"_key":767,"_type":289},"97c50caa5910",{"_key":769,"_type":242,"text":770},"a98a1e58d342",[771,779,787],{"_key":772,"_type":130,"children":773,"markDefs":778,"style":252},"50036cb64539",[774],{"_key":775,"_type":134,"marks":776,"text":777},"5f4ca35ba542",[],"Conclusion and outlook",[],{"_key":780,"_type":130,"children":781,"markDefs":786,"style":138},"bbad1fbf3f1b",[782],{"_key":783,"_type":134,"marks":784,"text":785},"0c8b5722b1f6",[],"The tools presented in this post address two recurring ways a run wastes the hardware it holds: running below its capacity, and failing in a way that is slow to diagnose. The profiler and performance monitor show where step time goes and whether performance changes during a run. The canary verifies the assigned hardware before training and detects degradation while the job runs. The health monitor records per-rank progress and preserves enough evidence to distinguish an originating error from the failures that follow it.",[],{"_key":788,"_type":130,"children":789,"markDefs":794,"style":138},"c2230ee3cd7e",[790],{"_key":791,"_type":134,"marks":792,"text":793},"962cef98a990",[],"Folding them into the closed loop of our AI Factory is the work ahead and the subject of a future post.",[],{"_key":796,"_type":797,"references":798},"3fdfd4cf995b","referencesSection",[799,800,801,802,803],"Dubey et al. 2024. The Llama 3 Herd of Models. arXiv:2407.21783.","PyTorch Trace Analysis for the Masses, https://pytorch.org/blog/trace-analysis-for-masses/ ","Chowdhery et al. 2022. PaLM: Scaling Language Modeling with Pathways. arXiv:2204.02311.","Jiang et al. 2024. MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs. USENIX NSDI 2024. arXiv:2402.15627.","DeepSeek-AI 2024. DeepSeek-V3 Technical Report. arXiv:2412.19437.","2026-07-29T13:00:00.000Z",{"_type":89,"asset":806},{"_ref":807,"_type":92},"image-57eb27fab7979b0e35b5be4803e54af4367c9ce6-2010x938-png",false,[810,813],{"_key":811,"_ref":812,"_type":92},"69c333d4addf","5a60d1e9-dfb8-4767-aff8-1524bb04047e",{"_key":814,"_ref":815,"_type":92},"dda64e991dfa","38a85cd4-0666-4e99-b794-b9b8371c12e9",{"_type":817,"current":818},"slug","Every-Run-Explains-Itself","How we instrument large training runs to reveal where performance is lost, trace failures to their source, and verify the hardware beneath them.",[821,1464],{"categories":822,"content":825,"date":1444,"featuredImage":1445,"slug":1448,"subHeading":1450,"title":1246,"topics":1451},[823],{"_key":824,"_ref":238,"_type":92},"4fe50cf56b15",[826,880,882,948,950,998,1004,1071,1073,1210,1216,1235,1237,1256,1262,1281,1283,1318,1324,1392,1394,1436],{"_key":827,"_type":242,"text":828},"69d51ba0eb7f",[829,836,844,856,868],{"_key":830,"_type":130,"children":831,"markDefs":835,"style":252},"dc546114bad8",[832],{"_key":833,"_type":134,"marks":834,"text":250},"7ae30ff716a1",[],[],{"_key":837,"_type":130,"children":838,"markDefs":843,"style":138},"35daf495c44a",[839],{"_key":840,"_type":134,"marks":841,"text":842},"fe2e6b919338",[],"On our journey toward training on a trillion tokens, we have learned what matters. In this post we share how we build:",[],{"_key":845,"_type":130,"children":846,"level":101,"listItem":273,"markDefs":855,"style":138},"92d0a7bcc16f",[847,851],{"_key":848,"_type":134,"marks":849,"text":850},"b439f69327b0",[267],"A stable and efficient training framework.",{"_key":852,"_type":134,"marks":853,"text":854},"0b806bdb7063",[]," Enabling seamless training resumption after preemption or hardware failure, packing strategies to keep compute on real tokens, and end-to-end testing to catch regressions before they become expensive.",[],{"_key":857,"_type":130,"children":858,"level":101,"listItem":273,"markDefs":867,"style":138},"15b3772e2b21",[859,863],{"_key":860,"_type":134,"marks":861,"text":862},"f0908f44052a",[267],"Thorough data observability.",{"_key":864,"_type":134,"marks":865,"text":866},"09d749fd6437",[]," We monitor what actually reaches the model, not just the recipe: the realized mixture and per-source statistics, collected live during the run.",[],{"_key":869,"_type":130,"children":870,"level":101,"listItem":273,"markDefs":879,"style":138},"2045655f21e5",[871,875],{"_key":872,"_type":134,"marks":873,"text":874},"c0869ece124f",[267],"Trustworthy benchmarks.",{"_key":876,"_type":134,"marks":877,"text":878},"53af3e506dea",[]," We don't take a score at face value: aggressive decontamination of the training data against every benchmark, and recording the generations behind a number.",[],{"_key":881,"_type":289},"705ca7d0a22e",{"_key":883,"_type":242,"text":884},"7c4a8e4216bf",[885,892,900,908,916,924,932,940],{"_key":886,"_type":130,"children":887,"markDefs":891,"style":252},"970c73cf1b5b",[888],{"_key":889,"_type":134,"marks":890,"text":299},"a8bc1dde2f73",[],[],{"_key":893,"_type":130,"children":894,"markDefs":899,"style":138},"ec6bcaa383e8",[895],{"_key":896,"_type":134,"marks":897,"text":898},"55a1bc359f8b",[],"Building frontier agentic reasoning models for clinical work means training a generalist model to navigate clinical workflows end-to-end: reasoning over long patient records, calling diagnostic tools, navigating radiology viewers, scrolling through whole-slide pathology images, and connecting what it learns from those tools back to a clinical question. These are tools that off-the-shelf foundation models never see during training and are not natively calibrated to operate.",[],{"_key":901,"_type":130,"children":902,"markDefs":907,"style":138},"e1bbd08160e0",[903],{"_key":904,"_type":134,"marks":905,"text":906},"148332c41fcc",[],"In a little over a year, we moved from training dense ~7B parameter multimodal vision-language models on a few hundred million tokens to 100B-token runs on 10-100x larger MoE models, with a roadmap toward trillion-token-scale continued pre-training (CPT). In our approach, the model architecture is treated as fixed. By benchmarking several open weight models that reported data sources and performances transparently, we settled on a strong base. This leaves the data as the primary variable we control in CPT and post-training. In what follows, we will therefore take a data-centric view on the stack that we’ve built.",[],{"_key":909,"_type":130,"children":910,"markDefs":915,"style":138},"0feb0297fe27",[911],{"_key":912,"_type":134,"marks":913,"text":914},"2f65f92a8251",[],"The first decision is the mixture itself: a three-way tug-of-war between installing new clinical knowledge, retaining the base's general skills, and warming up its latent ones. The rest is the engineering that delivers that mixture at scale:",[],{"_key":917,"_type":130,"children":918,"level":101,"listItem":273,"markDefs":923,"style":138},"899d88d4a02c",[919],{"_key":920,"_type":134,"marks":921,"text":922},"cf5869a8c799",[],"a robust training framework for 100B+ MoE models that tracks not just training progress but the dynamics, hardware utilization, and data statistics of a run",[],{"_key":925,"_type":130,"children":926,"level":101,"listItem":273,"markDefs":931,"style":138},"7ba6d98887b7",[927],{"_key":928,"_type":134,"marks":929,"text":930},"a0e19315943e",[],"efficient data streaming to focus the compute where it matters",[],{"_key":933,"_type":130,"children":934,"level":101,"listItem":273,"markDefs":939,"style":138},"3fa8ad03a873",[935],{"_key":936,"_type":134,"marks":937,"text":938},"2a27f23ca539",[],"observability to track down any failure and resume from a given state across data, model, and hardware (and to see what the model actually trained on)",[],{"_key":941,"_type":130,"children":942,"level":101,"listItem":273,"markDefs":947,"style":138},"4deb1bb0c509",[943],{"_key":944,"_type":134,"marks":945,"text":946},"3219294c411b",[],"an evaluation signal we can trust: broadly mapping the performance landscape we care about and rigorously decontaminating the test data",[],{"_key":949,"_type":289},"43567e705a3f",{"_key":951,"_type":242,"text":952},"c634445876a4",[953,961,976],{"_key":954,"_type":130,"children":955,"markDefs":960,"style":252},"106f6f5434aa",[956],{"_key":957,"_type":134,"marks":958,"text":959},"4faf2a1ff06e",[],"Mid-training as a three-way tug-of-war",[],{"_key":962,"_type":130,"children":963,"markDefs":975,"style":138},"85155a6b66aa",[964,968,971],{"_key":965,"_type":134,"marks":966,"text":967},"43cbeef9d998",[],"Rather than pre-training models from scratch, we start from capable models that have already been instruction-tuned and optimized for reasoning, and tool-use. We have previously verified that this yields stronger and better-performing models than starting from a checkpoint that was only pre-trained. However, a naïve continued pre-training on raw biomedical domain data would easily jeopardize the model’s learnt capabilities. To preserve these learnt skills, we craft a careful mix of raw, web-scale data (both text-only, image-captioning, and interleaved data), instruction-formatted Q&A and VQA, multi-turn conversations (with and without reasoning traces and images), as well as tool-use data. These instruction-formatted data components lean heavily on the general domain, while the raw, unformatted data is predominantly composed of domain-specific biomedical and clinical data intended to teach the model domain-specific knowledge. Technically, what we do is CPT, yet our data mixture moves this closer to what is often understood as mid-training, see e.g. in Liu et al. (2025)",{"_key":969,"_type":134,"marks":970,"text":328},"cb53f06731d0",[327],{"_key":972,"_type":134,"marks":973,"text":974},"4bbd257e7830",[],". Conventional CPT would rely exclusively on in-domain data and is often accompanied by a sharp performance drop in orthogonal domains – something that we aim to prevent by carefully crafting a balanced data mix.",[],{"_key":977,"_type":130,"children":978,"markDefs":995,"style":138},"3eab3a9d729e",[979,983,988,991],{"_key":980,"_type":134,"marks":981,"text":982},"e7924949fda1",[],"Thus, we view our CPT stage as a three-way tug-of-war between warmup, skill retention, and the target domain shift; see also ",{"_key":984,"_type":134,"marks":985,"text":987},"2d0705ed7503",[986],"5e7f12df0c22","PRISM",{"_key":989,"_type":134,"marks":990,"text":471},"0c8932053595",[327],{"_key":992,"_type":134,"marks":993,"text":994},"e9e15b276cc4",[]," for an in-depth analysis of the mid-training paradigm. Here we will briefly outline each of these directions:",[996],{"_key":986,"_type":477,"href":997},"https://arxiv.org/abs/2603.17074",{"_key":999,"_type":365,"caption":1000,"image":1001},"3df31b978f06","Mid-training as a tug-of-war between three data types, each defined by which two properties it combines. The circles are the properties (biomedical content, web-scale volume, instruction/tool-use/reasoning format); each data type sits in a pairwise overlap: domain shift (biomedical + web-scale) reinforces clinical knowledge, retention (web-scale + instruction) guards the base's general skills, warmup (biomedical + instruction) wakes latent clinical skills. Each pulls the mixture toward it with a force set by its abundance, so warmup, scarce and expensive to obtain, pulls with the thinnest rope. The triple overlap, representing web-scale biomedical instruction data, is not readily available and we reserve this for a future blog post on synthetic data generation for post-training.",{"_type":89,"asset":1002},{"_ref":1003,"_type":92},"image-c6b872846dd7f33141ab065174d0ecad477d3940-728x462-svg",{"_key":1005,"_type":242,"text":1006},"01cd7e1f72a0",[1007,1023,1039,1063],{"_key":1008,"_type":130,"children":1009,"markDefs":1022,"style":138},"3fabde72f18d",[1010,1014,1018],{"_key":1011,"_type":134,"marks":1012,"text":1013},"13fd156cf704",[],"The base model already has latent clinical core skills (instruction following, medical reasoning, basic tool-use), but calibrated for consumer-facing chatbots rather than assisting in professional clinical workflows. ",{"_key":1015,"_type":134,"marks":1016,"text":1017},"b305ca4180bd",[267],"Warmup",{"_key":1019,"_type":134,"marks":1020,"text":1021},"e6936aee7d2e",[]," turns up the model's responsiveness on these skills by exposing it to clinical-shaped versions of them, e.g. reasoning traces injected into biomedical articles, or clinical VQA.",[],{"_key":1024,"_type":130,"children":1025,"markDefs":1038,"style":138},"f0253901d30c",[1026,1030,1034],{"_key":1027,"_type":134,"marks":1028,"text":1029},"c22fb66d7184",[],"The second direction is ",{"_key":1031,"_type":134,"marks":1032,"text":1033},"dff92df3f7ba",[267],"retention",{"_key":1035,"_type":134,"marks":1036,"text":1037},"24e91c1880f0",[],": The base already knows English and a few other languages, math, code, and has some broad \"world knowledge\". We can't afford to erode any of that while chasing clinical capability. Catastrophic forgetting is the classic version of this risk. Our mixture is shaped accordingly: A large fraction of the data mix is general instruction-formatted data that uplifts the skills which otherwise would erode under the heavy weight of biomedical data.",[],{"_key":1040,"_type":130,"children":1041,"markDefs":1062,"style":138},"d40f040fe896",[1042,1046,1050,1054,1058],{"_key":1043,"_type":134,"marks":1044,"text":1045},"65bc8e908242",[],"Finally, we adapt the model to the clinical domain by exposing it to the knowledge and structures of its target domain: biomedical terminology, the shape of EHRs, the literature, the conventions and guidelines of clinical reasoning. This is the ",{"_key":1047,"_type":134,"marks":1048,"text":1049},"410e5284309c",[267],"domain shift",{"_key":1051,"_type":134,"marks":1052,"text":1053},"26c2981fcf45",[]," we are targeting and what most people think of when they hear ",{"_key":1055,"_type":134,"marks":1056,"text":1057},"747daafbab57",[383],"continued pre-training",{"_key":1059,"_type":134,"marks":1060,"text":1061},"c17d0cac1ee9",[],": raw, web-scale domain data.",[],{"_key":1064,"_type":130,"children":1065,"markDefs":1070,"style":138},"ba6a77785636",[1066],{"_key":1067,"_type":134,"marks":1068,"text":1069},"f85eef49fe42",[],"Each data type pulls the model in a different direction and striking the right balance to meet the criteria of a production-grade clinical model requires detailed understanding of the intended uses of the model, but equally important is a thorough understanding of the data mix in training. We will come back to this momentarily.",[],{"_key":1072,"_type":289},"8360cdaf10a7",{"_key":1074,"_type":242,"text":1075},"3b0612e19a60",[1076,1084,1092,1100,1108,1120,1132,1140,1148,1156,1172,1184],{"_key":1077,"_type":130,"children":1078,"markDefs":1083,"style":252},"2de0048034bc",[1079],{"_key":1080,"_type":134,"marks":1081,"text":1082},"872be5594ca6",[],"Building the foundation",[],{"_key":1085,"_type":130,"children":1086,"markDefs":1091,"style":138},"e010aac9b769",[1087],{"_key":1088,"_type":134,"marks":1089,"text":1090},"cd7d3e113836",[],"Our aim for the training framework is that scaling up model size and data is a change of configuration, not a reinvention of the wheel. Everything above rests on it, so the infrastructure underneath has to be solid. Here’s what we concluded on.",[],{"_key":1093,"_type":130,"children":1094,"markDefs":1099,"style":252},"0d5e1577609d",[1095],{"_key":1096,"_type":134,"marks":1097,"text":1098},"58f0fbcd59a2",[],"Training framework and data loading",[],{"_key":1101,"_type":130,"children":1102,"markDefs":1107,"style":138},"2429fb33025d",[1103],{"_key":1104,"_type":134,"marks":1105,"text":1106},"8079b16ff675",[],"After evaluating the alternatives, we settled on building on top of NVIDIA's NeMo framework: Megatron-Bridge for training and Energon for data loading, respectively. What we needed was production-readiness at scale with enough integration surface to layer our own callbacks, logging, and recipe defaults on top, and both Megatron-Bridge and Energon fit: they are fully integrated from data loading, model training, and performance tracking, all the way to checkpoint resumption, and at the same time allow us to build on top, add custom training logic and logging where we need it. As a bonus, NVIDIA's NeMo comes with a range of low-level hardware accelerations that result in 5x faster training iterations compared to our old research training framework built on top of PyTorch Lightning.",[],{"_key":1109,"_type":130,"children":1110,"markDefs":1119,"style":138},"338523a84915",[1111,1115],{"_key":1112,"_type":134,"marks":1113,"text":1114},"f0a387393db2",[267],"Tests at every scale of integration.",{"_key":1116,"_type":134,"marks":1117,"text":1118},"3e7c8c20e52c",[]," While unit tests ensure that every component in the pipeline behaves as expected, unforeseen long-range interactions both within one team and across team boundaries can lead to undesirable outcomes. Therefore, we test end-to-end: For any change to the data or the code we trigger integration tests for the entire pipeline including short training runs and full iterations of the affected datasets to ensure continuity in our production training runs.",[],{"_key":1121,"_type":130,"children":1122,"markDefs":1131,"style":138},"9b4f97296995",[1123,1127],{"_key":1124,"_type":134,"marks":1125,"text":1126},"9e0baf1e6048",[267],"Byte-identical resumability.",{"_key":1128,"_type":134,"marks":1129,"text":1130},"f62d3f6b2bfe",[]," A multi-week run will be interrupted at some point, either by preemption, hardware failure, or a planned restart. This is where the NeMo universe shines: We persist the full dataloader state alongside every model checkpoint, thereby ensuring byte-identical behavior between the uninterrupted run and its continuation.",[],{"_key":1133,"_type":130,"children":1134,"markDefs":1139,"style":252},"2bf9ba0cb06b",[1135],{"_key":1136,"_type":134,"marks":1137,"text":1138},"d7a762e1c697",[],"Packing and token efficiency",[],{"_key":1141,"_type":130,"children":1142,"markDefs":1147,"style":138},"abbfbc797779",[1143],{"_key":1144,"_type":134,"marks":1145,"text":1146},"63d23b52b4e4",[],"Why production-readiness matters at this scale is best illustrated by putting numbers to it. One example that's easy to overlook is sequence packing.",[],{"_key":1149,"_type":130,"children":1150,"markDefs":1155,"style":138},"29813ce26ead",[1151],{"_key":1152,"_type":134,"marks":1153,"text":1154},"e6d75aae884f",[],"Biomedical and clinical corpora are wildly heterogeneous in document length: a PubMed abstract runs around 200 tokens, clinical notes vary broadly (500-3K), full papers and tool traces easily exceed 10K. Packing sequentially loaded samples from such a heterogeneous distribution into a standard pre-training sequence of length, say, 4k tokens, inevitably results in a substantial amount of padding tokens: long sequences are truncated, long and short sequences get combined into sub-optimal combinations. To alleviate this, we use a greedy Knapsack algorithm that selects packing candidates optimally from a pre-loaded buffer. Moreover, we choose a packing strategy that minimizes the padding content. Three packing regimes are worth comparing:",[],{"_key":1157,"_type":130,"children":1158,"level":101,"listItem":273,"markDefs":1171,"style":138},"374fc10da694",[1159,1163,1167],{"_key":1160,"_type":134,"marks":1161,"text":1162},"5f8528990b14",[267],"No packing.",{"_key":1164,"_type":134,"marks":1165,"text":1166},"6c7307634e50",[]," One document per sequence, padded out to the context length. On a biomedical document-length mix at 4K context, realistic padding waste is ",{"_key":1168,"_type":134,"marks":1169,"text":1170},"00595e8629fc",[267],"30-40%.",[],{"_key":1173,"_type":130,"children":1174,"level":101,"listItem":273,"markDefs":1183,"style":138},"575ade0b68a1",[1175,1179],{"_key":1176,"_type":134,"marks":1177,"text":1178},"3084a27232c9",[267],"Naïve packing.",{"_key":1180,"_type":134,"marks":1181,"text":1182},"aacf4e83e498",[]," Concatenate samples into one sequence, no attention masking. Padding waste drops to near zero, but attention across document boundaries can contaminate the gradient signal and has sub-optimal memory scaling.",[],{"_key":1185,"_type":130,"children":1186,"level":101,"listItem":273,"markDefs":1207,"style":138},"5a74c027d94a",[1187,1191,1195,1200,1203],{"_key":1188,"_type":134,"marks":1189,"text":1190},"7cc761332000",[267],"Boundary-aware packing.",{"_key":1192,"_type":134,"marks":1193,"text":1194},"74f97f6436e5",[]," Concatenate samples into a single sequence, but mask attention so each token only attends to others within its own document. This is the approach chosen by us and other frontier labs, see e.g. ",{"_key":1196,"_type":134,"marks":1197,"text":1199},"cd23054625a3",[1198],"1d1c311fb0db","DeepSeek-V4",{"_key":1201,"_type":134,"marks":1202,"text":611},"21210b1e6afa",[327],{"_key":1204,"_type":134,"marks":1205,"text":1206},"89641ab631f8",[],". It guarantees near-zero padding waste and each document is treated as an individual sequence.",[1208],{"_key":1198,"_type":477,"href":1209},"https://arxiv.org/abs/2606.19348",{"_key":1211,"_type":365,"caption":1212,"image":1213},"445ea7d6238d","Top: six documents of different lengths. One document per sequence (left) leaves 30 - 40% of tokens as padding (note that for multi-node training, a global sequence length must be fixed); concatenating the documents (right) removes most of it, with a small residual if sequences are split only at paragraph or sentence boundaries. Bottom: the causal attention matrix over one packed sequence of three documents (A, B, C). Within-document attention (blue) is identical in both cases; naïve packing (left) additionally attends across document boundaries (gradient contamination and suboptimal memory scaling), while boundary-aware packing (right) masks those cross-document positions, so each document is treated as an individual sequence.",{"_type":89,"asset":1214},{"_ref":1215,"_type":92},"image-fd14a1c0d14abf45a5e9c08332953644f2cf7535-960x780-svg",{"_key":1217,"_type":242,"text":1218},"40929a7023b1",[1219],{"_key":1220,"_type":130,"children":1221,"markDefs":1234,"style":138},"d19cc193a915",[1222,1226,1230],{"_key":1223,"_type":134,"marks":1224,"text":1225},"20935da01ddb",[],"In GPU-hours this would mean the following. The public Qwen3.5-VL 122B-A10B SFT recipe runs at global batch size 36, with a maximum sequence length of 4,096, for 300k steps: roughly 44B training tokens, or about 2'100 GPU-hours on 48 H100s at 35% MFU. At a worst-case 40% padding waste, ~840 of those GPU-hours are spent processing padded positions that contribute no useful training signal. Scaling the same math up, a 100B-token run wastes around 1'900 GPU-hours per recipe ablation, and the 1T-token target wastes around 19'000. That's roughly ",{"_key":1227,"_type":134,"marks":1228,"text":1229},"92e681c9e784",[267],"$76K per run",{"_key":1231,"_type":134,"marks":1232,"text":1233},"ed6e374b63f6",[]," at typical H100 cloud pricing of around 4$/GPU-hour.",[],{"_key":1236,"_type":289},"9a8765b71e4c",{"_key":1238,"_type":242,"text":1239},"8d3ef62cd2fb",[1240,1248],{"_key":1241,"_type":130,"children":1242,"markDefs":1247,"style":252},"2f1d3ec018ee",[1243],{"_key":1244,"_type":134,"marks":1245,"text":1246},"449673190e49",[],"What did the model actually see?",[],{"_key":1249,"_type":130,"children":1250,"markDefs":1255,"style":138},"c935a02a3b9e",[1251],{"_key":1252,"_type":134,"marks":1253,"text":1254},"2aa229c788ff",[],"One subtlety matters here specifically since the realized and planned data mixtures can deviate substantially: The planned mix is a list of sources with their target shares, a set of filter parameters, and a training schedule. However, what reaches the model might deviate substantially from this planned mix. We illustrate this with the figure below (left panel). A declared (sample level) recipe would contain 71% web-scale biomedical text data; however, after transformations and filters are applied, sequences have been packed, and bad or uninformative samples removed (think of corrupted images that previously weren’t caught), the true token level mixture looks quite different, with a sharply decreased share of the web scale text dataset. Further looking at the loss token share per dataset, the tension relaxes but still strongly differs from the declared mix.",[],{"_key":1257,"_type":365,"caption":1258,"image":1259},"e7e060cc1dba","A real run (~20B tokens; sources anonymized), counted three ways. Left: the same realized data weighted by sample, by token, and by loss-token. By sample it matches the declared blend (71% biomedical text / 10% biomedical image-text / 19% skill additions); by token, biomedical text falls to 43% as token-dense skill data more than doubles its share; by loss-token (prompt, system, and image tokens masked out) biomedical text climbs back to 51%, while masked tool-use data collapses from 12% to 5%, far less than its token share suggests. Right: sources differ in size, so a small tool-use subset is cycled ~9 times before the largest corpus finishes a fifth of one pass.",{"_type":89,"asset":1260},{"_ref":1261,"_type":92},"image-22de86cabced38eb9772f42314b21a3c93b7acdb-1120x460-svg",{"_key":1263,"_type":242,"text":1264},"be90546e868a",[1265,1273],{"_key":1266,"_type":130,"children":1267,"markDefs":1272,"style":138},"89efc4513b11",[1268],{"_key":1269,"_type":134,"marks":1270,"text":1271},"9cf280f773d2",[],"Another aspect that is crucial to monitor is the number of times each dataset in the mix is seen. A first best judgement blend will be altered through the aforementioned dynamics and consequently we might be seeing the same samples more than intended. The right panel of the figure illustrates this, where a small tool-use dataset is heavily oversampled. The mechanism is as follows: While the dataset might have been correctly weighted by token share (tool-use data tends to come with fewer samples but very long sequences), many of the samples would exceed the context window of the initial stage of CPT already with the system message and user prompt, which depending on the training configuration may leave no tokens relevant for gradient calculation at all and thus are filtered on the fly. The resulting over-representation of the dataset in the mix requires a subsequent re-adjustment.",[],{"_key":1274,"_type":130,"children":1275,"markDefs":1280,"style":138},"4b03cf391aaf",[1276],{"_key":1277,"_type":134,"marks":1278,"text":1279},"335716809993",[],"The cost of being wrong grows with run size. At the trillion-token scale we're working toward, a 1% drift in the realized mixture is 10B tokens of unintended training spent on the wrong distribution (about the entire budget of a respectable open research run). Our token-level observability stack ensures that any drift is visible at training time and not hidden in a silent performance drift in evaluation.",[],{"_key":1282,"_type":289},"c70574fa1114",{"_key":1284,"_type":242,"text":1285},"78af5b6faf74",[1286,1294,1302,1310],{"_key":1287,"_type":130,"children":1288,"markDefs":1293,"style":252},"7aae45ba6e15",[1289],{"_key":1290,"_type":134,"marks":1291,"text":1292},"85ca49ba3bf3",[],"A trustworthy evaluation signal",[],{"_key":1295,"_type":130,"children":1296,"markDefs":1301,"style":138},"4be2a7958365",[1297],{"_key":1298,"_type":134,"marks":1299,"text":1300},"dd6e8aa47276",[],"Checkpoint validation is our main signal for whether the upstream work paid off, so it carries the same engineering investment as training and data.",[],{"_key":1303,"_type":130,"children":1304,"markDefs":1309,"style":138},"22e53ea0f1e8",[1305],{"_key":1306,"_type":134,"marks":1307,"text":1308},"f0c25d2f4d9c",[],"Contamination is a sharper risk for us than for a research model. A research model is judged on a fixed, public benchmark set it can decontaminate against; our models are judged by users on clinical tasks we may not have anticipated. Therefore, we evaluate against both critical public benchmarks and internal private ones, and keep adding and revising them as the product surface grows. One uncaught test-set leak turns that signal into a misleading one.",[],{"_key":1311,"_type":130,"children":1312,"markDefs":1317,"style":138},"d4e39d0546a2",[1313],{"_key":1314,"_type":134,"marks":1315,"text":1316},"77274cdcdeb2",[],"Therefore, we deduplicate every training source against every active benchmark. Exact matching runs on every pair; we add the costly fuzzy and semantic passes where source and benchmark draw on shared material. Overlap on the rest is tracked as a tripwire, escalating a pair if it climbs above a threshold.",[],{"_key":1319,"_type":365,"caption":1320,"image":1321},"a508836edf99","Leakage risk varies across (training source × benchmark) pairs, so we target the expensive matching rather than spread it evenly. Every pair is deduplicated by exact match; where a source and benchmark draw on shared material (high-risk pairs) we add the full fuzzy and semantic stack (dense-embedding matching for images). Overlap on the rest is tracked as a tripwire and escalated if it climbs. New sources and benchmarks are matched against everything active as they come online.",{"_type":89,"asset":1322},{"_ref":1323,"_type":92},"image-7ef14539e1371164e7cf6dbda271faf29106a01e-780x470-svg",{"_key":1325,"_type":242,"text":1326},"e04dea550fa1",[1327,1335,1347,1365,1384],{"_key":1328,"_type":130,"children":1329,"markDefs":1334,"style":138},"a74f34b0467f",[1330],{"_key":1331,"_type":134,"marks":1332,"text":1333},"a6f511a6172f",[],"A trustworthy signal needs two things: a benchmark the model has not already seen, and a way to read why a score moved. For the first, we escalate the matcher, each step catching what the one before it cannot:",[],{"_key":1336,"_type":130,"children":1337,"level":101,"listItem":273,"markDefs":1346,"style":138},"8355dc6045b8",[1338,1342],{"_key":1339,"_type":134,"marks":1340,"text":1341},"883ddbe07a8c",[267],"Exact match.",{"_key":1343,"_type":134,"marks":1344,"text":1345},"b97e43c51e87",[]," Verbatim copies fall out of a hash match, run on every (source × benchmark) pair.",[],{"_key":1348,"_type":130,"children":1349,"level":101,"listItem":273,"markDefs":1364,"style":138},"0dba7b474d3d",[1350,1354,1358,1361],{"_key":1351,"_type":134,"marks":1352,"text":1353},"5296869c12eb",[267],"Fuzzy matching.",{"_key":1355,"_type":134,"marks":1356,"text":1357},"4789fc6e165f",[]," Reformatted and reworded copies survive it: MinHash-LSH surfaces near-duplicates by Jaccard overlap",{"_key":1359,"_type":134,"marks":1360,"text":656},"b025fbb6c976",[327],{"_key":1362,"_type":134,"marks":1363,"text":332},"5228c870ab07",[],[],{"_key":1366,"_type":130,"children":1367,"level":101,"listItem":273,"markDefs":1383,"style":138},"e717ab4975c6",[1368,1372,1376,1379],{"_key":1369,"_type":134,"marks":1370,"text":1371},"1c449255b409",[267],"Semantic matching.",{"_key":1373,"_type":134,"marks":1374,"text":1375},"407039cff2af",[]," What survives fuzzy is caught by embeddings: a cosine pass over text",{"_key":1377,"_type":134,"marks":1378,"text":739},"9f4a886e94d7",[327],{"_key":1380,"_type":134,"marks":1381,"text":1382},"f0f305801ae8",[],", and dense image embeddings with approximate nearest-neighbour search that flag a training image answering a benchmark question even when its pixels differ. Both carry a higher false-positive cost, so we reserve them for pairs where a source and benchmark share material, as illustrated in the figure.",[],{"_key":1385,"_type":130,"children":1386,"markDefs":1391,"style":138},"a130dec621f4",[1387],{"_key":1388,"_type":134,"marks":1389,"text":1390},"0c39c1608f5c",[],"For the second point, we store all generations and not just the score. We pull actual generations from each checkpoint and inspect a few of them by hand and with an LLM judge at scale, because a drop in a benchmark has many possible causes a single number hides. A common example is a drop in performance, where the model gives the right answer in the wrong format, so the failure is instruction following, not knowledge. A score alone cannot tell the two apart.",[],{"_key":1393,"_type":289},"0fa9e390015a",{"_key":1395,"_type":242,"text":1396},"204aae08f9bd",[1397,1404,1412,1428],{"_key":1398,"_type":130,"children":1399,"markDefs":1403,"style":252},"dd7d700a4605",[1400],{"_key":1401,"_type":134,"marks":1402,"text":777},"3111078cd748",[],[],{"_key":1405,"_type":130,"children":1406,"markDefs":1411,"style":138},"d6a2d91f522f",[1407],{"_key":1408,"_type":134,"marks":1409,"text":1410},"4dd4e4410448",[],"Every stage in this post acts on the same data mixture: the framework and packing deliver it, observability records what the model actually trained on, and decontaminated evaluation shows whether it worked. Each component is built to hold as runs scale, so moving from today's 100B-token runs to the trillion-token target on 100B+ MoE models means changing the configuration, not rewriting the stack. What remains is to compose them into a single loop.",[],{"_key":1413,"_type":130,"children":1414,"markDefs":1427,"style":138},"8ac1185c23f4",[1415,1419,1423],{"_key":1416,"_type":134,"marks":1417,"text":1418},"1de81c4b8bf6",[],"That composition is what we are building toward in kaiko's ",{"_key":1420,"_type":134,"marks":1421,"text":1422},"60d25eb0e4a7",[267],"AI Factory",{"_key":1424,"_type":134,"marks":1425,"text":1426},"6552009fc270",[],", a pipeline where evaluation results feed back into the recipe directly, the recipe as a versioned, structured artifact and the observability primitives from the preceding sections making each stage's output legible to the next. The orchestration already runs on Dagster, workloads are distributed via Ray, and Kubernetes provisions and schedules the resources. The components exist today (framework, loader, packing, decontamination, a trustworthy evaluation signal, an emerging recipe artifact). Composing them into a single closed loop is the work ahead, and the subject of a future deep dive.",[],{"_key":1429,"_type":130,"children":1430,"markDefs":1435,"style":138},"f6c7f0283de9",[1431],{"_key":1432,"_type":134,"marks":1433,"text":1434},"1a31159132d3",[],"None of these practices are novel on their own. The literature has assembled most of the pieces. What rarely makes it into a paper are the steps behind a headline: preempted and crashed runs, the mixture drifting from the recipe, or a dataset silently cycled nine times. With this article, we want to shed some light on what it takes to move toward the trillion-token frontier.",[],{"_key":1437,"_type":797,"references":1438},"81a6af91cf49",[1439,1440,1441,1442,1443],"Liu et al. 2025, Midtraining Bridges Pretraining and Posttraining Distributions, \nhttps://arxiv.org/abs/2510.14865","PRISM: Demystifying Retention and Interaction in Mid-Training, https://arxiv.org/abs/2603.17074","DeepSeek-V4: Towards Highly Efficient Million-Token Context Intelligence\nhttps://arxiv.org/abs/2606.19348","Lee et al. 2021 Deduplicating Training Data Makes Language Models Better, https://arxiv.org/abs/2107.06499 ","Abbas et al. 2023, SemDeDup: Data-efficient learning at web-scale through semantic deduplication, https://arxiv.org/abs/2303.09540","2026-07-09T16:12:54.910Z",{"_type":89,"asset":1446},{"_ref":1447,"_type":92},"image-d9a09026393fbffde93f88b44b36fe2f68e105d2-2010x938-png",{"_type":817,"current":1449},"Training-clinical-reasoning-model-at-scale","Toward training at the trillion-token scale: Why reliability, stability, and observability matter and how we engineer them.",[1452,1455,1458,1461],{"_key":1453,"_ref":1454,"_type":92},"3b76795f501c","ffc879d7-9be3-47d1-bda5-5817c9d7cf8c",{"_key":1456,"_ref":1457,"_type":92},"c2dc7fdd784c","0e7a76c9-607b-4b29-b0f5-48940b955680",{"_key":1459,"_ref":1460,"_type":92},"a476af4cdeb6","038e94a3-7be4-458d-972c-d3b572c832a0",{"_key":1462,"_ref":1463,"_type":92},"279baccd7561","5f860d5a-1a0d-4041-ab25-eb07132ec4bd",{"categories":1465,"content":1468,"date":2140,"featuredImage":2141,"slug":2144,"subHeading":2146,"title":2147,"topics":2148},[1466],{"_key":1467,"_ref":238,"_type":92},"a78b4a06e8fc",[1469,1475,1558,1560,1582,1588,1627,1629,1657,1663,1674,1680,1727,1746,1752,1754,1781,1792,1798,1825,1880,1882,1953,1955,2112,2114],{"_key":1470,"_type":365,"caption":1471,"image":1472},"9566649717a4","Figure 1-4: Ground-truth annotations vs CoralBay predictions on various datasets.",{"_type":89,"asset":1473},{"_ref":1474,"_type":92},"image-f2f8cd6857b5225f5e7a438ea0fbd3b8520ccbbb-1280x720-gif",{"_key":1476,"_type":242,"text":1477},"64220c0acaa7",[1478,1486,1502,1510,1522,1534,1546],{"_key":1479,"_type":130,"children":1480,"markDefs":1485,"style":138},"35765f76c825",[1481],{"_key":1482,"_type":134,"marks":1483,"text":1484},"8c81e04012ee",[],"Each year, over 300 million CT scans are captured as rich, high-fidelity 3D volumes that give doctors a detailed view of the human body for diagnosis and surgical planning. However, the status quo in current AI models is to process these volumes as stacks of 2D slices, effectively flattening the data and discarding critical spatial relationships, or to rely on complicated, highly specialised methods that are difficult to scale and build upon.",[],{"_key":1487,"_type":130,"children":1488,"markDefs":1501,"style":138},"b331bc6804ad",[1489,1493,1497],{"_key":1490,"_type":134,"marks":1491,"text":1492},"7bca31b804f0",[],"Enter ",{"_key":1494,"_type":134,"marks":1495,"text":1496},"cd673be05ab4",[267],"CoralBay",{"_key":1498,"_type":134,"marks":1499,"text":1500},"aacb86bcdcce",[],": our native 3D foundation model for CT, pre-trained on only 11K unlabeled volumes. With a simple architecture and training pipeline, it produces meaningful representations from raw data that generalize across the full clinical stack, from scan-level classification down to fine-grained lesion segmentation.",[],{"_key":1503,"_type":130,"children":1504,"markDefs":1509,"style":252},"54b86dc85d82",[1505],{"_key":1506,"_type":134,"marks":1507,"text":1508},"6a2cc7267166",[267],"Quick Highlights",[],{"_key":1511,"_type":130,"children":1512,"level":101,"listItem":273,"markDefs":1521,"style":138},"4d251260c971",[1513,1517],{"_key":1514,"_type":134,"marks":1515,"text":1516},"05b0b7dc5891",[267],"Native 3D Intelligence:",{"_key":1518,"_type":134,"marks":1519,"text":1520},"e07ecffd95d4",[]," Understands anatomy as continuous volumetric structures.",[],{"_key":1523,"_type":130,"children":1524,"level":101,"listItem":273,"markDefs":1533,"style":138},"8caec661b008",[1525,1529],{"_key":1526,"_type":134,"marks":1527,"text":1528},"abe2ae64d09d",[267],"Simple, Strong, Generalizable:",{"_key":1530,"_type":134,"marks":1531,"text":1532},"1b030a7961bb",[]," Strong multi-task performance from a simple training pipeline and just 11K unlabeled volumes, ready to adapt to various downstream tasks.",[],{"_key":1535,"_type":130,"children":1536,"level":101,"listItem":273,"markDefs":1545,"style":138},"fee963bbe5c7",[1537,1541],{"_key":1538,"_type":134,"marks":1539,"text":1540},"18b325e8ae18",[267],"Clinical Depth:",{"_key":1542,"_type":134,"marks":1543,"text":1544},"55efa3c6f50e",[]," A single backbone with built-in invariance to HU values, excelling across the full task spectrum from broad organ identification to fine-grained tumour segmentation",[],{"_key":1547,"_type":130,"children":1548,"level":101,"listItem":273,"markDefs":1557,"style":138},"43c283f1419c",[1549,1553],{"_key":1550,"_type":134,"marks":1551,"text":1552},"21d4328d42dc",[267],"Open Science:",{"_key":1554,"_type":134,"marks":1555,"text":1556},"bfdcf0b49a91",[]," Fully open-source weights and benchmarks to accelerate global medical AI research.",[],{"_key":1559,"_type":289},"4be33ed573b9",{"_key":1561,"_type":242,"text":1562},"f8ff5b1d9a48",[1563,1571],{"_key":1564,"_type":130,"children":1565,"markDefs":1570,"style":252},"bd0845d89d25",[1566],{"_key":1567,"_type":134,"marks":1568,"text":1569},"5bd09ab7d8c7",[267],"The Challenge: 3D is Different",[],{"_key":1572,"_type":130,"children":1573,"markDefs":1581,"style":138},"b8528c9d14ad",[1574,1578],{"_key":1575,"_type":134,"marks":1576,"text":1577},"1576c6dfdc1d",[],"Standard vision models assume 2D RGB images; three colour channels, consistent resolution, and millions of available labeled examples. CT scans violate every one of these assumptions.",{"_key":1567,"_type":134,"marks":1579,"text":1580},[267],"\n",[],{"_key":1583,"_type":365,"caption":1584,"image":1585},"b34e97709062","Figure 5: Challenges in 3D data representation. Left: Narrow windows clarify soft tissue; wide windows preserve high-density data. Center: Thin slices increase resolution; thick slices improve SNR but cause blurring. Right: 3D consistency is required across all orthogonal planes.",{"_type":89,"asset":1586},{"_ref":1587,"_type":92},"image-b15c3a2413e2743044aa8a6f40419291816065e1-2857x860-png",{"_key":1589,"_type":242,"text":1590},"7170efbff7f2",[1591,1603,1615],{"_key":1592,"_type":130,"children":1593,"level":101,"listItem":273,"markDefs":1602,"style":138},"13c3b67b8f91",[1594,1598],{"_key":1595,"_type":134,"marks":1596,"text":1597},"951db6b47c8a",[267],"Hounsfield Units, not pixels",{"_key":1599,"_type":134,"marks":1600,"text":1601},"68968e710212",[],": Intensities reflect tissue density, not color. Different window settings highlight different anatomy (e.g., lungs vs. soft tissue), so models must be robust to multiple visualizations of the same scan.",[],{"_key":1604,"_type":130,"children":1605,"level":101,"listItem":273,"markDefs":1614,"style":138},"d5c75cd01feb",[1606,1610],{"_key":1607,"_type":134,"marks":1608,"text":1609},"fbec56375347",[267],"Anisotropic resolution",{"_key":1611,"_type":134,"marks":1612,"text":1613},"48c80c0a9f45",[],": Slice thickness varies (e.g., 1–5 mm), causing partial volume effects. Thicker slices improve signal-to-noise but blur fine structures, unlike typical image noise.",[],{"_key":1616,"_type":130,"children":1617,"level":101,"listItem":273,"markDefs":1626,"style":138},"a5fd44829f4a",[1618,1622],{"_key":1619,"_type":134,"marks":1620,"text":1621},"88bb51682b5d",[267],"Volumetric spatial complexity",{"_key":1623,"_type":134,"marks":1624,"text":1625},"e4263fe818ca",[],": CT is inherently 3D. Small findings, including early-stage tumours that may occupy only a handful of voxels, require integrating information across the full 3D space to be reliably detected. Treating slices independently breaks the spatial continuity that makes these critical, hard-to-spot findings visible in the first place.",[],{"_key":1628,"_type":289},"17ece86911bc",{"_key":1630,"_type":242,"text":1631},"910b8f205540",[1632,1640,1649],{"_key":1633,"_type":130,"children":1634,"markDefs":1639,"style":252},"e1031f7eaa2b",[1635],{"_key":1636,"_type":134,"marks":1637,"text":1638},"1958c0300072",[267],"The CoralBay Approach",[],{"_key":1641,"_type":130,"children":1642,"markDefs":1647,"style":1648},"24cf30dce250",[1643],{"_key":1644,"_type":134,"marks":1645,"text":1646},"cfef1df31105",[267],"Hierarchical 3D Feature Learning",[],"h3",{"_key":1650,"_type":130,"children":1651,"markDefs":1656,"style":138},"22efd7486e59",[1652],{"_key":1653,"_type":134,"marks":1654,"text":1655},"26123db28de3",[],"CoralBay extends DINO self-supervised learning to native 3D CT volumes. Using a hierarchical 3D Swin-Transformer, it captures long-range spatial relationships efficiently and learns multi-scale features ranging from organ-level anatomy to fine structures such as vessels.",[],{"_key":1658,"_type":365,"caption":1659,"image":1660},"fc44b831cb37","Figure 6: CoralBay’s technical overviews of the training pipeline",{"_type":89,"asset":1661},{"_ref":1662,"_type":92},"image-dbb59b4f247bc0d580e639560876d93759308654-2983x830-png",{"_key":1664,"_type":242,"text":1665},"e24b6ad5a981",[1666],{"_key":1667,"_type":130,"children":1668,"markDefs":1673,"style":1648},"b5b27746eff8",[1669],{"_key":1670,"_type":134,"marks":1671,"text":1672},"17448af3114b",[267],"Radiology-Specific Augmentations",[],{"_key":1675,"_type":365,"caption":1676,"image":1677},"d5b91e32677e","Figure 7: HU ranges for pre-training data augmentation.",{"_type":89,"asset":1678},{"_ref":1679,"_type":92},"image-81151700e6702f95aea34f954a36249de02bdf7b-1871x662-png",{"_key":1681,"_type":242,"text":1682},"cef125988b64",[1683,1691,1703,1715],{"_key":1684,"_type":130,"children":1685,"markDefs":1690,"style":138},"0183fefb6cf8",[1686],{"_key":1687,"_type":134,"marks":1688,"text":1689},"0aebd20a500b",[],"CoralBay replaces generic image augmentations with CT-specific transformations that reflect real-world imaging variability.",[],{"_key":1692,"_type":130,"children":1693,"level":101,"listItem":273,"markDefs":1702,"style":138},"6cb032b32dc8",[1694,1698],{"_key":1695,"_type":134,"marks":1696,"text":1697},"ed6a8d78c81c",[267],"Random HU Windowing:",{"_key":1699,"_type":134,"marks":1700,"text":1701},"4950f307840f",[]," Samples from multiple clinically relevant HU ranges (e.g., lung, liver, brain, abdomen, and full CT) to learn features that are robust across viewing settings.",[],{"_key":1704,"_type":130,"children":1705,"level":101,"listItem":273,"markDefs":1714,"style":138},"ce21392c405f",[1706,1710],{"_key":1707,"_type":134,"marks":1708,"text":1709},"4773b4688429",[267],"Scanner robustness:",{"_key":1711,"_type":134,"marks":1712,"text":1713},"05f3208926e3",[]," Gaussian smoothing and histogram shifts simulate differences in scanners and reconstruction protocols, improving generalization across sites.",[],{"_key":1716,"_type":130,"children":1717,"level":101,"listItem":273,"markDefs":1726,"style":138},"2acdfbb129d2",[1718,1722],{"_key":1719,"_type":134,"marks":1720,"text":1721},"cfec638253be",[267],"Local-to-global 3D context:",{"_key":1723,"_type":134,"marks":1724,"text":1725},"e6d600c2cfe9",[]," Combines large and small 3D crops during training, helping the model link fine pathological details with their surrounding anatomy.\n\n",[],{"_key":1728,"_type":242,"text":1729},"15ad5094a40b",[1730,1738],{"_key":1731,"_type":130,"children":1732,"markDefs":1737,"style":1648},"be04f790ac1e",[1733],{"_key":1734,"_type":134,"marks":1735,"text":1736},"8da62902f193",[267],"Scan-Level Inference via Sliding Window",[],{"_key":1739,"_type":130,"children":1740,"markDefs":1745,"style":138},"b01bb1b885eb",[1741],{"_key":1742,"_type":134,"marks":1743,"text":1744},"2e22e58a177a",[],"While the backbone is trained on 96×96×96 crops, inference uses a sliding-window approach: the full scan is divided into overlapping 3D patches, each encoded independently, then stitched together. For classification, pooling merges the patch-level features into a single scan-level vector. For segmentation, features are passed to a Swin-UNETR decoder with skip connections, which preserves fine boundary details for voxel-wise segmentation.",[],{"_key":1747,"_type":365,"caption":1748,"image":1749},"b53c3ec1a4f0","Figure 8a-b: 3D scan-level classification (left) and segmentation (right) sliding window Inference",{"_type":89,"asset":1750},{"_ref":1751,"_type":92},"image-9c7d3b91caa94b4ccf1f486106b39d3adff64310-1080x608-gif",{"_key":1753,"_type":289},"9d1f2386f430",{"_key":1755,"_type":242,"text":1756},"373492d8a965",[1757,1765,1773],{"_key":1758,"_type":130,"children":1759,"markDefs":1764,"style":252},"27742a556a73",[1760],{"_key":1761,"_type":134,"marks":1762,"text":1763},"49a15d39b191",[267],"How CoralBay Compares",[],{"_key":1766,"_type":130,"children":1767,"markDefs":1772,"style":138},"d9730d5a19bc",[1768],{"_key":1769,"_type":134,"marks":1770,"text":1771},"5ef80e386a2f",[],"As the CT foundation model space matures, CoralBay can be understood along three axes: how well it transfers to downstream tasks with both frozen and fine-tuned encoders, how little pre-training data it requires, and how broadly it generalises across classification and segmentation tasks.",[],{"_key":1774,"_type":130,"children":1775,"markDefs":1780,"style":138},"edf6185361e2",[1776],{"_key":1777,"_type":134,"marks":1778,"text":1779},"9a9870265e97",[],"To test this, we evaluated CoralBay on 11 datasets spanning classification and segmentation across diverse anatomical targets, with two model variants: CoralBayU96B (53.2M) and CoralBayU96H (847M).",[],{"_key":1782,"_type":242,"text":1783},"546419008d33",[1784],{"_key":1785,"_type":130,"children":1786,"markDefs":1791,"style":252},"454658f66a2e",[1787],{"_key":1788,"_type":134,"marks":1789,"text":1790},"000230fac4c6",[],"Quantitative Results",[],{"_key":1793,"_type":365,"caption":1794,"image":1795},"7a5cce16e0ad","Table A: Quantitative performance across classification (Multi-class Accuracy/Binary AUROC) and segmentation (Dice score) tasks, as evaluated via the eva framework.",{"_type":89,"asset":1796},{"_ref":1797,"_type":92},"image-d3bd456a4f8a0285bf1592543ff451c5d2655910-3881x1864-png",{"_key":1799,"_type":242,"text":1800},"f0202c030fc1",[1801,1813],{"_key":1802,"_type":130,"children":1803,"markDefs":1812,"style":138},"444f6936a4f7",[1804,1808],{"_key":1805,"_type":134,"marks":1806,"text":1807},"0a2fa2c6c444",[267],"- Classification:",{"_key":1809,"_type":134,"marks":1810,"text":1811},"246c3c863c32",[]," CoralBayU96H achieves the best or tied-best frozen-encoder results across all four classification benchmarks — organ identification (OrganMNIST3D), lung nodule malignancy (NoduleMNIST3D, LUNA25), and COVID-19 classification (CC-CCII).",[],{"_key":1814,"_type":130,"children":1815,"markDefs":1824,"style":138},"72429b2441da",[1816,1820],{"_key":1817,"_type":134,"marks":1818,"text":1819},"65e6b282dfa5",[267],"- Segmentation:",{"_key":1821,"_type":134,"marks":1822,"text":1823},"66e3b870c73b",[]," With a frozen encoder and a lightweight 22.8M-parameter decoder, CoralBay performs comparably to VoCo across seven segmentation benchmarks despite the significant data gap.\n\n",[],{"_key":1826,"_type":242,"text":1827},"11e22cc37247",[1828,1836,1844,1856,1868],{"_key":1829,"_type":130,"children":1830,"markDefs":1835,"style":252},"038664d34317",[1831],{"_key":1832,"_type":134,"marks":1833,"text":1834},"a88282b576bf",[],"Qualitative Analysis",[],{"_key":1837,"_type":130,"children":1838,"markDefs":1843,"style":138},"6e4b26c4cad3",[1839],{"_key":1840,"_type":134,"marks":1841,"text":1842},"eb8962ac6119",[],"Visualising the model's performance (Figures 1-4) reveals its precision across diverse anatomical challenges:",[],{"_key":1845,"_type":130,"children":1846,"level":101,"listItem":273,"markDefs":1855,"style":138},"ff3b8ca7829f",[1847,1851],{"_key":1848,"_type":134,"marks":1849,"text":1850},"bb4643edcecb",[267],"Multi-Organ Segmentation:",{"_key":1852,"_type":134,"marks":1853,"text":1854},"9b40d7d71e77",[]," On benchmarks like BTCV and FLARE22, the model accurately delineates complex abdominal structures including the liver, spleen, and kidneys with high spatial consistency.",[],{"_key":1857,"_type":130,"children":1858,"level":101,"listItem":273,"markDefs":1867,"style":138},"24bc2f23c9f5",[1859,1863],{"_key":1860,"_type":134,"marks":1861,"text":1862},"b15a6cd2934f",[267],"Lesion and tumour boundaries:",{"_key":1864,"_type":134,"marks":1865,"text":1866},"b5f90a41d68e",[]," Evaluation on LiTS17 and MSD Task 7 demonstrates its ability to resolve fine-grained tumor boundaries within the liver and pancreas, critical for clinical diagnostics.",[],{"_key":1869,"_type":130,"children":1870,"level":101,"listItem":273,"markDefs":1879,"style":138},"e824eaf2088d",[1871,1875],{"_key":1872,"_type":134,"marks":1873,"text":1874},"7101abc76ded",[267],"Reliability:",{"_key":1876,"_type":134,"marks":1877,"text":1878},"93edb776f4ca",[]," The difference maps highlight minimal variance between CoralBay segmentations and ground-truth annotations, confirming robust feature capture in low-contrast regions.",[],{"_key":1881,"_type":289},"98cc8424d8fb",{"_key":1883,"_type":242,"text":1884},"28f603846357",[1885,1893,1901,1921,1941],{"_key":1886,"_type":130,"children":1887,"markDefs":1892,"style":252},"2958a8aa097b",[1888],{"_key":1889,"_type":134,"marks":1890,"text":1891},"5eb56301d4d0",[267],"What the Ablations Reveal",[],{"_key":1894,"_type":130,"children":1895,"markDefs":1900,"style":138},"624b97cc39b6",[1896],{"_key":1897,"_type":134,"marks":1898,"text":1899},"e79ff7601a16",[],"Ablation studies in the paper confirm that CoralBay’s performance arises from a combination of structural and data-driven innovations:",[],{"_key":1902,"_type":130,"children":1903,"level":101,"listItem":273,"markDefs":1920,"style":138},"9786e5156e10",[1904,1908,1912,1916],{"_key":1905,"_type":134,"marks":1906,"text":1907},"aabe76e00128",[267],"Native 3D Dominance:",{"_key":1909,"_type":134,"marks":1910,"text":1911},"adce4cd9683d",[]," Switching from the native dino 2D to CoralBay 3D spatial modeling provides a substantial improvement of ",{"_key":1913,"_type":134,"marks":1914,"text":1915},"12e0269e2fba",[267],"20%",{"_key":1917,"_type":134,"marks":1918,"text":1919},"3a46a6f7539b",[]," in Dice score (BTCV), proving that consistent 3D inductive biases are essential for medical volumes.",[],{"_key":1922,"_type":130,"children":1923,"level":101,"listItem":273,"markDefs":1940,"style":138},"fe64aed92a06",[1924,1928,1932,1936],{"_key":1925,"_type":134,"marks":1926,"text":1927},"e7647ed0f39d",[267],"Effective Scaling",{"_key":1929,"_type":134,"marks":1930,"text":1931},"195dd081c52b",[],": Segmentation accuracy improves consistently as both the model size and the pre-training dataset grow, resulting in an overall ",{"_key":1933,"_type":134,"marks":1934,"text":1935},"80af9713004a",[267],"15% improvement",{"_key":1937,"_type":134,"marks":1938,"text":1939},"5d5a95f56819",[]," (LiTS17). This demonstrates the framework’s ability to benefit from larger data pools and suggests that training with even more data could further strengthen the model.",[],{"_key":1942,"_type":130,"children":1943,"level":101,"listItem":273,"markDefs":1952,"style":138},"bdce70c68041",[1944,1948],{"_key":1945,"_type":134,"marks":1946,"text":1947},"eefc607d4e17",[267],"Superior Label Efficiency:",{"_key":1949,"_type":134,"marks":1950,"text":1951},"5b3b36cdcf4e",[]," On challenging tumor tasks, self-supervised pre-training acts as a powerful prior, outperforming heavily tuned models in low-data settings.",[],{"_key":1954,"_type":289},"8c07382648bb",{"_key":1956,"_type":242,"text":1957},"fcdb71887c26",[1958,1966,2000,2008,2016,2024,2036,2059,2071],{"_key":1959,"_type":130,"children":1960,"markDefs":1965,"style":252},"6da8f2643ea9",[1961],{"_key":1962,"_type":134,"marks":1963,"text":1964},"509fc72c3ffe",[267],"Why this matters for kaiko",[],{"_key":1967,"_type":130,"children":1968,"markDefs":1999,"style":138},"39467400a5c4",[1969,1972,1976,1979,1983,1987,1991,1995],{"_key":1970,"_type":134,"marks":1971,"text":1496},"c0638d99be1f",[267],{"_key":1973,"_type":134,"marks":1974,"text":1975},"f43458ba626d",[]," establishes a powerful 3D foundation for radiological data. Today, it masters native spatial reasoning and physics-informed intensity invariance, delivering highly data-efficient, clinically robust models for classification and segmentation across organs, pathologies, and scanners. This raises the ceiling for high‑resolution pathology detection while also providing a standard, open benchmark through the 3D radiology leaderboard. \n\nLooking ahead, ",{"_key":1977,"_type":134,"marks":1978,"text":1496},"941e381f5bd5",[267],{"_key":1980,"_type":134,"marks":1981,"text":1982},"4614f3441733",[]," serves as a visual anchor for ",{"_key":1984,"_type":134,"marks":1985,"text":1986},"23837e082ca5",[267],"Multimodal Medical Intelligence",{"_key":1988,"_type":134,"marks":1989,"text":1990},"6b0544eff428",[],", where AI systems jointly interpret imaging, text, lab values, and other clinical signals in a unified clinical context. Built on a strong 3D visual backbone, it moves toward an ",{"_key":1992,"_type":134,"marks":1993,"text":1994},"641b143a9e24",[267],"agentic vision paradigm",{"_key":1996,"_type":134,"marks":1997,"text":1998},"469ed812d54a",[],", where models go beyond interpretation to actively reason over scans, plan multi-step analyses, and coordinate tool use across modalities and time.",[],{"_key":2001,"_type":130,"children":2002,"markDefs":2007,"style":138},"f884f3cd83a7",[2003],{"_key":2004,"_type":134,"marks":2005,"text":2006},"71631a68b7ae",[],"",[],{"_key":2009,"_type":130,"children":2010,"markDefs":2015,"style":252},"f17bcd4148c4",[2011],{"_key":2012,"_type":134,"marks":2013,"text":2014},"3506c3bc6681",[267],"Open Science",[],{"_key":2017,"_type":130,"children":2018,"markDefs":2023,"style":138},"8ad16fdfcb89",[2019],{"_key":2020,"_type":134,"marks":2021,"text":2022},"b2767c4e8ed4",[],"CoralBay promotes open, reproducible evaluation in medical AI through core contributions:",[],{"_key":2025,"_type":130,"children":2026,"level":101,"listItem":273,"markDefs":2035,"style":138},"e8a7a8dcb4c7",[2027,2031],{"_key":2028,"_type":134,"marks":2029,"text":2030},"1aa0531f239e",[267],"Comprehensive Whitepaper:",{"_key":2032,"_type":134,"marks":2033,"text":2034},"ae32bdf52146",[]," We provide a detailed whitepaper describing the model, training process, datasets, evaluation setup, and results.",[],{"_key":2037,"_type":130,"children":2038,"level":101,"listItem":273,"markDefs":2056,"style":138},"f4b10baeca14",[2039,2043,2047,2052],{"_key":2040,"_type":134,"marks":2041,"text":2042},"11e7b8fa0e9e",[267],"Standardized Benchmarking:",{"_key":2044,"_type":134,"marks":2045,"text":2046},"17b681ae0a2b",[]," We’re scaling 🔗 ",{"_key":2048,"_type":134,"marks":2049,"text":2051},"556e5f7ffd40",[267,2050],"ca96badbe275","eva",{"_key":2053,"_type":134,"marks":2054,"text":2055},"14d64945d831",[]," — our open-source evaluation framework — to include comprehensive support for 3D radiology. This includes standardized data loaders, model backbones, and a public leaderboard designed to make 3D medical AI research transparent, reproducible, and easily comparable.",[2057],{"_key":2050,"_type":477,"href":2058},"https://github.com/kaiko-ai/eva",{"_key":2060,"_type":130,"children":2061,"level":101,"listItem":273,"markDefs":2070,"style":138},"9057a5c47442",[2062,2066],{"_key":2063,"_type":134,"marks":2064,"text":2065},"71e126998f48",[267],"Open Access:",{"_key":2067,"_type":134,"marks":2068,"text":2069},"f3b605346cff",[]," Model weights are available on GitHub and Hugging Face.",[],{"_key":2072,"_type":130,"children":2073,"markDefs":2105,"style":138},"402b38cf12a7",[2074,2078,2083,2087,2092,2096,2101],{"_key":2075,"_type":134,"marks":2076,"text":2077},"68c6ea89183c",[],"Explore the 🔗 ",{"_key":2079,"_type":134,"marks":2080,"text":2082},"390e9ca438c4",[2081],"3ca4e837b5ae","paper",{"_key":2084,"_type":134,"marks":2085,"text":2086},"f9ffddb50552",[],", 🔗 ",{"_key":2088,"_type":134,"marks":2089,"text":2091},"3633c31e6832",[2090],"f06242a5f1f7","leaderboard",{"_key":2093,"_type":134,"marks":2094,"text":2095},"aea443f75bfb",[],", and 🔗 ",{"_key":2097,"_type":134,"marks":2098,"text":2100},"44c5f83a91c5",[2099],"bd67fdf5c7ad","model weights",{"_key":2102,"_type":134,"marks":2103,"text":2104},"e77c19555e00",[]," to learn more.",[2106,2108,2110],{"_key":2081,"_type":477,"href":2107},"https://arxiv.org/abs/2606.03888",{"_key":2090,"_type":477,"href":2109},"https://github.com/kaiko-ai/eva#radiology",{"_key":2099,"_type":477,"href":2111},"https://huggingface.co/kaiko-ai/coralbay",{"_key":2113,"_type":289},"906c8f045b1f",{"_key":2115,"_type":242,"text":2116},"7e8094857822",[2117,2126,2133],{"_key":2118,"_type":130,"children":2119,"markDefs":2124,"style":2125},"08d8e875db33",[2120],{"_key":2121,"_type":134,"marks":2122,"text":2123},"d50b6c35f89e",[],"CoralBay is released under the MIT License and is intended solely for research purposes. This model has not been validated or certified for clinical use and must not be used to inform, guide, or replace clinical decision-making, diagnosis, treatment, or patient management.",[],"disclaimer",{"_key":2127,"_type":130,"children":2128,"markDefs":2132,"style":2125},"28407ab80679",[2129],{"_key":2121,"_type":134,"marks":2130,"text":2131},[],"\nCoralBay's capabilities — including tumor segmentation, volume estimation, lesion tracking, and scan-level diagnosis — have been evaluated in research settings only. Performance may not generalise across patient populations, scanner hardware, imaging protocols, or clinical environments. The model has not been reviewed or approved by any regulatory authority (including but not limited to the FDA or under the EU MDR).",[],{"_key":2134,"_type":130,"children":2135,"markDefs":2139,"style":2125},"85d8875c981d",[2136],{"_key":2121,"_type":134,"marks":2137,"text":2138},[],"\nUsers are solely responsible for assessing suitability for their intended use. Kaiko strongly recommends independent validation, ethical review, and compliance with applicable laws before any downstream application — particularly in healthcare AI contexts. CoralBay is provided \"as is\", without warranty of any kind. Kaiko accepts no liability for outcomes arising from its use.",[],"2026-06-10T09:00:00.000Z",{"_type":89,"asset":2142},{"_ref":2143,"_type":92},"image-0327c08822c6c84776553e305bd13275fb6dc913-2011x941-png",{"_type":817,"current":2145},"CoralBay-3D-Foundation-Model","Introducing CoralBay: A fully open-source, native 3D foundation model for radiology that delivers state-of-the-art CT scan analysis with unmatched data efficiency.","CoralBay: The 3D Foundation Model for Radiology CT-Scans by kaiko",[2149,2152,2154],{"_key":2150,"_ref":2151,"_type":92},"ab8ee8167c09","87071b12-f012-4898-9928-e78a41e6f88e",{"_key":2153,"_ref":1454,"_type":92},"8e1606b99a64",{"_key":2155,"_ref":2156,"_type":92},"bdc5cc8677ce","fd4ff9cb-2a4f-41a0-a2af-371ea8baafe4","Every run explains itself",[2159,2162],{"_key":2160,"_ref":2161,"_type":92},"dc9c8fc1e4c9","f006e877-46a9-4d49-9d08-ad783336cf59",{"_key":2163,"_ref":1463,"_type":92},"d7df9b62181d",{"documents":2165,"paths":2169,"mappings":2181},[2166,2167,2168],{"_id":815,"_type":227},{"_id":812,"_type":227},{"_id":222,"_type":227},[145,146,147,148,149,150,2170,2171,2172,2173,2174,2175,2176,2177,2178,2179,2180],"$['area']","$['authors']","$['categories']","$['content']","$['date']","$['featuredImage']","$['relatedPosts']","$['slug']","$['subHeading']","$['title']","$['topics']",{"$['_createdAt']":2182,"$['_id']":2184,"$['_rev']":2186,"$['_system']":2188,"$['_type']":2190,"$['_updatedAt']":2192,"$['area']":2194,"$['authors']":2196,"$['categories']":2198,"$['content']":2200,"$['date']":2202,"$['featuredImage']":2204,"$['relatedPosts']":2206,"$['slug']":2208,"$['subHeading']":2210,"$['suggestedPosts'][0]['categories']":2212,"$['suggestedPosts'][0]['content']":2214,"$['suggestedPosts'][0]['date']":2216,"$['suggestedPosts'][0]['featuredImage']":2218,"$['suggestedPosts'][0]['slug']":2220,"$['suggestedPosts'][0]['subHeading']":2222,"$['suggestedPosts'][0]['title']":2224,"$['suggestedPosts'][0]['topics']":2226,"$['suggestedPosts'][1]['categories']":2228,"$['suggestedPosts'][1]['content']":2230,"$['suggestedPosts'][1]['date']":2232,"$['suggestedPosts'][1]['featuredImage']":2234,"$['suggestedPosts'][1]['slug']":2236,"$['suggestedPosts'][1]['subHeading']":2238,"$['suggestedPosts'][1]['title']":2240,"$['suggestedPosts'][1]['topics']":2242,"$['title']":2244,"$['topics']":2246},{"source":2183,"type":168},{"document":173,"path":166,"type":167},{"source":2185,"type":168},{"document":173,"path":101,"type":167},{"source":2187,"type":168},{"document":173,"path":173,"type":167},{"source":2189,"type":168},{"document":173,"path":176,"type":167},{"source":2191,"type":168},{"document":173,"path":179,"type":167},{"source":2193,"type":168},{"document":173,"path":182,"type":167},{"source":2195,"type":168},{"document":173,"path":185,"type":167},{"source":2197,"type":168},{"document":173,"path":188,"type":167},{"source":2199,"type":168},{"document":173,"path":191,"type":167},{"source":2201,"type":168},{"document":173,"path":194,"type":167},{"source":2203,"type":168},{"document":173,"path":197,"type":167},{"source":2205,"type":168},{"document":173,"path":200,"type":167},{"source":2207,"type":168},{"document":173,"path":203,"type":167},{"source":2209,"type":168},{"document":173,"path":206,"type":167},{"source":2211,"type":168},{"document":173,"path":209,"type":167},{"source":2213,"type":168},{"document":101,"path":191,"type":167},{"source":2215,"type":168},{"document":101,"path":194,"type":167},{"source":2217,"type":168},{"document":101,"path":197,"type":167},{"source":2219,"type":168},{"document":101,"path":200,"type":167},{"source":2221,"type":168},{"document":101,"path":206,"type":167},{"source":2223,"type":168},{"document":101,"path":209,"type":167},{"source":2225,"type":168},{"document":101,"path":212,"type":167},{"source":2227,"type":168},{"document":101,"path":218,"type":167},{"source":2229,"type":168},{"document":166,"path":191,"type":167},{"source":2231,"type":168},{"document":166,"path":194,"type":167},{"source":2233,"type":168},{"document":166,"path":197,"type":167},{"source":2235,"type":168},{"document":166,"path":200,"type":167},{"source":2237,"type":168},{"document":166,"path":206,"type":167},{"source":2239,"type":168},{"document":166,"path":209,"type":167},{"source":2241,"type":168},{"document":166,"path":212,"type":167},{"source":2243,"type":168},{"document":166,"path":218,"type":167},{"source":2245,"type":168},{"document":173,"path":212,"type":167},{"source":2247,"type":168},{"document":173,"path":218,"type":167},1785418824685]