{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_559aebf47dbf4e2c5c65ac4e26382c814d80952a073f6bbd3f4fb5e481d094f2","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_559aebf47dbf4e2c5c65ac4e26382c814d80952a073f6bbd3f4fb5e481d094f2","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"618b04cb16c09e9038c6d259c97c5761052614e3f3b2610ca48142d7f436b229","published":"Tue, 09 Jun 2026 00:00:00 -0400","receipt_hash":"618b04cb16c09e9038c6d259c97c5761052614e3f3b2610ca48142d7f436b229","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"618b04cb16c09e9038c6d259c97c5761052614e3f3b2610ca48142d7f436b229","observed_at":"2026-06-09T04:43:45.619596Z","parent_run_hash":"f2344865fd128464efd1bacba326b5a7ccea707694b8c5650dd51ae8c46ac8a1","published":"Tue, 09 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.09370v3 Announce Type: replace-cross \nAbstract: Large-scale AI training is now fundamentally a distributed systems problem, and hardware failures have become routine operating conditions rather than rare exceptions. Public operational evidence from production training clusters, however, remains scarce. This technical report presents an empirical analysis of a 63-node NVIDIA B200 production cluster (504 GPUs), using 55 days of Prometheus time-series data and 73 days of operational logs covering 224 multi-node training sessions. The cluster operates within a cross-organizational environment in which five parties (SKT, Upstage, Lablup, NVIDIA Korea, and VAST Data) share a unified monitoring pipeline. This arrangement enabled joint diagnosis of a 60-node-scale storage I/O bottleneck that did not appear at 2-4-node scale, a production-scale phenomenon no single team could isolate alone. Drawing on a months-long pre-training campaign, we perform three quantitative analyses yieldin","title":"From Detection to Recovery: Operational Analysis on LLM Pre-training with 504 GPUs","url":"https://arxiv.org/abs/2605.09370","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.09370v3 Announce Type: replace-cross \nAbstract: Large-scale AI training is now fundamentally a distributed systems problem, and hardware failures have become routine operating conditions rather than rare exceptions. Public operational evidence from production training clusters, however, remains scarce. This technical report presents an empirical analysis of a 63-node NVIDIA B200 production cluster (504 GPUs), using 55 days of Prometheus time-series data and 73 days of operational logs covering 224 multi-node training sessions. The cluster operates within a cross-organizational environment in which five parties (SKT, Upstage, Lablup, NVIDIA Korea, and VAST Data) share a unified monitoring pipeline. This arrangement enabled joint diagnosis of a 60-node-scale storage I/O bottleneck that did not appear at 2-4-node scale, a production-scale phenomenon no single team could isolate alone. Drawing on a months-long pre-training campaign, we perform three quantitative analyses yieldin","title":"From Detection to Recovery: Operational Analysis on LLM Pre-training with 504 GPUs","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-09T04:43:45Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.09370"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4c79ace81857646ff3d5f50617bf990aae730af5268460970ba315a2b9811e5eb425f0407d4368090884328d22a751c27289bad99c0468ff8ed6428837a03e0f","signer":"crovia.substrate","subject":{"observed_at":"2026-06-09T04:43:45Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.09370"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8d9a47215a58a3e6349112e87c357ec30ebe123a90392a5f6f843fb735d7a543","leaf_index":224645,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"5845626817900ee5e2e98da520d2ebd3921486428dd6431f333795e2bd0ff796","side":"left"},{"sibling":"22f774ea67650a4ce3f5afd2c72acc563c367c0edf4527ba0fd4c93190a1ab5e","side":"right"},{"sibling":"6ee69bfd2843c5c98f91f3e2c5d408ddfdbef6146b528cb6e201c5f938e7f44e","side":"left"},{"sibling":"c57ab4b95fa23cababf8e7cd05578c5f301aa5c05126d154b40306f0cac791a6","side":"right"},{"sibling":"dc8031487681d9c8a00e59f98bd1caf5e48e7c725e3740c891cbe3628c5c715a","side":"right"},{"sibling":"105f34a4f2ed82d9182cc94407ecc093a2e4250d0134717e6fa385c929721bfe","side":"right"},{"sibling":"4cbc852b1c806f43d24027ee52115a44c6a67dd27328071cc246f77bbd500a57","side":"right"},{"sibling":"61f5edd06f165b7eb528418b7a3a490a1a565a01078352981f9463fd513fc29e","side":"left"},{"sibling":"f54580a307d4bb82e453931aa73e9a0c590486cb8df06af79ad4674f1c4f6963","side":"left"},{"sibling":"b2df6a4bb3e928f0b447931cc688ae01d2415773a2b07cfed0b1cba689078aed","side":"right"},{"sibling":"b1c9ec856caa0fd46bb47b46f18c59ebcd295d774ca17adb3b46f05d394a6a5d","side":"left"},{"sibling":"24fdc29d461691aedb6fa920758206b5bb43851f477ef7a04c34aaed84b8971b","side":"left"},{"sibling":"036922da4e1e2c46d948f070454bfad299b7406fb00735ea9d8bd1e687f5f445","side":"right"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"87c6b850dfec08ac35a693d9db3a3315250a68adb1cfab9b1015f212b63b15bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":224761,"merkle_root":"e9f7b49b652e869ab97ffba9c5a31356b2d0e3dc5d00bb28944adf737c46b1e7","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260609T103805Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-09T14:15:34Z","sig_algorithm":"ed25519","signature":"8ad8076fb12c8e486ae1d1559a9a7ba8e2ee996a9ad3d8ba7bcdbdbd88ab3a15bcb429707aca6d3e9d8b97e2ba755b3dcc77b1abb6601ccb829842719a6fb30d","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_559aebf47dbf4e2c5c65ac4e26382c814d80952a073f6bbd3f4fb5e481d094f2"}}