{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_69068d3c8dee6b7ac9473f90a619438be6ca2c0fca337a5b30136abd517386ce","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_69068d3c8dee6b7ac9473f90a619438be6ca2c0fca337a5b30136abd517386ce","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"9e7b5ea773e92ff6745c19ed044464d9c51bd582f0942bcfa8a5bb7cb6315fb5","published":"Mon, 25 May 2026 00:00:00 -0400","receipt_hash":"9e7b5ea773e92ff6745c19ed044464d9c51bd582f0942bcfa8a5bb7cb6315fb5","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"9e7b5ea773e92ff6745c19ed044464d9c51bd582f0942bcfa8a5bb7cb6315fb5","observed_at":"2026-05-25T04:43:48.841018Z","parent_run_hash":"56713422f06ad87427cd8cdcdb1ed341feb016b33c37198d9b168328d1df15fb","published":"Mon, 25 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.11215v2 Announce Type: replace-cross \nAbstract: Pre-training large language models on massive GPU clusters has made hardware faults routine rather than rare, driving the need for resilient training systems. Yet existing frameworks either focus on specific parallelism schemes or risk drifting away from a failure-free training trajectory. We propose ReCoVer, a resilient LLM pre-training system that upholds a single invariant: each iteration keeps the number of microbatches constant, ensuring per-iteration gradients remain stochastically equivalent to a failure-free run. The framework is organized as three decoupled protocol layers: (1) Fault-tolerant collectives that isolate faults from propagating across replicas; (2) in-step fine-grained recovery that preserves intra-iteration progress and prevents gradient corruption; (3) versatile-workload policy that dynamically redistributes microbatch quotas across the survivors. The design is parallelism-agnostic, integrating directly ","title":"ReCoVer: Resilient LLM Pre-Training System via Fault-Tolerant Collective and Versatile Workload","url":"https://arxiv.org/abs/2605.11215","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.11215v2 Announce Type: replace-cross \nAbstract: Pre-training large language models on massive GPU clusters has made hardware faults routine rather than rare, driving the need for resilient training systems. Yet existing frameworks either focus on specific parallelism schemes or risk drifting away from a failure-free training trajectory. We propose ReCoVer, a resilient LLM pre-training system that upholds a single invariant: each iteration keeps the number of microbatches constant, ensuring per-iteration gradients remain stochastically equivalent to a failure-free run. The framework is organized as three decoupled protocol layers: (1) Fault-tolerant collectives that isolate faults from propagating across replicas; (2) in-step fine-grained recovery that preserves intra-iteration progress and prevents gradient corruption; (3) versatile-workload policy that dynamically redistributes microbatch quotas across the survivors. The design is parallelism-agnostic, integrating directly ","title":"ReCoVer: Resilient LLM Pre-Training System via Fault-Tolerant Collective and Versatile Workload","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-25T04:43:48Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.11215"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:99ce4379e60be7adddc61f8f7cf2286cefaabb9b2c9a1e7818d4bca19c5263045cc471ec2d885534d6f7bd4393b51bdf9c5efe976641ad1af2b69dda68ed0804","signer":"crovia.substrate","subject":{"observed_at":"2026-05-25T04:43:48Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.11215"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"409a439dcb7238a2c664470d5019202acdf714740b0a0175f794da1a4aeba10c","leaf_index":149759,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"045336e0e73d1592aa9d7fade48726abb2bd9f0e8d5dbcf2af91c30604160ed0","side":"left"},{"sibling":"f5dead321938485710e8f755d65e0951a09f84d3d63a353b9c0e7b9c8b2eadef","side":"left"},{"sibling":"ebac63627de8c0ee09e1e2d35f2e9168281159402d75089fbf0247cdf1b48e71","side":"left"},{"sibling":"38b8ae88996a3848062d4ba221afdebd8e1f683f195a333f58540e761ca921fc","side":"left"},{"sibling":"2b3bb63503691b69feadf67374d3b1f5c8c9470823bc18f151aba03919ccd87d","side":"left"},{"sibling":"f4d732126f5433063a27ed518354e49a3c94e9c4c015a328c34b22b871d62002","side":"left"},{"sibling":"9a9dec70c246f07136fd93efad0647d42e469fcfca31c6741dcb1d715bd9ce45","side":"left"},{"sibling":"cb71a04113e54ec457f452a34b2778696e1747afe98ba9da255ce1f5fd8619d7","side":"left"},{"sibling":"e044acc52c31efe134c902c984299ea3452b42df993d096ed61ad8be9d530bbc","side":"right"},{"sibling":"6be461ecf12dedf98a31921aa7b5c32d4a32df6897e0f697b8bf1a4ba3e2d324","side":"right"},{"sibling":"c2861a8cef3eb66bf2726aa377c24a6bf6e8b7489dcd0e870b61d62a35ccadfb","side":"right"},{"sibling":"134949308b15cffd6792ee2cf678119af34d69a64764d7c89cd47573c94e1cda","side":"left"},{"sibling":"debbc1a6232ea7970009b91bdc2345041b2de5ac59551f955855d579001e512c","side":"right"},{"sibling":"216869846f40bd905626884f58cb67b9019e488b3656946a7808c436ab7339ac","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"e734b0c6d6d13acb5f4cf00367b3913e5bbf1aa717377aa4e4820c6de24b7673","side":"right"},{"sibling":"4bbb7f78e96179bb9cdd06d7207b66b3503ab68e4880438263025b04ebe8f7ec","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":149835,"merkle_root":"51484a548bc0cf3d178eec868e98935c87f0aa83369143b7c96b048585e5ba01","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260525T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-25T05:37:33Z","sig_algorithm":"ed25519","signature":"99a657e53a015ace0bade5d7fbb3f60f950b6d1e1128251a63afcc454a492422cae5efdc0c18503b4f7e62f371bd48f2889be78d9048c4681d6c1e73f754d705","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_69068d3c8dee6b7ac9473f90a619438be6ca2c0fca337a5b30136abd517386ce"}}